From 2eb30dbf6212715f48542ddaa7738969f53ea479 Mon Sep 17 00:00:00 2001 From: darmie Date: Fri, 25 Sep 2026 20:51:46 +0100 Subject: [PATCH 1/3] Tiered: each body is HIR-optimised once and every tier compiles it Each lazy function's optimised body lives in a once-cell, filled by the first caller under the scratch lock, and every compile path (requester, broker, optimize_function, record_call, late resume points, closure promotion) reads it through tier_body. A region cut from a body already through the pipeline takes only the finishing passes and the entry cleanup; a region of a lowered body is outlined once per loop. The LLVM tier gets the Arc it is handed, and a reload swaps the cells. git-bug: 73e6e26fd19da79528b2bafffa99d1b4de019d1298246dbfd5afa52cd7621920 --- crates/compiler/src/beadie_adapter.rs | 6 +- crates/compiler/src/lib.rs | 3 + crates/compiler/src/llvm_jit_backend.rs | 24 +- crates/compiler/src/opt_audit.rs | 102 ++++ crates/compiler/src/tiered_backend.rs | 521 +++++++++++++------ crates/compiler/tests/optimized_once.rs | 114 ++++ crates/compiler/tests/optimized_once_race.rs | 87 ++++ 7 files changed, 699 insertions(+), 158 deletions(-) create mode 100644 crates/compiler/src/opt_audit.rs create mode 100644 crates/compiler/tests/optimized_once.rs create mode 100644 crates/compiler/tests/optimized_once_race.rs diff --git a/crates/compiler/src/beadie_adapter.rs b/crates/compiler/src/beadie_adapter.rs index 52bba0e5..7f4da7ea 100644 --- a/crates/compiler/src/beadie_adapter.rs +++ b/crates/compiler/src/beadie_adapter.rs @@ -32,7 +32,9 @@ use crate::hir::{HirFunction, HirId, HirModule}; #[derive(Clone)] pub struct ZyntaxFunctionDef { pub id: HirId, - pub function: HirFunction, + /// Shared with whatever handed it over: a tier compiles the body it + /// is given, not a copy made on the way. + pub function: std::sync::Arc, /// Module-level effect, handler, global, and callee context required when /// a single hot function is recompiled outside the initial bulk pass. pub module: std::sync::Arc, @@ -426,6 +428,7 @@ mod llvm_impl { if sites.is_empty() { return Vec::new(); } + crate::opt_audit::note_llvm_body(def.id, &def.function); self.with_lock(|backend| { backend.set_compile_tier(def.tier); backend.set_module_context(std::sync::Arc::clone(&def.module)); @@ -470,6 +473,7 @@ mod llvm_impl { // Resume points where a frame can take one: the sites frames // asked at, and an outlined region's own header. let sites = crate::osr::wanted_resume_points(def.bead_id, &def.function); + crate::opt_audit::note_llvm_body(def.id, &def.function); self.with_lock(|backend| { backend.set_compile_tier(tier); backend.set_module_context(std::sync::Arc::clone(&def.module)); diff --git a/crates/compiler/src/lib.rs b/crates/compiler/src/lib.rs index 5f28f7a5..d7c93fe8 100644 --- a/crates/compiler/src/lib.rs +++ b/crates/compiler/src/lib.rs @@ -72,6 +72,8 @@ pub mod memory_optimization; // Memory-aware optimizations pub mod memory_pass; pub mod monomorphize; pub mod move_insert; // Owning parameters become `Move` the borrow check can see +#[doc(hidden)] +pub mod opt_audit; // Per-function counts of optimiser runs, for tests pub mod optimization; pub mod parallel_dispatch; // A loop with independent iterations becomes a band dispatch pub mod parallel_safe; // Which counted loops have independent iterations @@ -1970,6 +1972,7 @@ fn run_interp_safe_opts_with( { return stats; } + opt_audit::note_pipeline(module); // Alloca → Malloc promotion runs ONCE up front, before the // fixed-point sweep. Two reasons it goes here: diff --git a/crates/compiler/src/llvm_jit_backend.rs b/crates/compiler/src/llvm_jit_backend.rs index be5d8044..e58f1379 100644 --- a/crates/compiler/src/llvm_jit_backend.rs +++ b/crates/compiler/src/llvm_jit_backend.rs @@ -107,6 +107,12 @@ pub struct LLVMJitBackend<'ctx> { /// [`Self::set_cross_tier_links`]. cross_tier_key: Option, global_resolver: Option Option + Send + Sync>>, + /// The body each tier compiles of a module function, when it is not + /// the module context's own; a callee compiled alongside a promoted + /// function takes it. See [`Self::set_body_source`]. + #[allow(clippy::type_complexity)] + body_source: + Option Option> + Send + Sync>>, /// What the module being compiled reaches across tiers, handed to /// the lowering. pending_cross_tier: HashMap, @@ -207,6 +213,7 @@ impl<'ctx> LLVMJitBackend<'ctx> { address_taken: std::collections::HashSet::new(), cross_tier_key: None, global_resolver: None, + body_source: None, pending_cross_tier: HashMap::new(), pending_shared_globals: HashMap::new(), pending_entry_abi: None, @@ -1084,6 +1091,17 @@ impl<'ctx> LLVMJitBackend<'ctx> { self.global_resolver = Some(globals); } + /// Where a callee compiled alongside a promoted function takes its + /// body from: `bodies` answers with the body every tier compiles of + /// a function, or `None` for one whose module-context body is that. + #[allow(clippy::type_complexity)] + pub fn set_body_source( + &mut self, + bodies: std::sync::Arc Option> + Send + Sync>, + ) { + self.body_source = Some(bodies); + } + /// Compile a single function, together with everything it calls. /// /// The callees come from the module context, so they are recompiled at @@ -1169,7 +1187,11 @@ impl<'ctx> LLVMJitBackend<'ctx> { if callee == id { continue; } - if let Some(f) = ctx.functions.get(&callee) { + let tier_body = self.body_source.as_ref().and_then(|body| body(callee)); + if let Some(f) = tier_body { + crate::opt_audit::note_llvm_body(callee, &f); + functions.insert(callee, (*f).clone()); + } else if let Some(f) = ctx.functions.get(&callee) { functions.insert(callee, f.clone()); } } diff --git a/crates/compiler/src/opt_audit.rs b/crates/compiler/src/opt_audit.rs new file mode 100644 index 00000000..6aca9276 --- /dev/null +++ b/crates/compiler/src/opt_audit.rs @@ -0,0 +1,102 @@ +//! Per-function counts of the optimiser's work, for tests that hold +//! each body to one run of the pipeline and every compile above the +//! baseline to that body. Off until [`enable`]; while off, each hook +//! costs one relaxed load. + +use std::collections::HashMap; +use std::sync::atomic::{AtomicBool, Ordering}; +use std::sync::{Arc, Mutex, OnceLock}; + +use crate::hir::{HirFunction, HirId, HirModule}; + +static ON: AtomicBool = AtomicBool::new(false); + +#[derive(Default)] +struct Counts { + pipeline: HashMap, + finishing: HashMap, + llvm: HashMap>, +} + +fn counts() -> &'static Mutex { + static C: OnceLock> = OnceLock::new(); + C.get_or_init(|| Mutex::new(Counts::default())) +} + +/// Start counting, for the rest of the process. +pub fn enable() { + ON.store(true, Ordering::Relaxed); +} + +fn on() -> bool { + ON.load(Ordering::Relaxed) +} + +/// A run of the whole pipeline over `module`: one for each function it +/// optimises, those not through it already. +pub(crate) fn note_pipeline(module: &HirModule) { + if !on() { + return; + } + let mut c = counts().lock().unwrap(); + for (id, f) in &module.functions { + if !f.attributes.optimized && !f.is_external { + *c.pipeline.entry(*id).or_default() += 1; + } + } +} + +/// The incremental passes run over `id`, a body that arrived optimised. +pub(crate) fn note_finishing(id: HirId) { + if on() { + *counts().lock().unwrap().finishing.entry(id).or_default() += 1; + } +} + +/// `body` handed to the LLVM tier as the HIR of `id`. +#[cfg_attr(not(feature = "llvm-backend"), allow(dead_code))] +pub(crate) fn note_llvm_body(id: HirId, body: &Arc) { + if on() { + counts() + .lock() + .unwrap() + .llvm + .entry(id) + .or_default() + .push(Arc::as_ptr(body) as usize); + } +} + +/// How many runs of the whole pipeline optimised `id`. +pub fn pipeline_runs(id: HirId) -> usize { + counts() + .lock() + .unwrap() + .pipeline + .get(&id) + .copied() + .unwrap_or(0) +} + +/// How many times the incremental passes ran over `id`. +pub fn finishing_runs(id: HirId) -> usize { + counts() + .lock() + .unwrap() + .finishing + .get(&id) + .copied() + .unwrap_or(0) +} + +/// The address of each body the LLVM tier was handed for `id`, in the +/// order it was handed them. +pub fn llvm_bodies(id: HirId) -> Vec { + counts() + .lock() + .unwrap() + .llvm + .get(&id) + .cloned() + .unwrap_or_default() +} diff --git a/crates/compiler/src/tiered_backend.rs b/crates/compiler/src/tiered_backend.rs index 6eed5572..2a772854 100644 --- a/crates/compiler/src/tiered_backend.rs +++ b/crates/compiler/src/tiered_backend.rs @@ -241,6 +241,82 @@ struct FunctionEntry { bead_id: u64, } +/// A function's optimised body; see [`OptimizedBodies`]. +type BodyCell = Arc>>; + +/// Each lazy function's optimised body, beside the lowered one its +/// module holds. A cell is filled at most once per load or reload, by +/// the first caller to reach it while any other waits, and every compile +/// of the function at any tier reads its HIR from its cell or from the +/// body a reload swapped in. A cell is replaced whole, by a reload or to +/// let its body go, and never refilled in place. +#[derive(Default)] +struct OptimizedBodies(RwLock>); + +impl OptimizedBodies { + fn cell(&self, id: HirId) -> Option { + self.0.read().unwrap().get(&id).cloned() + } + + /// The body in `id`'s cell, if it is filled. + fn get(&self, id: HirId) -> Option> { + self.0.read().unwrap().get(&id)?.get().cloned() + } + + /// An empty cell for each of `ids` that has none; a cell already + /// there is kept, filled or not. + fn add(&self, ids: impl IntoIterator) { + let mut cells = self.0.write().unwrap(); + for id in ids { + cells.entry(id).or_default(); + } + } + + /// `cell` in place of `id`'s, handing back the one it replaced. + fn replace(&self, id: HirId, cell: BodyCell) -> Option { + self.0.write().unwrap().insert(id, cell) + } + + /// Put back what [`Self::replace`] handed back. + fn restore(&self, id: HirId, cell: Option) { + let mut cells = self.0.write().unwrap(); + match cell { + Some(cell) => cells.insert(id, cell), + None => cells.remove(&id), + }; + } + + fn filled(body: Arc) -> BodyCell { + let cell = std::sync::OnceLock::new(); + let _ = cell.set(body); + Arc::new(cell) + } +} + +/// The HIR every tier compiles of `func_id`: the body a reload swapped +/// in, else its optimised body, made now if a lazy function has none +/// yet, else the module's, which an eager function was optimised in at +/// load. +fn tier_body( + func_id: HirId, + bead_id: u64, + swapped: Option<&Arc>, + bodies: &OptimizedBodies, + lazy: bool, + module: &HirModule, +) -> Option> { + if let Some(f) = swapped { + return Some(Arc::clone(f)); + } + if let Some(f) = bodies.get(func_id) { + return Some(f); + } + if lazy && let Some(f) = osr::lazy_optimized_body(bead_id) { + return Some(f); + } + module.functions.get(&func_id).map(|f| Arc::new(f.clone())) +} + /// What the tier above answered a bead's promotion request with, kept /// per bead and tier so the answer is not worked out again. #[derive(Clone, Copy)] @@ -598,6 +674,11 @@ struct UndoSwap { /// nothing was published). old_published: usize, old_body: Arc, + /// The body a reload before this one swapped in, if one had. + old_function: Option>, + /// The function's optimised-body cell before the reload, if it had + /// one. + old_body_cell: Option, /// OSR resume points this reload published, to be unpublished. helper_sites: Vec<(u64, u64)>, /// Dispatch-table slots this reload patched: (slot address, value @@ -642,10 +723,9 @@ pub struct TieredBackend { /// in their cell; the interpreter reaches their baseline the same /// way, so one compile is made of each. lazy: HashSet, - /// The body the first-call compile optimised for each function it - /// compiled, which a later tier compiles again rather than the - /// module's unoptimised one. - optimized_bodies: Arc>>>, + /// Each lazy function's optimised body, made once and compiled by + /// every tier in place of the module's unoptimised one. + optimized_bodies: Arc, /// Of the lazy functions, those that arrived optimised (a linked /// snapshot's), across every module compiled so far: the interpreter /// takes their body from the optimiser, which has little to do to @@ -657,9 +737,11 @@ pub struct TieredBackend { /// code, and asks for the optimizing tier as soon as it has some. static_hot: HashMap, /// The optimiser for a function that is not the module's: an - /// outlined resume point, optimised in the scratch module of the - /// bead it belongs to. Installed with the lazy compiler. - optimize_extra: Option HirFunction + Send + Sync>>, + /// outlined resume point of the bead it belongs to, and whether it + /// was cut from a body through the pipeline already, which it then + /// does not go through again. Installed with the lazy compiler. + #[allow(clippy::type_complexity)] + optimize_extra: Option HirFunction + Send + Sync>>, /// The body the interpreter runs a lazy function's first calls on: /// the module's, as lowered, with its releases placed (see /// [`Self::interpreter_body_source`]). Made at the first run and @@ -793,7 +875,7 @@ impl TieredBackend { _llvm_context, functions: HashMap::new(), lazy: HashSet::new(), - optimized_bodies: Arc::new(Mutex::new(HashMap::new())), + optimized_bodies: Arc::new(OptimizedBodies::default()), finished: HashSet::new(), static_hot: HashMap::new(), optimize_extra: None, @@ -1040,6 +1122,25 @@ impl TieredBackend { let (all_lazy, all_finished) = (self.lazy.clone(), self.finished.clone()); self.install_lazy_compiler(&all_lazy, &all_finished); } + // A promotion that compiles its callee closure takes each + // callee's tier body. + #[cfg(feature = "llvm-backend")] + if let Some(llvm) = &self.llvm { + let bodies = Arc::clone(&self.optimized_bodies); + let beads: HashMap = self + .functions + .iter() + .filter(|(id, _)| self.lazy.contains(id)) + .map(|(id, e)| (*id, e.bead_id)) + .collect(); + llvm.with_lock(|be| { + be.set_body_source(Arc::new(move |id| { + bodies + .get(id) + .or_else(|| beads.get(&id).and_then(|b| osr::lazy_optimized_body(*b))) + })) + }); + } if !joining { self.install_promotion_requester(); self.precompile_static_hot(&lazy); @@ -1519,6 +1620,8 @@ impl TieredBackend { let mut old_entry = 0usize; let mut old_body: Option> = None; + let mut old_function: Option> = None; + let mut old_body_cell: Option = None; let old_cell = crate::reload::call_target(reload_key, old_id); let old_published = osr::published_entry(bead_id); if let Some(fn_entry) = self.functions.get_mut(&old_id) { @@ -1540,7 +1643,14 @@ impl TieredBackend { { fn_entry.bound.bead().eager_install(entry_ptr as *mut ()); } - fn_entry.function = Some(Arc::new(body.clone())); + // The edit's body is what every tier compiles from now + // on, the interpreter's getter included, which reads the + // cell. + let swapped = Arc::new(body.clone()); + old_function = fn_entry.function.replace(Arc::clone(&swapped)); + old_body_cell = self + .optimized_bodies + .replace(old_id, OptimizedBodies::filled(swapped)); } // A stub whose address was kept calls the edit's code too. osr::set_published_entry(bead_id, entry_ptr); @@ -1623,6 +1733,8 @@ impl TieredBackend { old_cell, old_published, old_body: old_body.unwrap_or_else(|| Arc::new(old_fn.clone())), + old_function, + old_body_cell, helper_sites, vtable_slots, }); @@ -1769,7 +1881,8 @@ impl TieredBackend { // restored body. fn_entry.bound.bead().reload(); } - fn_entry.function = Some(Arc::clone(&swap.old_body)); + fn_entry.function = swap.old_function.clone(); + self.optimized_bodies.restore(swap.id, swap.old_body_cell); } let cell = if swap.old_cell != 0 { swap.old_cell @@ -2401,24 +2514,23 @@ impl TieredBackend { let code = adapter.on_invoke(&bound, move |tier, bead| { let c = &*ctx; // The baseline was compiled at load; the ladder above it - // compiles anew, from the body the first-call compile - // optimised when there is one. A quick baseline has none - // until its replacement: the body is optimised now. + // compiles anew, from the function's tier body. A quick + // baseline has none until its replacement: the body is + // optimised now, once, for both. if tier == 0 { if let Some(p) = c.cranelift.with_lock(|be| be.get_function_ptr(c.func_id)) { return p as *mut (); } } - let optimized = optimized_bodies.lock().unwrap().get(&c.func_id).cloned(); - let optimized = optimized - .or_else(|| lazy.then(|| osr::lazy_optimized_body(c.bead_id)).flatten()); - let body = match (&c.swapped, optimized) { - (Some(f), _) => Arc::clone(f), - (None, Some(f)) => f, - (None, None) => match c.module.functions.get(&c.func_id) { - Some(f) => Arc::new(f.clone()), - None => return ptr::null_mut(), - }, + let Some(body) = tier_body( + c.func_id, + c.bead_id, + c.swapped.as_ref(), + &optimized_bodies, + lazy, + &c.module, + ) else { + return ptr::null_mut(); }; let entry = compile_at_tier( tier, @@ -2468,8 +2580,8 @@ impl TieredBackend { let finished = self.finished.clone(); Box::new(move |id: HirId| { let bead = beads.get(&id)?; - if let Some(body) = optimized_bodies.lock().unwrap().get(&id) { - return Some(Arc::clone(body)); + if let Some(body) = optimized_bodies.get(id) { + return Some(body); } if finished.contains(&id) { return osr::lazy_optimized_body(*bead); @@ -2567,8 +2679,11 @@ impl TieredBackend { None => return, }; - // Build a closure beadie can call from any tier broker thread. - let func_arc = entry.body(func_id); + // Build a closure beadie can call from any tier broker thread. The + // body is looked up when a compile needs it, not per call. + let swapped = entry.function.clone(); + let bodies = Arc::clone(&self.optimized_bodies); + let lazy = self.lazy.contains(&func_id); let module_arc = Arc::clone(&entry.module); let bead_id = entry.bead_id; let cranelift = Arc::clone(&self.cranelift); @@ -2578,6 +2693,16 @@ impl TieredBackend { let verbosity = self.config.verbosity; let closure = move |tier_idx: usize, bead: &Arc| -> *mut () { + let Some(func_arc) = tier_body( + func_id, + bead_id, + swapped.as_ref(), + &bodies, + lazy, + &module_arc, + ) else { + return ptr::null_mut(); + }; compile_at_tier( tier_idx, bead, @@ -2653,6 +2778,8 @@ impl TieredBackend { struct Scratch { module: HirModule, cache: crate::OptCache, + /// The functions of `module` through the pipeline in it. + done: HashSet, } // One scratch per module: a program that compiles a chunk of // itself has bodies in more than one. @@ -2660,18 +2787,30 @@ impl TieredBackend { /// The scratch of `module` in `slots`, made if none is: every /// function marked through the pipeline and deferred, so a /// pass that walks the module touches the one being optimised - /// alone. + /// alone. A program function whose cell is filled goes in as + /// its cell holds it, so a body optimised later inlines it as + /// every tier runs it; one that arrived optimised stays as it + /// arrived, box readers and all, as the release pass reads it. fn of<'a>( slots: &'a mut HashMap, module: &Arc, facts: &Facts, + bodies: &OptimizedBodies, + program: &HashSet, ) -> &'a mut Scratch { let key = Arc::as_ptr(module) as usize; slots.entry(key).or_insert_with(|| { let started = web_time::Instant::now(); let facts = facts.of(module); let mut module: HirModule = (**module).clone(); - for f in module.functions.values_mut() { + let mut done = HashSet::new(); + for (id, f) in module.functions.iter_mut() { + if program.contains(id) + && let Some(body) = bodies.get(*id) + { + *f = (*body).clone(); + done.insert(*id); + } f.attributes.optimized = true; f.attributes.deferred = true; } @@ -2683,7 +2822,11 @@ impl TieredBackend { std::thread::current().name().unwrap_or("?") ); } - Scratch { module, cache } + Scratch { + module, + cache, + done, + } }) } } @@ -2719,11 +2862,30 @@ impl TieredBackend { let keeps_bodies = matches!(tier2_backend, Tier2Backend::LLVM); #[cfg(not(feature = "llvm-backend"))] let keeps_bodies = false; + // What a body that arrived optimised (a linked snapshot's) still + // needs, the module's own pass having skipped it: box readers to + // loads, then what those loads let move. Incremental; never the + // pipeline. + let finish = { + let externs = externs.clone(); + let pure_fns = pure_fns.clone(); + move |f: &mut HirFunction| { + crate::opt_audit::note_finishing(f.id); + let boxed = crate::boxes::run_function(f, &externs); + if boxed.expanded + boxed.made + boxed.released + boxed.shared > 0 { + crate::licm::run(f); + crate::cse::eliminate_with(f, &pure_fns); + } + } + }; + let finish: Arc = Arc::new(finish); // The body of a lazy function as every tier runs it: the // interpreter, the baseline and the tier above compile one body, - // so a frame in any of them can move to the next. Made once, at - // the first call or the first compile, whichever comes first, - // and kept in `optimized_bodies` from then. + // so a frame in any of them can move to the next. Made in the + // function's cell by the first call or compile to ask, whichever + // comes first; any other asking meanwhile waits for it. + self.optimized_bodies + .add(by_bead.values().map(|(id, _, _)| *id)); let optimize_body = { let by_bead: HashMap)> = by_bead .iter() @@ -2734,68 +2896,72 @@ impl TieredBackend { let facts = Arc::clone(&facts); let finished = finished.clone(); let lazy = lazy.clone(); - let externs = externs.clone(); - let pure_fns = pure_fns.clone(); + let finish = Arc::clone(&finish); move |bead_id: u64| -> Option> { let (func_id, module_arc) = by_bead.get(&bead_id)?; - if let Some(body) = optimized_bodies.lock().unwrap().get(func_id) { + let cell = optimized_bodies.cell(*func_id)?; + if let Some(body) = cell.get() { return Some(Arc::clone(body)); } - let body = if finished.contains(func_id) { - // Optimised with its snapshot; what the module's own - // pass would have done to it, done to it alone: box - // readers to loads, then what those loads let move. - let mut f = module_arc.functions.get(func_id)?.clone(); - f.attributes.deferred = false; - let boxed = crate::boxes::run_function(&mut f, &externs); - if boxed.expanded + boxed.made + boxed.released + boxed.shared > 0 { - crate::licm::run(&mut f); - crate::cse::eliminate_with(&mut f, &pure_fns); - } - Arc::new(f) - } else { - if !lazy.contains(func_id) { - return None; + let finished = finished.contains(func_id); + if !module_arc.functions.contains_key(func_id) + || (!finished && !lazy.contains(func_id)) + { + return None; + } + let body = cell.get_or_init(|| { + if finished { + let mut f = module_arc.functions[func_id].clone(); + f.attributes.deferred = false; + finish(&mut f); + return Arc::new(f); } + // The scratch is held until the body is in the cell's + // hands, so no other caller sees it half made. let mut optimized = optimized.lock().unwrap(); - let scratch = Scratch::of(&mut optimized, module_arc, &facts); - let f = scratch.module.functions.get_mut(func_id)?; - f.attributes.optimized = false; - f.attributes.deferred = false; - // `ZYNTAX_DUMP_HIR_DIR` gets the body before and after - // its own optimisation, as `fn--lowered.hir` and - // `fn--body.hir`. - let name = f.name.resolve_global().unwrap_or_default(); - if std::env::var_os("ZYNTAX_DUMP_HIR_DIR").is_some() { - let lowered = f.clone(); - crate::hir_dump::dump_function_to_dir( - &lowered, - &scratch.module, - &format!("{name}-lowered"), - ); + let scratch = + Scratch::of(&mut optimized, module_arc, &facts, &optimized_bodies, &lazy); + if !scratch.done.contains(func_id) { + let f = scratch + .module + .functions + .get_mut(func_id) + .expect("the scratch is a copy of the function's module"); + f.attributes.optimized = false; + f.attributes.deferred = false; + // `ZYNTAX_DUMP_HIR_DIR` gets the body before and + // after its own optimisation, as + // `fn--lowered.hir` and `fn--body.hir`. + if std::env::var_os("ZYNTAX_DUMP_HIR_DIR").is_some() { + let name = f.name.resolve_global().unwrap_or_default(); + let lowered = f.clone(); + crate::hir_dump::dump_function_to_dir( + &lowered, + &scratch.module, + &format!("{name}-lowered"), + ); + } + crate::run_interp_safe_opts_cached(&mut scratch.module, &scratch.cache); + scratch.done.insert(*func_id); } - crate::run_interp_safe_opts_cached(&mut scratch.module, &scratch.cache); - let f = scratch.module.functions.get_mut(func_id)?; + let f = scratch + .module + .functions + .get_mut(func_id) + .expect("no pass removes a function"); f.attributes.optimized = true; f.attributes.deferred = true; let mut body = f.clone(); body.attributes.deferred = false; + let name = body.name.resolve_global().unwrap_or_default(); crate::hir_dump::dump_function_to_dir( &body, &scratch.module, &format!("{name}-body"), ); Arc::new(body) - }; - // Another thread may have made it meanwhile; the first - // one in stays, so every tier reads the same body. - Some(Arc::clone( - optimized_bodies - .lock() - .unwrap() - .entry(*func_id) - .or_insert(body), - )) + }); + Some(Arc::clone(body)) } }; let optimize_body: Arc Option> + Send + Sync> = @@ -2806,8 +2972,12 @@ impl TieredBackend { move |bead_id| optimize_body(bead_id) }); // A function of the module's making that is not in it, an - // outlined resume point: optimised in the same scratch, alongside - // the bodies it may inline, and taken out again. + // outlined resume point. Cut from a body already through the + // pipeline, it takes what the cut calls for alone: the finishing + // passes a body that arrived optimised takes, and the cleanup of + // the entry the cut made. Cut from a lowered body, it is + // optimised in the same scratch, alongside the bodies it may + // inline, and taken out again. self.optimize_extra = Some(Arc::new({ let by_bead: HashMap> = by_bead .iter() @@ -2815,12 +2985,24 @@ impl TieredBackend { .collect(); let optimized = Arc::clone(&optimized); let facts = Arc::clone(&facts); - move |bead_id: u64, f: HirFunction| -> HirFunction { + let optimized_bodies = Arc::clone(&optimized_bodies); + let finish = Arc::clone(&finish); + let lazy = lazy.clone(); + move |bead_id: u64, mut f: HirFunction, already_optimized: bool| -> HirFunction { + if already_optimized { + finish(&mut f); + crate::cfg_simplify::run(&mut f); + crate::phi_prune::run_function(&mut f); + f.attributes.optimized = false; + f.attributes.deferred = false; + return f; + } let Some(module_arc) = by_bead.get(&bead_id) else { return f; }; let mut optimized = optimized.lock().unwrap(); - let scratch = Scratch::of(&mut optimized, module_arc, &facts); + let scratch = + Scratch::of(&mut optimized, module_arc, &facts, &optimized_bodies, &lazy); let id = f.id; scratch.module.functions.insert(id, f); crate::run_interp_safe_opts_cached(&mut scratch.module, &scratch.cache); @@ -2972,7 +3154,7 @@ impl TieredBackend { } } let lazy_started = web_time::Instant::now(); - let optimized = optimized_bodies.lock().unwrap().contains_key(func_id); + let optimized = optimized_bodies.get(*func_id).is_some(); if quick && !optimized && let Some(queue) = &queue @@ -3087,9 +3269,10 @@ impl TieredBackend { } // A body with a loop is kept for the resume points an // interpreted frame of it may still ask for; without one, or - // a tier above, the compile was its last reader. + // a tier above, the compile was its last reader, and its + // cell is let go for an empty one. if !keeps_bodies && osr::find_loop_headers(&body).is_empty() { - optimized_bodies.lock().unwrap().remove(func_id); + optimized_bodies.replace(*func_id, BodyCell::default()); } publish(entry as usize, false); // A promotion held for the quick baseline goes ahead now @@ -3338,6 +3521,17 @@ impl TieredBackend { ) }) .collect(); + // Of those, the functions that arrived optimised. + let finished_beads: HashSet = self + .functions + .iter() + .filter(|(id, _)| self.finished.contains(id)) + .map(|(_, e)| e.bead_id) + .collect(); + // The loops a region was outlined at, by (bead, body tag, loop + // ordinal): each is outlined, and optimised, once. + let outlined_at: Arc>> = + Arc::new(Mutex::new(HashSet::new())); // After every install at the optimizing tier, whichever path // made it, the sites frames asked at meanwhile get its resume @@ -3411,9 +3605,7 @@ impl TieredBackend { if let Some(llvm) = &llvm && late.lock().unwrap().claim_if_installed(bead_id, site) { - let body = swapped - .clone() - .or_else(|| optimized_bodies.lock().unwrap().get(func_id).cloned()); + let body = swapped.clone().or_else(|| optimized_bodies.get(*func_id)); let bodies = late_resume_bodies( body, module_arc.functions.get(func_id), @@ -3475,19 +3667,19 @@ impl TieredBackend { return true; } } - // The body: the one a reload swapped in, else the one the - // first-call compile optimised, else the module's. - let optimized = optimized_bodies.lock().unwrap().get(func_id).cloned(); - let func_arc = match (swapped, optimized) { - (Some(f), _) => Arc::clone(f), - (None, Some(f)) => f, - (None, None) => match module_arc.functions.get(func_id) { - Some(f) => Arc::new(f.clone()), - None => return false, - }, - }; let module_arc = Arc::clone(module_arc); let func_id = *func_id; + // The body every tier compiles, made now if it was not yet. + let body_now = || { + tier_body( + func_id, + bead_id, + swapped.as_ref(), + &optimized_bodies, + *lazy, + &module_arc, + ) + }; // An interpreted frame can enter no code mid-loop without a // resume point, and the baseline's body has only probes. Its // resume points come first, before the body: the frame is @@ -3498,21 +3690,23 @@ impl TieredBackend { if let osr::Requester::Interpreted { site } = from { let (body_tag, ordinal, _) = osr::decode_osr_site(site); // The frame runs the body its tag names: the one the - // interpreter made for the first runs, or the module's - // as lowered, when it started before any optimised body - // existed; else that one. - let unoptimized = interp_bodies + // interpreter made for the first runs, or the module's, + // when it started before any optimised body existed; else + // the tier body. The module's is through the pipeline + // already for a function that arrived optimised. + let untiered = interp_bodies .lock() .unwrap() .get(&func_id) .cloned() .filter(|f| osr::body_tag(f) == body_tag) + .map(|f| (f, false)) .or_else(|| { module_arc .functions .get(&func_id) .filter(|f| osr::body_tag(f) == body_tag) - .map(|f| Arc::new(f.clone())) + .map(|f| (Arc::new(f.clone()), finished_beads.contains(&bead_id))) }); // `ZYNTAX_DISABLE_OUTLINE=1` gives such a frame the // baseline's resume point instead, to bisect; safe to @@ -3520,24 +3714,32 @@ impl TieredBackend { let outline = optimize_extra .as_ref() .filter(|_| std::env::var_os("ZYNTAX_DISABLE_OUTLINE").is_none()); - match (unoptimized, outline) { - // A frame on an unoptimised body leaves into the - // region optimised on its own: the body itself was - // never optimised, and need not be for this call. - (Some(body), Some(optimize)) => { - let optimize = Arc::clone(optimize); - publish_outlined_resume_point( - &cranelift, - func_id, - bead_id, - &body, - &module_arc, - ordinal, - &*optimize, - &outlined_regions, - ); + match (untiered, outline) { + // A frame on a body other than the tier body leaves + // into the region optimised on its own: the body + // itself need not be for this call. A loop another + // request outlined already has its resume point, or + // is getting it. + (Some((body, optimized)), Some(optimize)) => { + let key = (bead_id, body_tag, ordinal); + if outlined_at.lock().unwrap().insert(key) { + let optimize = Arc::clone(optimize); + let attempted = publish_outlined_resume_point( + &cranelift, + func_id, + bead_id, + &body, + &module_arc, + ordinal, + &|f| optimize(bead_id, f, optimized), + &outlined_regions, + ); + if !attempted { + outlined_at.lock().unwrap().remove(&key); + } + } } - (Some(body), None) => { + (Some((body, _)), None) => { publish_baseline_resume_points( &cranelift, func_id, @@ -3547,16 +3749,25 @@ impl TieredBackend { ordinal, ); } - (None, _) => publish_baseline_resume_points( - &cranelift, - func_id, - bead_id, - &func_arc, - &module_arc, - ordinal, - ), + // A frame on neither runs the tier body, which is + // made by now. + (None, _) => { + if let Some(body) = body_now() { + publish_baseline_resume_points( + &cranelift, + func_id, + bead_id, + &body, + &module_arc, + ordinal, + ); + } + } } } + let Some(func_arc) = body_now() else { + return false; + }; // The baseline, so the promotion has something to promote and // the next call has code. if !ensure_baseline( @@ -3591,18 +3802,6 @@ impl TieredBackend { } return true; } - // The tier above compiles the body the baseline was optimised - // from, which a first-call compile just made when the request - // came from the interpreter. - let func_arc = match swapped { - Some(_) => func_arc, - None => optimized_bodies - .lock() - .unwrap() - .get(&func_id) - .cloned() - .unwrap_or(func_arc), - }; #[cfg(feature = "llvm-backend")] if !crate::abi::llvm_entry_abi_supported(&func_arc, false) { if osr::osr_trace_enabled() { @@ -3733,16 +3932,24 @@ impl TieredBackend { .get(&func_id) .ok_or_else(|| CompilerError::Backend(format!("Function {:?} not found", func_id)))?; - let func_arc = entry.body(func_id); - let module_arc = Arc::clone(&entry.module); + let lazy = self.lazy.contains(&func_id); let bead_id = entry.bead_id; + let func_arc = tier_body( + func_id, + bead_id, + entry.function.as_ref(), + &self.optimized_bodies, + lazy, + &entry.module, + ) + .ok_or_else(|| CompilerError::Backend(format!("Function {:?} has no body", func_id)))?; + let module_arc = Arc::clone(&entry.module); let cranelift = Arc::clone(&self.cranelift); #[cfg(feature = "llvm-backend")] let llvm = self.llvm.as_ref().map(Arc::clone); let tier2_backend = self.config.tier2_backend; let verbosity = self.config.verbosity; let tier_idx = target_tier.index(); - let lazy = self.lazy.contains(&func_id); if !ensure_baseline( &entry.bound, func_id, @@ -4166,7 +4373,7 @@ fn publish_late_resume_point( }; let def = ZyntaxFunctionDef { id: *id, - function: (**body).clone(), + function: Arc::clone(body), module: Arc::clone(module_arc), tier: OptimizationTier::Optimized.index(), bead_id, @@ -4300,7 +4507,8 @@ fn ensure_baseline( /// The region goes into `regions` under the bead before the site is /// published, for the tier above to give the frame resume points of its /// own (see [`promote_outlined_region`]): the frame may ask for that -/// promotion from the region the moment it is in it. +/// promotion from the region the moment it is in it. Whether the region +/// was attempted: false when there was no such loop or no frame waits. #[allow(clippy::too_many_arguments, clippy::type_complexity)] fn publish_outlined_resume_point( cranelift: &Arc, @@ -4309,12 +4517,12 @@ fn publish_outlined_resume_point( func_arc: &Arc, module_arc: &Arc, asked_at: u64, - optimize: &(dyn Fn(u64, HirFunction) -> HirFunction + Send + Sync), + optimize: &(dyn Fn(HirFunction) -> HirFunction + Send + Sync), regions: &Mutex)>>>, -) { +) -> bool { let def = ZyntaxFunctionDef { id: func_id, - function: (**func_arc).clone(), + function: Arc::clone(func_arc), module: Arc::clone(module_arc), tier: OptimizationTier::Baseline.index(), bead_id, @@ -4323,7 +4531,7 @@ fn publish_outlined_resume_point( .get(asked_at as usize) .copied() else { - return; + return false; }; if !osr::frame_waiting(bead_id) { if osr::osr_trace_enabled() { @@ -4332,7 +4540,7 @@ fn publish_outlined_resume_point( func_arc.name.resolve_global().unwrap_or_default() ); } - return; + return false; } let outlined = std::thread::scope(|scope| { std::thread::Builder::new() @@ -4340,7 +4548,7 @@ fn publish_outlined_resume_point( .stack_size(16 << 20) .spawn_scoped(scope, || { cranelift - .outlined_resume_point_at(&def, header, &|f| optimize(bead_id, f)) + .outlined_resume_point_at(&def, header, optimize) .map(|out| { let sites: Vec<(u64, usize)> = out .sites @@ -4361,7 +4569,7 @@ fn publish_outlined_resume_point( ); } publish_baseline_resume_points(cranelift, func_id, bead_id, func_arc, module_arc, asked_at); - return; + return true; }; regions .lock() @@ -4381,6 +4589,7 @@ fn publish_outlined_resume_point( osr::publish_helper(bead_id, site, code); } } + true } /// The optimizing tier's resume points for an outlined region: the @@ -4410,7 +4619,7 @@ fn promote_outlined_region( crate::hir_dump::dump_function_to_dir(region, module_arc, &format!("{name}-tier{tier_idx}")); let def = ZyntaxFunctionDef { id: region_id, - function: (**region).clone(), + function: Arc::clone(region), module: Arc::clone(module_arc), tier: tier_idx, bead_id, @@ -4450,7 +4659,7 @@ fn publish_baseline_resume_points( ) { let def = ZyntaxFunctionDef { id: func_id, - function: (**func_arc).clone(), + function: Arc::clone(func_arc), module: Arc::clone(module_arc), tier: OptimizationTier::Baseline.index(), bead_id, @@ -4548,7 +4757,7 @@ pub fn compile_at_tier( ) -> *mut () { let def = ZyntaxFunctionDef { id: func_id, - function: (**func_arc).clone(), + function: Arc::clone(func_arc), module: Arc::clone(module_arc), tier: tier_idx, bead_id, diff --git a/crates/compiler/tests/optimized_once.rs b/crates/compiler/tests/optimized_once.rs new file mode 100644 index 00000000..fe30f907 --- /dev/null +++ b/crates/compiler/tests/optimized_once.rs @@ -0,0 +1,114 @@ +//! A lazy function's HIR goes through the pipeline once, and every tier +//! above the interpreter compiles that body: the baseline, the LLVM tier +//! through the promotion requester, and the LLVM tier through an +//! explicit `optimize_function`. + +#![cfg(all(feature = "llvm-backend", feature = "cranelift-backend"))] + +mod common; + +use std::collections::HashSet; +use std::sync::Arc; + +use zyntax_compiler::hir::{HirFunction, HirId, HirModule}; +use zyntax_compiler::opt_audit; +use zyntax_compiler::osr; +use zyntax_compiler::tiered_backend::{ + OptimizationTier, Tier2Backend, TieredBackend, TieredConfig, +}; +use zyntax_typed_ast::InternedString; + +/// `count_to`, left for its first call as the runtime leaves a +/// program's functions. +fn lazy_loop() -> HirFunction { + let (mut function, _) = common::counted_loop(); + function.attributes.optimized = true; + function.attributes.deferred = true; + function +} + +fn call(entry: *const u8, n: i32) -> i32 { + // SAFETY: the entry is `count_to`'s code, `fn(i32) -> i32`. + let f: extern "C" fn(i32) -> i32 = unsafe { std::mem::transmute(entry) }; + f(n) +} + +/// Waits for the LLVM tier to have been handed `id`'s HIR `n` times. +fn llvm_compiles(id: HirId, n: usize) -> Vec { + for _ in 0..1000 { + let bodies = opt_audit::llvm_bodies(id); + if bodies.len() >= n { + return bodies; + } + std::thread::sleep(std::time::Duration::from_millis(10)); + } + opt_audit::llvm_bodies(id) +} + +/// One test: the lazy compiler and the requester a backend installs are +/// the process's. +#[test] +fn every_tier_compiles_the_one_optimised_body() { + opt_audit::enable(); + let (by_requester, by_request) = (lazy_loop(), lazy_loop()); + let (a, b) = (by_requester.id, by_request.id); + let mut module = HirModule::new(InternedString::new_global("optimized_once")); + module.functions.insert(a, by_requester); + module.functions.insert(b, by_request); + let config = TieredConfig { + tier2_backend: Tier2Backend::LLVM, + enable_osr: true, + ..TieredConfig::default() + }; + let mut backend = TieredBackend::new(config).expect("tiered backend"); + backend.set_emit_osr_probes(true); + backend + .compile_module_lazily(module, None, HashSet::from([a, b]), HashSet::new(), false) + .expect("module compiles"); + + // Each is called through its stub, which optimises the body and + // compiles it at the baseline; a call too short for its loop to ask + // for the tier above. + let (_, entry, bead) = backend.interpreter_bridge(); + for id in [a, b] { + assert_eq!(call(entry(id).expect("a stub"), 5), 10); + } + let (bead_a, bead_b) = (bead(a).expect("a bead"), bead(b).expect("a bead")); + let stored = |bead| osr::lazy_optimized_body(bead).expect("an optimised body"); + let (body_a, body_b) = (stored(bead_a), stored(bead_b)); + + // The requester, as a compiled frame asks. + assert!(osr::run_promotion( + bead_a, + osr::Requester::Compiled { site: osr::NO_SITE } + )); + let handed = llvm_compiles(a, 1); + assert_eq!( + handed, + vec![Arc::as_ptr(&body_a) as usize], + "the requester's LLVM compile took a body other than the optimised one" + ); + + // An explicit promotion. + backend + .optimize_function(b, OptimizationTier::Optimized) + .expect("promotion"); + let handed = llvm_compiles(b, 1); + assert_eq!( + handed, + vec![Arc::as_ptr(&body_b) as usize], + "optimize_function's LLVM compile took a body other than the optimised one" + ); + + for id in [a, b] { + assert_eq!( + opt_audit::pipeline_runs(id), + 1, + "the pipeline ran more than once over one function" + ); + } + // The interpreter runs the same body. + let mut source = backend.interpreter_body_source(); + let interp = source(a).expect("a body for the interpreter"); + assert!(Arc::ptr_eq(&interp, &body_a)); +} diff --git a/crates/compiler/tests/optimized_once_race.rs b/crates/compiler/tests/optimized_once_race.rs new file mode 100644 index 00000000..fc6e1689 --- /dev/null +++ b/crates/compiler/tests/optimized_once_race.rs @@ -0,0 +1,87 @@ +//! Callers asking for a lazy function's optimised body at once, while +//! the compile worker makes it too, all get one body, made by one run of +//! the pipeline. + +#![cfg(feature = "cranelift-backend")] + +mod common; + +use std::collections::HashSet; +use std::sync::{Arc, Barrier}; + +use zyntax_compiler::hir::{ + HirConstant, HirInstruction, HirModule, HirType, HirValue, HirValueKind, +}; +use zyntax_compiler::opt_audit; +use zyntax_compiler::osr; +use zyntax_compiler::tiered_backend::{TieredBackend, TieredConfig}; +use zyntax_typed_ast::InternedString; + +const ASKERS: usize = 8; + +/// One test: the lazy optimizer a backend installs is the process's. +#[test] +fn racing_callers_share_one_run_of_the_pipeline() { + opt_audit::enable(); + for _ in 0..200 { + // A loop hot before its first call, so the worker is asked to + // compile it as the module loads. + let (mut function, header) = common::counted_loop(); + let bound = zyntax_compiler::hir::HirId::new(); + function.values.insert( + bound, + HirValue { + id: bound, + ty: HirType::I32, + kind: HirValueKind::Constant(HirConstant::I32(5000)), + uses: Default::default(), + span: None, + }, + ); + for inst in &mut function.blocks.get_mut(&header).unwrap().instructions { + if let HirInstruction::Binary { right, .. } = inst { + *right = bound; + } + } + function.attributes.optimized = true; + function.attributes.deferred = true; + let id = function.id; + let mut module = HirModule::new(InternedString::new_global("optimized_once_race")); + module.functions.insert(id, function); + let config = TieredConfig { + enable_osr: true, + ..TieredConfig::default() + }; + let mut backend = TieredBackend::new(config).expect("tiered backend"); + backend + .compile_module_lazily(module, None, HashSet::from([id]), HashSet::new(), false) + .expect("module compiles"); + let (_, _, bead) = backend.interpreter_bridge(); + let bead = bead(id).expect("a bead"); + + let barrier = Arc::new(Barrier::new(ASKERS)); + let askers: Vec<_> = (0..ASKERS) + .map(|_| { + let barrier = Arc::clone(&barrier); + std::thread::spawn(move || { + barrier.wait(); + osr::lazy_optimized_body(bead).expect("an optimised body") + }) + }) + .collect(); + let bodies: Vec<_> = askers + .into_iter() + .map(|t| t.join().expect("asker")) + .collect(); + assert!( + bodies.iter().all(|b| Arc::ptr_eq(b, &bodies[0])), + "callers were handed different bodies" + ); + drop(backend); + assert_eq!( + opt_audit::pipeline_runs(id), + 1, + "the pipeline ran more than once over one function" + ); + } +} From 9ca2b340f4cd454729b67483a59f594e168ef732 Mon Sep 17 00:00:00 2001 From: darmie Date: Fri, 25 Sep 2026 20:51:46 +0100 Subject: [PATCH 2/3] Interpreter: a frame holds the body its bytecode came from A frame of a function with loops holds the body the body source gave while it runs, so a resume point it asks for is made from that body whatever the runtime keeps. Bytecode whose body nobody holds is made again. The runtime is told when the first frame returns and may let the body, and the bytecode, go. --- crates/compiler/src/hir_interp.rs | 76 ++++++++++++++++++++++++++++++- 1 file changed, 74 insertions(+), 2 deletions(-) diff --git a/crates/compiler/src/hir_interp.rs b/crates/compiler/src/hir_interp.rs index 2e940171..9af2c41e 100644 --- a/crates/compiler/src/hir_interp.rs +++ b/crates/compiler/src/hir_interp.rs @@ -2674,6 +2674,15 @@ impl std::error::Error for InterpError {} // Interpreter (dispatch loop) // ───────────────────────────────────────────────────────────────────────────── +/// The body a function's bytecode was made from, when the body source +/// gave it. Held weakly: the runtime decides how long it lives, and a +/// running frame of a function with loops holds it besides. +struct SourceBody { + body: std::sync::Weak, + /// The bytecode has loop headers a frame can ask resume points at. + loops: bool, +} + pub struct HirInterpreter { symbols: HashMap, pub profile: IdMap, @@ -2697,6 +2706,13 @@ pub struct HirInterpreter { /// Where a function's body comes from when the runtime keeps one /// apart from the module's: see [`Self::set_body_source`]. body_source: Option Option> + Send>>, + /// The body each function's bytecode was made from, for those the + /// body source gave: see [`SourceBody`]. + source_bodies: IdMap, + /// Told when the first frame run from a body the body source gave + /// returns; see [`Self::set_frame_exit_hook`]. + #[allow(clippy::type_complexity)] + frame_exit_hook: Option bool + Send>>, /// Compiles, or finds, the thunk that calls native code of a given /// shape: `fn(target, words, out)`. Installed by a runtime with a /// native tier; without one, calls into native code use the fixed @@ -2850,6 +2866,8 @@ impl HirInterpreter { uncompilable: IdMap::default(), tick_callbacks: IdMap::default(), body_source: None, + source_bodies: IdMap::default(), + frame_exit_hook: None, thunk_source: None, entry_source: None, bead_source: None, @@ -2980,6 +2998,14 @@ impl HirInterpreter { self.body_source = Some(source); } + /// Called with a function's id when the first frame run from a body + /// the body source gave returns. Answers whether the runtime let that + /// body go; the bytecode made from it goes too then, and the next + /// call asks the body source again. + pub fn set_frame_exit_hook(&mut self, hook: Box bool + Send>) { + self.frame_exit_hook = Some(hook); + } + /// Drop the bridge and every tick callback. They hold the native /// tiers' backends, which the runtime shuts down after them. pub fn clear_native_bridge(&mut self) { @@ -2987,6 +3013,7 @@ impl HirInterpreter { self.thunk_source = None; self.entry_source = None; self.bead_source = None; + self.frame_exit_hook = None; } /// Install the bridge to a native tier: `thunk` compiles the caller @@ -3520,6 +3547,26 @@ impl HirInterpreter { } } + // A frame of a function with loops holds the body its bytecode + // was made from, so a request for resume points it makes finds + // that body while it runs. Bytecode whose body nobody holds any + // more is made again from the body the source gives now. + let mut held: Option> = None; + let mut stale = false; + if let Some(source) = self.source_bodies.get(&func_id) + && source.loops + { + held = source.body.upgrade(); + stale = held.is_none(); + } + // Whether this frame made the bytecode from a body the source + // gave: the runtime is told when it returns. + let mut first = false; + if stale { + self.source_bodies.remove(&func_id); + self.cache.remove(&func_id); + } + // Compile-on-first-use, and refuse-once. What the interpreter // cannot run, native code runs, when there is native code. if !self.cache.contains_key(&func_id) { @@ -3551,6 +3598,21 @@ impl HirInterpreter { let taken = self.is_address_taken(module, func_id); match compile_function_with(module, &mut self.memory, func, taken) { Ok(cf) => { + // A recursive call compiles the body again while + // the frame that made the entry runs. + if let Some(body) = &shared { + let loops = !cf.osr_sites.is_empty(); + self.source_bodies.entry(func_id).or_insert_with(|| { + first = true; + SourceBody { + body: std::sync::Arc::downgrade(body), + loops, + } + }); + if loops { + held = Some(std::sync::Arc::clone(body)); + } + } self.cache.insert(func_id, cf); } Err(InterpError::UnsupportedInstruction(why)) => { @@ -3574,8 +3636,18 @@ impl HirInterpreter { // calls. The map ownership returns at the end. let cf = self.cache.remove(&func_id).unwrap(); let result = self.run(module, &cf, args, func_id, dest); - // Put the (immutable) compiled function back. - self.cache.insert(func_id, cf); + drop(held); + let released = first + && self + .frame_exit_hook + .as_mut() + .is_some_and(|hook| hook(func_id)); + if released { + self.source_bodies.remove(&func_id); + } else { + // Put the (immutable) compiled function back. + self.cache.insert(func_id, cf); + } result } From 460de2e5b2ce8e12acb2370e9439a50125b2c70d Mon Sep 17 00:00:00 2001 From: darmie Date: Fri, 25 Sep 2026 20:51:46 +0100 Subject: [PATCH 3/3] Tiered: the scratch holds what it optimises; bodies go when unused The scratch module copies in only the bodies an optimisation reaches through direct calls and is dropped once no optimisation is under way and no compile waits. An interp body goes once its function has native code, or when a large body's first frame returns; an optimised body goes after the ladder's last compile unless it is small enough to be inlined. A frame still running a body finds it, and a body asked for again is made again, once per cell. git-bug: 08988fa9d5b1d1f4ef42a0f9281172a42e87167a460a92dadc84c5972840ebee --- crates/compiler/src/inline.rs | 6 + crates/compiler/src/tiered_backend.rs | 615 +++++++++++++++++----- crates/compiler/tests/body_retention.rs | 303 +++++++++++ crates/zyntax_embed/src/runtime/tiered.rs | 1 + 4 files changed, 779 insertions(+), 146 deletions(-) create mode 100644 crates/compiler/tests/body_retention.rs diff --git a/crates/compiler/src/inline.rs b/crates/compiler/src/inline.rs index 8bbce6c1..bdbaea3e 100644 --- a/crates/compiler/src/inline.rs +++ b/crates/compiler/src/inline.rs @@ -181,6 +181,12 @@ fn count_insts(function: &HirFunction) -> usize { function.blocks.values().map(|b| b.instructions.len()).sum() } +/// Whether `function` is small enough that some call to it may be +/// inlined. +pub(crate) fn may_inline(function: &HirFunction) -> bool { + count_insts(function) <= MAX_INLINE_INSTS_MULTI_BLOCK +} + /// Inline every eligible direct call within `module`. Iterates a /// fixed-point: inlining one call can expose a now-eligible callee /// (when the now-inlined body's prior nested call shape was a diff --git a/crates/compiler/src/tiered_backend.rs b/crates/compiler/src/tiered_backend.rs index 2a772854..78ac755d 100644 --- a/crates/compiler/src/tiered_backend.rs +++ b/crates/compiler/src/tiered_backend.rs @@ -317,6 +317,321 @@ fn tier_body( module.functions.get(&func_id).map(|f| Arc::new(f.clone())) } +impl OptimizedBodies { + /// An empty cell in place of `id`'s, if that cell still holds `body`: + /// a cell a reload filled since keeps the reload's body. The next + /// request for the body makes it again, once, in the new cell. + fn release(&self, id: HirId, body: &Arc) -> bool { + let mut cells = self.0.write().unwrap(); + let holds = cells + .get(&id) + .and_then(|cell| cell.get()) + .is_some_and(|held| Arc::ptr_eq(held, body)); + if holds { + cells.insert(id, BodyCell::default()); + } + holds + } +} + +/// Whether the optimised body of a function is kept once every tier +/// that compiles it has: a body small enough to be inlined is what a +/// caller optimised later inlines. +fn kept_for_inlining(body: &HirFunction) -> bool { + crate::inline::may_inline(body) +} + +/// A body at least this many instructions long whose first interpreted +/// frame returns before the function has native code is let go then: +/// such a body is mostly run once, and costs more kept than made again. +const RUN_ONCE_INSTRUCTIONS: usize = 512; + +/// What the interpreter was given of one lazy function. +#[derive(Default)] +struct FrameBodies { + /// The body made for the interpreter's first runs, while it is kept: + /// until the function has native code, or, for a large body, until + /// its first frame returns. A frame running it holds it besides. + interp: Option>, + /// The body `interp` held last, while anyone holds it. + interp_weak: std::sync::Weak, + /// The tier body last handed to the interpreter, while a frame + /// running it holds it. + tier_weak: std::sync::Weak, + /// An interp body was made; one made after it was let go is kept + /// until the function has native code. + made: bool, + remade: bool, +} + +impl FrameBodies { + /// The interp body, while anyone holds it. + fn interp(&self) -> Option> { + self.interp.clone().or_else(|| self.interp_weak.upgrade()) + } +} + +/// The interp bodies of the lazy functions, by function id. +type InterpBodies = Mutex>; + +/// The interp body of `func_id`: the one held, else one made now. +#[allow(clippy::type_complexity)] +fn interp_body( + bodies: &InterpBodies, + func_id: HirId, + bead_id: u64, + make: Option<&Arc Option> + Send + Sync>>, +) -> Option> { + if let Some(body) = bodies + .lock() + .unwrap() + .get(&func_id) + .and_then(FrameBodies::interp) + { + return Some(body); + } + let body = make?(bead_id)?; + let mut bodies = bodies.lock().unwrap(); + let entry = bodies.entry(func_id).or_default(); + if let Some(body) = entry.interp() { + return Some(body); + } + entry.remade |= entry.made; + entry.made = true; + entry.interp_weak = Arc::downgrade(&body); + entry.interp = Some(Arc::clone(&body)); + Some(body) +} + +/// Let `func_id`'s interp body go; a frame running it keeps it alive +/// for as long as it runs. +fn release_interp_body(bodies: &InterpBodies, func_id: HirId) { + if let Some(entry) = bodies.lock().unwrap().get_mut(&func_id) { + entry.interp = None; + } +} + +fn instruction_count(f: &HirFunction) -> usize { + f.blocks.values().map(|b| b.instructions.len()).sum() +} + +/// The functions `f` calls directly. +fn direct_callees(f: &HirFunction) -> impl Iterator + '_ { + use crate::hir::{HirCallable, HirInstruction, HirTerminator}; + f.blocks.values().flat_map(|b| { + let calls = b.instructions.iter().filter_map(|inst| match inst { + HirInstruction::Call { + callee: HirCallable::Function(target), + .. + } => Some(*target), + _ => None, + }); + let invoke = match &b.terminator { + HirTerminator::Invoke { + callee: HirCallable::Function(target), + .. + } => Some(*target), + _ => None, + }; + calls.chain(invoke) + }) +} + +/// A module the program's lazy functions are optimised in, one at a +/// time: the module's declarations and extern functions, and of its +/// defined functions those an optimised body reaches through direct +/// calls, each copied in when first reached. Every function in it has +/// its direct callees in it too, which is all any pass reads of a +/// function other than the one it optimises; what the passes know of +/// the module as a whole comes from [`Facts`]. +struct Scratch { + module: HirModule, + /// The functions of `module` through the pipeline in it. + done: HashSet, +} + +impl Scratch { + /// The scratch of `module` in `slots`, made if none is. + fn of<'a>(slots: &'a mut HashMap, module: &Arc) -> &'a mut Scratch { + let key = Arc::as_ptr(module) as usize; + slots.entry(key).or_insert_with(|| { + let started = web_time::Instant::now(); + let externs = module + .functions + .iter() + .filter(|(_, f)| f.is_external) + .map(|(id, f)| { + let mut f = f.clone(); + f.attributes.optimized = true; + f.attributes.deferred = true; + (*id, f) + }) + .collect(); + let scratch = HirModule { + id: module.id, + name: module.name, + functions: externs, + globals: module.globals.clone(), + types: module.types.clone(), + imports: module.imports.clone(), + exports: module.exports.clone(), + version: module.version, + dependencies: module.dependencies.clone(), + effects: module.effects.clone(), + handlers: module.handlers.clone(), + automatic_release: module.automatic_release, + }; + if std::env::var_os("ZYNTAX_TRACE_LAZY").is_some() { + eprintln!( + "[lazy] scratch module made in {:.2} ms on {}", + started.elapsed().as_secs_f64() * 1e3, + std::thread::current().name().unwrap_or("?") + ); + } + Scratch { + module: scratch, + done: HashSet::new(), + } + }) + } + + /// Copy in `roots` and every function they reach through direct + /// calls that is not in yet, each marked through the pipeline and + /// deferred, so a pass that walks the module touches the one being + /// optimised alone. A program function whose cell is filled goes in + /// as its cell holds it, so a body optimised later inlines it as + /// every tier runs it; one that arrived optimised stays as it + /// arrived, box readers and all, as the release pass reads it. + fn reach( + &mut self, + roots: impl IntoIterator, + module: &HirModule, + bodies: &OptimizedBodies, + program: &HashSet, + ) { + let mut work: Vec = roots.into_iter().collect(); + while let Some(id) = work.pop() { + if self.module.functions.contains_key(&id) { + continue; + } + let Some(lowered) = module.functions.get(&id) else { + continue; + }; + let optimized = program.contains(&id).then(|| bodies.get(id)).flatten(); + let mut f = match optimized { + Some(body) => { + self.done.insert(id); + (*body).clone() + } + None => lowered.clone(), + }; + f.attributes.optimized = true; + f.attributes.deferred = true; + work.extend(direct_callees(&f)); + self.module.functions.insert(id, f); + } + } +} + +/// The scratch modules of one lazy compiler, and the optimisations under +/// way in them. They are let go once none is under way and no compile +/// is waiting, and made again as asked. +#[derive(Default)] +struct Scratches { + slots: Mutex>, + pending: std::sync::atomic::AtomicUsize, +} + +impl Scratches { + /// Note an optimisation about to use a scratch, until the returned + /// guard drops; `quiet` answers whether no compile is waiting. + fn enter<'a>(&'a self, quiet: &'a (dyn Fn() -> bool + Send + Sync)) -> ScratchUse<'a> { + self.pending + .fetch_add(1, std::sync::atomic::Ordering::SeqCst); + ScratchUse { + scratches: self, + quiet, + } + } + + /// Let every scratch go, once the optimisation in one, if any, is + /// done. + fn clear(&self) { + let gone = std::mem::take(&mut *self.slots.lock().unwrap()); + drop(gone); + } + + /// Let every scratch go if no optimisation is under way. + fn release_if_unused(&self) { + let mut slots = self.slots.lock().unwrap(); + if self.pending.load(std::sync::atomic::Ordering::SeqCst) != 0 || slots.is_empty() { + return; + } + let gone = std::mem::take(&mut *slots); + drop(slots); + drop(gone); + } +} + +struct ScratchUse<'a> { + scratches: &'a Scratches, + quiet: &'a (dyn Fn() -> bool + Send + Sync), +} + +impl Drop for ScratchUse<'_> { + fn drop(&mut self) { + let last = self + .scratches + .pending + .fetch_sub(1, std::sync::atomic::Ordering::SeqCst) + == 1; + if last && (self.quiet)() { + self.scratches.release_if_unused(); + } + } +} + +/// What the passes know of each module as a whole, built once, by the +/// first thread that needs it: the release pass's facts, and the call +/// cycles the inliner keeps within. A linked library's functions carry +/// their facts, so the build covers the program's own. +#[derive(Default)] +struct Facts { + facts: Mutex>>, + caches: Mutex>>, +} + +impl Facts { + fn of(&self, module: &Arc) -> Arc { + let key = Arc::as_ptr(module) as usize; + let mut built = self.facts.lock().unwrap(); + Arc::clone(built.entry(key).or_insert_with(|| { + let started = web_time::Instant::now(); + let facts = Arc::new(crate::drop_insert::facts_of(module)); + if std::env::var_os("ZYNTAX_TRACE_LAZY").is_some() { + eprintln!( + "[lazy] release facts built in {:.2} ms on {}", + started.elapsed().as_secs_f64() * 1e3, + std::thread::current().name().unwrap_or("?") + ); + } + facts + })) + } + + /// The pipeline's cache for `module`, over the module as lowered: + /// inlining keeps every function's reach, so its call cycles hold + /// for the bodies optimised since. + fn cache(&self, module: &Arc) -> Arc { + let key = Arc::as_ptr(module) as usize; + if let Some(cache) = self.caches.lock().unwrap().get(&key) { + return Arc::clone(cache); + } + let cache = Arc::new(crate::OptCache::with_facts(self.of(module), module)); + Arc::clone(self.caches.lock().unwrap().entry(key).or_insert(cache)) + } +} + /// What the tier above answered a bead's promotion request with, kept /// per bead and tier so the answer is not worked out again. #[derive(Clone, Copy)] @@ -393,6 +708,15 @@ impl CompileQueue { !self.busy.load(std::sync::atomic::Ordering::Acquire) } + /// Whether the worker is in no job and no request waits for it. + fn quiet(&self) -> bool { + if !self.idle() { + return false; + } + let promote = self.promote.lock().unwrap().is_empty(); + promote && self.compile.lock().unwrap().is_empty() + } + /// Whether a compile of `bead_id` was asked for already. fn is_queued(&self, bead_id: u64) -> bool { self.queued.lock().unwrap().contains(&bead_id) @@ -567,7 +891,7 @@ type FirstCompiles = ( struct QuickBaseline { /// The program's own lazy functions; a finished one arrived optimised. lazy: HashSet, - interp_bodies: Arc>>>, + interp_bodies: Arc, #[allow(clippy::type_complexity)] make_interp_body: Option Option> + Send + Sync>>, /// What each quick baseline's call cell held when it was published: @@ -588,17 +912,12 @@ impl QuickBaseline { { return None; } - if let Some(body) = self.interp_bodies.lock().unwrap().get(&func_id) { - return Some(Arc::clone(body)); - } - let body = self.make_interp_body.as_ref()?(bead_id)?; - Some(Arc::clone( - self.interp_bodies - .lock() - .unwrap() - .entry(func_id) - .or_insert(body), - )) + interp_body( + &self.interp_bodies, + func_id, + bead_id, + self.make_interp_body.as_ref(), + ) } fn note_published(&self, bead_id: u64) { @@ -744,9 +1063,12 @@ pub struct TieredBackend { optimize_extra: Option HirFunction + Send + Sync>>, /// The body the interpreter runs a lazy function's first calls on: /// the module's, as lowered, with its releases placed (see - /// [`Self::interpreter_body_source`]). Made at the first run and - /// kept: the sites a frame on it asks at are this body's. - interp_bodies: Arc>>>, + /// [`Self::interpreter_body_source`]). Made at the first run, and + /// found by a frame running it for as long as it runs: the sites the + /// frame asks at are this body's. + interp_bodies: Arc, + /// The current lazy compiler's scratch modules. + scratches: Arc, /// Makes an entry of `interp_bodies` for a bead. Installed with the /// lazy compiler. #[allow(clippy::type_complexity)] @@ -880,6 +1202,7 @@ impl TieredBackend { static_hot: HashMap::new(), optimize_extra: None, interp_bodies: Arc::new(Mutex::new(HashMap::new())), + scratches: Arc::new(Scratches::default()), make_interp_body: None, current_module: None, loaded: Vec::new(), @@ -1836,6 +2159,8 @@ impl TieredBackend { // The promotion requester captured each function's body when it // was installed; reinstall so a later promotion compiles the // edited bodies rather than the ones it captured. + // A scratch holds the bodies it copied in before the swap. + self.scratches.clear(); self.install_promotion_requester(); // A reload that changed nothing keeps the previous undo record: @@ -1918,6 +2243,7 @@ impl TieredBackend { } } + self.scratches.clear(); self.install_promotion_requester(); Ok(restored) } @@ -2580,19 +2906,54 @@ impl TieredBackend { let finished = self.finished.clone(); Box::new(move |id: HirId| { let bead = beads.get(&id)?; - if let Some(body) = optimized_bodies.get(id) { + let tier = match optimized_bodies.get(id) { + Some(body) => Some(body), + None if finished.contains(&id) => Some(osr::lazy_optimized_body(*bead)?), + None => None, + }; + if let Some(body) = tier { + interp_bodies + .lock() + .unwrap() + .entry(id) + .or_default() + .tier_weak = Arc::downgrade(&body); return Some(body); } - if finished.contains(&id) { - return osr::lazy_optimized_body(*bead); - } - if let Some(body) = interp_bodies.lock().unwrap().get(&id) { - return Some(Arc::clone(body)); + interp_body(&interp_bodies, id, *bead, make_interp_body.as_ref()) + }) + } + + /// What the interpreter tells when the first frame run from a body + /// [`Self::interpreter_body_source`] gave returns: the interp body is + /// let go once the function has native code, or then for a large + /// body the first time, which is mostly run once. Answers whether + /// it was. + pub fn interpreter_frame_exit_hook(&self) -> Box bool + Send> { + let beads: HashMap = self + .functions + .iter() + .filter(|(id, _)| self.lazy.contains(id)) + .map(|(id, e)| (*id, e.bound.clone())) + .collect(); + let interp_bodies = Arc::clone(&self.interp_bodies); + Box::new(move |id: HirId| { + let Some(bound) = beads.get(&id) else { + return false; + }; + let mut bodies = interp_bodies.lock().unwrap(); + let Some(entry) = bodies.get_mut(&id) else { + return false; + }; + let Some(body) = &entry.interp else { + return false; + }; + let native = bound.bead().compiled().is_some(); + if native || (!entry.remade && instruction_count(body) >= RUN_ONCE_INSTRUCTIONS) { + entry.interp = None; + return true; } - let body = make_interp_body.as_ref()?(*bead)?; - Some(Arc::clone( - interp_bodies.lock().unwrap().entry(id).or_insert(body), - )) + false }) } @@ -2771,91 +3132,24 @@ impl TieredBackend { None => (HashMap::new(), HashSet::new()), }; // The program's own bodies were left as lowered. Each is - // optimised on its own at its first compile, in one scratch copy - // of the module: a body optimised earlier is what a later one + // optimised on its own at its first compile, in a scratch module + // (see [`Scratch`]): a body optimised earlier is what a later one // inlines, and what the passes know about the module as a whole // is built once, since optimising a body changes none of it. - struct Scratch { - module: HirModule, - cache: crate::OptCache, - /// The functions of `module` through the pipeline in it. - done: HashSet, - } - // One scratch per module: a program that compiles a chunk of - // itself has bodies in more than one. - impl Scratch { - /// The scratch of `module` in `slots`, made if none is: every - /// function marked through the pipeline and deferred, so a - /// pass that walks the module touches the one being optimised - /// alone. A program function whose cell is filled goes in as - /// its cell holds it, so a body optimised later inlines it as - /// every tier runs it; one that arrived optimised stays as it - /// arrived, box readers and all, as the release pass reads it. - fn of<'a>( - slots: &'a mut HashMap, - module: &Arc, - facts: &Facts, - bodies: &OptimizedBodies, - program: &HashSet, - ) -> &'a mut Scratch { - let key = Arc::as_ptr(module) as usize; - slots.entry(key).or_insert_with(|| { - let started = web_time::Instant::now(); - let facts = facts.of(module); - let mut module: HirModule = (**module).clone(); - let mut done = HashSet::new(); - for (id, f) in module.functions.iter_mut() { - if program.contains(id) - && let Some(body) = bodies.get(*id) - { - *f = (*body).clone(); - done.insert(*id); - } - f.attributes.optimized = true; - f.attributes.deferred = true; - } - let cache = crate::OptCache::with_facts(facts, &module); - if std::env::var_os("ZYNTAX_TRACE_LAZY").is_some() { - eprintln!( - "[lazy] scratch module made in {:.2} ms on {}", - started.elapsed().as_secs_f64() * 1e3, - std::thread::current().name().unwrap_or("?") - ); - } - Scratch { - module, - cache, - done, - } - }) - } - } - /// What the release pass knows of each module, built once, by - /// the first thread that needs it. A linked library's functions - /// carry their facts, so the build covers the program's own. - #[derive(Default)] - struct Facts(Mutex>>); - impl Facts { - fn of(&self, module: &Arc) -> Arc { - let key = Arc::as_ptr(module) as usize; - let mut built = self.0.lock().unwrap(); - Arc::clone(built.entry(key).or_insert_with(|| { - let started = web_time::Instant::now(); - let facts = Arc::new(crate::drop_insert::facts_of(module)); - if std::env::var_os("ZYNTAX_TRACE_LAZY").is_some() { - eprintln!( - "[lazy] release facts built in {:.2} ms on {}", - started.elapsed().as_secs_f64() * 1e3, - std::thread::current().name().unwrap_or("?") - ); - } - facts - })) - } - } let facts: Arc = Arc::new(Facts::default()); - let optimized: Arc>> = Arc::new(Mutex::new(HashMap::new())); + let optimized: Arc = Arc::new(Scratches::default()); + self.scratches = Arc::clone(&optimized); let scratch_shared = Arc::clone(&optimized); + // The queue is made after the closures below, which read it + // through this cell once it exists. + let queue_for_callees: Arc>>> = Arc::new(Mutex::new(None)); + let queue_cell = Arc::clone(&queue_for_callees); + // Whether no compile waits for the worker, which would take a + // scratch again. + let quiet: Arc bool + Send + Sync> = { + let queue = Arc::clone(&queue_for_callees); + Arc::new(move || queue.lock().unwrap().as_ref().is_none_or(|q| q.quiet())) + }; // The optimised body outlives the baseline compile for the tier // above it; a ladder that ends at the baseline drops it then. #[cfg(feature = "llvm-backend")] @@ -2897,6 +3191,7 @@ impl TieredBackend { let finished = finished.clone(); let lazy = lazy.clone(); let finish = Arc::clone(&finish); + let quiet = Arc::clone(&quiet); move |bead_id: u64| -> Option> { let (func_id, module_arc) = by_bead.get(&bead_id)?; let cell = optimized_bodies.cell(*func_id)?; @@ -2918,9 +3213,10 @@ impl TieredBackend { } // The scratch is held until the body is in the cell's // hands, so no other caller sees it half made. - let mut optimized = optimized.lock().unwrap(); - let scratch = - Scratch::of(&mut optimized, module_arc, &facts, &optimized_bodies, &lazy); + let _using = optimized.enter(&*quiet); + let mut slots = optimized.slots.lock().unwrap(); + let scratch = Scratch::of(&mut slots, module_arc); + scratch.reach([*func_id], module_arc, &optimized_bodies, &lazy); if !scratch.done.contains(func_id) { let f = scratch .module @@ -2941,7 +3237,10 @@ impl TieredBackend { &format!("{name}-lowered"), ); } - crate::run_interp_safe_opts_cached(&mut scratch.module, &scratch.cache); + crate::run_interp_safe_opts_cached( + &mut scratch.module, + &facts.cache(module_arc), + ); scratch.done.insert(*func_id); } let f = scratch @@ -2988,6 +3287,7 @@ impl TieredBackend { let optimized_bodies = Arc::clone(&optimized_bodies); let finish = Arc::clone(&finish); let lazy = lazy.clone(); + let quiet = Arc::clone(&quiet); move |bead_id: u64, mut f: HirFunction, already_optimized: bool| -> HirFunction { if already_optimized { finish(&mut f); @@ -3000,12 +3300,14 @@ impl TieredBackend { let Some(module_arc) = by_bead.get(&bead_id) else { return f; }; - let mut optimized = optimized.lock().unwrap(); - let scratch = - Scratch::of(&mut optimized, module_arc, &facts, &optimized_bodies, &lazy); + let _using = optimized.enter(&*quiet); + let mut slots = optimized.slots.lock().unwrap(); + let scratch = Scratch::of(&mut slots, module_arc); + let callees: Vec = direct_callees(&f).collect(); + scratch.reach(callees, module_arc, &optimized_bodies, &lazy); let id = f.id; scratch.module.functions.insert(id, f); - crate::run_interp_safe_opts_cached(&mut scratch.module, &scratch.cache); + crate::run_interp_safe_opts_cached(&mut scratch.module, &facts.cache(module_arc)); // No pass removes a function, so it is there to take back. let mut f = scratch .module @@ -3049,10 +3351,6 @@ impl TieredBackend { })); // What compiling a function on its first call does, once off the // caller's stack. - // The queue is made after this closure, which reads it through - // this cell once it exists. - let queue_for_callees: Arc>>> = Arc::new(Mutex::new(None)); - let queue_cell = Arc::clone(&queue_for_callees); let bead_of: HashMap = by_bead.iter().map(|(b, (id, _, _))| (*id, *b)).collect(); let published = Arc::clone(&done); @@ -3191,6 +3489,7 @@ impl TieredBackend { quick_baseline.note_published(bead_id); } publish(entry as usize, true); + release_interp_body(&quick_baseline.interp_bodies, *func_id); queue.request_replacement(bead_id); if trace { eprintln!( @@ -3267,13 +3566,6 @@ impl TieredBackend { bound.bead().eager_install(entry); } } - // A body with a loop is kept for the resume points an - // interpreted frame of it may still ask for; without one, or - // a tier above, the compile was its last reader, and its - // cell is let go for an empty one. - if !keeps_bodies && osr::find_loop_headers(&body).is_empty() { - optimized_bodies.replace(*func_id, BodyCell::default()); - } publish(entry as usize, false); // A promotion held for the quick baseline goes ahead now // that its optimised body exists. Taken after the publish, @@ -3288,6 +3580,15 @@ impl TieredBackend { if keeps_bodies && static_hot.contains(&bead_id) { promote_static_hot(bead_id, &body); } + // Calls from here on run the code. A frame still interpreting + // holds the body it runs, and finds it by that, so the interp + // body goes. A ladder that ends here had its last reader of + // the optimised body in this compile, but for a caller + // optimised later that inlines it. + release_interp_body(&quick_baseline.interp_bodies, *func_id); + if !keeps_bodies && !kept_for_inlining(&body) { + optimized_bodies.release(*func_id, &body); + } // `ZYNTAX_TRACE_LAZY=1` names each first-call compile with // the time it took, the wait for the backend included, what // kind of body it was, and the thread that did it; and each @@ -3352,7 +3653,6 @@ impl TieredBackend { .stack_size(16 << 20) .spawn(move || { ON_WARM_UP.with(|on| on.set(true)); - let mut idle_since: Option = None; loop { if stop.load(std::sync::atomic::Ordering::Acquire) { return; @@ -3360,11 +3660,9 @@ impl TieredBackend { queue.busy.store(true, std::sync::atomic::Ordering::Release); match queue.take() { Some(Job::Promote(bead_id, site)) => { - idle_since = None; queue.run_promotions(bead_id, site); } Some(Job::Compile(bead_id, count, at)) => { - idle_since = None; if queue.still_wanted(bead_id, count, at) { compile(bead_id, false); } else if std::env::var_os("ZYNTAX_TRACE_LAZY").is_some() { @@ -3374,13 +3672,10 @@ impl TieredBackend { } } None => { - // A quiet spell: the scratch modules the + // Nothing waits: the scratch modules the // program's functions are optimised in are // let go, and made again if asked. - let since = *idle_since.get_or_insert_with(web_time::Instant::now); - if since.elapsed() > std::time::Duration::from_millis(250) { - scratch_shared.lock().unwrap().clear(); - } + scratch_shared.release_if_unused(); queue .busy .store(false, std::sync::atomic::Ordering::Release); @@ -3547,10 +3842,19 @@ impl TieredBackend { .iter() .map(|(bead, (id, _, _, module, _))| (*bead, (*id, Arc::clone(module)))) .collect(); + let bodies = Arc::clone(&self.optimized_bodies); Box::new(move |bead_id: u64, body: &Arc| { let Some((func_id, module_arc)) = modules.get(&bead_id) else { return; }; + // The top of the ladder: a loop-free body has no + // frame left to ask anything of it, but for a + // caller optimised later that inlines it. A body + // with a loop is kept for the resume points a frame + // of the baseline may still ask for. + if osr::find_loop_headers(body).is_empty() && !kept_for_inlining(body) { + bodies.release(*func_id, body); + } let sites = late.lock().unwrap().installed(bead_id); if sites.is_empty() { return; @@ -3698,7 +4002,7 @@ impl TieredBackend { .lock() .unwrap() .get(&func_id) - .cloned() + .and_then(FrameBodies::interp) .filter(|f| osr::body_tag(f) == body_tag) .map(|f| (f, false)) .or_else(|| { @@ -3732,7 +4036,9 @@ impl TieredBackend { &module_arc, ordinal, &|f| optimize(bead_id, f, optimized), - &outlined_regions, + // Kept for the tier above, when there + // is one. + optimizing.then_some(&*outlined_regions), ); if !attempted { outlined_at.lock().unwrap().remove(&key); @@ -3750,9 +4056,15 @@ impl TieredBackend { ); } // A frame on neither runs the tier body, which is - // made by now. + // made by now, and which the frame holds. (None, _) => { - if let Some(body) = body_now() { + let held = interp_bodies + .lock() + .unwrap() + .get(&func_id) + .and_then(|b| b.tier_weak.upgrade()) + .filter(|f| osr::body_tag(f) == body_tag); + if let Some(body) = held.or_else(body_now) { publish_baseline_resume_points( &cranelift, func_id, @@ -3765,6 +4077,14 @@ impl TieredBackend { } } } + // A bead with its baseline, and with the tier above's answer + // in, has nothing left to ask of its body, which may have + // been let go. + if bound.bead().compiled().is_some() + && (!optimizing || outcomes.lock().unwrap().contains_key(&(bead_id, tier_idx))) + { + return true; + } let Some(func_arc) = body_now() else { return false; }; @@ -4504,8 +4824,9 @@ fn ensure_baseline( /// `optimize` and compiled as a function of its own (see /// [`osr::outline`]); the site gets the adapter that enters it. Where the /// region cannot stand alone, the baseline's own resume point instead. -/// The region goes into `regions` under the bead before the site is -/// published, for the tier above to give the frame resume points of its +/// The region goes into `regions`, given when there is a tier above, +/// under the bead before the site is published, for that tier to give +/// the frame resume points of its /// own (see [`promote_outlined_region`]): the frame may ask for that /// promotion from the region the moment it is in it. Whether the region /// was attempted: false when there was no such loop or no frame waits. @@ -4518,7 +4839,7 @@ fn publish_outlined_resume_point( module_arc: &Arc, asked_at: u64, optimize: &(dyn Fn(HirFunction) -> HirFunction + Send + Sync), - regions: &Mutex)>>>, + regions: Option<&Mutex)>>>>, ) -> bool { let def = ZyntaxFunctionDef { id: func_id, @@ -4571,12 +4892,14 @@ fn publish_outlined_resume_point( publish_baseline_resume_points(cranelift, func_id, bead_id, func_arc, module_arc, asked_at); return true; }; - regions - .lock() - .unwrap() - .entry(bead_id) - .or_default() - .push((region_id, Arc::new(region))); + if let Some(regions) = regions { + regions + .lock() + .unwrap() + .entry(bead_id) + .or_default() + .push((region_id, Arc::new(region))); + } for (site, code) in sites { let code = code as *mut (); if !code.is_null() && osr::helper_for(bead_id, site).is_null() { diff --git a/crates/compiler/tests/body_retention.rs b/crates/compiler/tests/body_retention.rs new file mode 100644 index 00000000..7960e9da --- /dev/null +++ b/crates/compiler/tests/body_retention.rs @@ -0,0 +1,303 @@ +//! What the tiers keep of a lazy function's HIR. Its interp body and its +//! optimised body go once it has native code and no interpreted frame +//! runs them; a frame still running one finds it for as long as it +//! runs; a large body whose first frame returns before there is native +//! code goes then; and a body asked for again after it went is made +//! again, once. + +#![cfg(feature = "cranelift-backend")] + +use std::collections::HashSet; +use std::sync::Arc; + +use indexmap::IndexMap; +use zyntax_compiler::hir::{ + BinaryOp, HirBlock, HirConstant, HirFunction, HirFunctionSignature, HirId, HirInstruction, + HirModule, HirParam, HirPhi, HirTerminator, HirType, HirValue, HirValueKind, +}; +use zyntax_compiler::opt_audit; +use zyntax_compiler::osr; +use zyntax_compiler::tiered_backend::{TieredBackend, TieredConfig}; +use zyntax_typed_ast::InternedString; + +/// Steps of the chain each body runs: two instructions a step, more than +/// any size a body is kept at. +const STEPS: usize = 300; + +fn key(k: usize) -> i32 { + (k as i32).wrapping_mul(0x9e37) ^ 0x5a5a +} + +/// The chain, as the compiled code computes it. +fn chain(mut x: i32) -> i32 { + for k in 0..STEPS { + x = x.wrapping_mul(31) ^ key(k); + } + x +} + +struct Builder { + values: IndexMap, +} + +impl Builder { + fn value(&mut self, ty: HirType, kind: HirValueKind) -> HirId { + let id = HirId::new(); + self.values.insert( + id, + HirValue { + id, + ty, + kind, + uses: Default::default(), + span: None, + }, + ); + id + } + + fn constant(&mut self, v: i32) -> HirId { + self.value(HirType::I32, HirValueKind::Constant(HirConstant::I32(v))) + } + + fn binary( + &mut self, + out: &mut Vec, + op: BinaryOp, + left: HirId, + right: HirId, + ) -> HirId { + let ty = if matches!(op, BinaryOp::Lt) { + HirType::Bool + } else { + HirType::I32 + }; + let result = self.value(ty.clone(), HirValueKind::Instruction); + out.push(HirInstruction::Binary { + op, + result, + ty, + left, + right, + }); + result + } + + /// `x * 31 ^ key(k)` for each step, into `out`. + fn chain(&mut self, out: &mut Vec, mut x: HirId) -> HirId { + let thirty_one = self.constant(31); + for k in 0..STEPS { + let key = self.constant(key(k)); + let m = self.binary(out, BinaryOp::Mul, x, thirty_one); + x = self.binary(out, BinaryOp::Xor, m, key); + } + x + } +} + +fn block( + id: HirId, + phis: Vec, + instructions: Vec, + terminator: HirTerminator, +) -> HirBlock { + HirBlock { + id, + label: None, + phis, + instructions, + terminator, + dominance_frontier: Default::default(), + predecessors: vec![], + successors: vec![], + } +} + +fn function(name: &str, param: HirId, b: Builder, blocks: Vec) -> HirFunction { + let mut f = HirFunction::new( + InternedString::new_global(name), + HirFunctionSignature { + params: vec![HirParam { + id: param, + name: InternedString::new_global("n"), + ty: HirType::I32, + attributes: Default::default(), + ownership: Default::default(), + }], + returns: vec![HirType::I32], + type_params: vec![], + const_params: vec![], + lifetime_params: vec![], + is_variadic: false, + is_async: false, + is_fiber: false, + effects: vec![], + is_pure: true, + }, + ); + f.entry_block = blocks[0].id; + f.blocks = blocks.into_iter().map(|b| (b.id, b)).collect(); + f.values = b.values; + // Left for its first call, as the runtime leaves a program's + // functions. + f.attributes.optimized = true; + f.attributes.deferred = true; + f +} + +/// `looped(n)`: `x = n`, then `n` times `x = chain(x)`; returns `x`. +fn looped() -> HirFunction { + let mut b = Builder { + values: IndexMap::new(), + }; + let n = b.value(HirType::I32, HirValueKind::Parameter(0)); + let zero = b.constant(0); + let one = b.constant(1); + let (entry, header, body, exit) = (HirId::new(), HirId::new(), HirId::new(), HirId::new()); + let i = b.value(HirType::I32, HirValueKind::Instruction); + let x = b.value(HirType::I32, HirValueKind::Instruction); + let mut test = Vec::new(); + let cmp = b.binary(&mut test, BinaryOp::Lt, i, n); + let mut step = Vec::new(); + let x_next = b.chain(&mut step, x); + let i_next = b.binary(&mut step, BinaryOp::Add, i, one); + let blocks = vec![ + block( + entry, + vec![], + vec![], + HirTerminator::Branch { target: header }, + ), + block( + header, + vec![ + HirPhi { + result: i, + ty: HirType::I32, + incoming: vec![(zero, entry), (i_next, body)], + }, + HirPhi { + result: x, + ty: HirType::I32, + incoming: vec![(n, entry), (x_next, body)], + }, + ], + test, + HirTerminator::CondBranch { + condition: cmp, + true_target: body, + false_target: exit, + }, + ), + block(body, vec![], step, HirTerminator::Branch { target: header }), + block( + exit, + vec![], + vec![], + HirTerminator::Return { values: vec![x] }, + ), + ]; + function("looped", n, b, blocks) +} + +/// `straight(n)`: `chain(n)`, with no loop. +fn straight() -> HirFunction { + let mut b = Builder { + values: IndexMap::new(), + }; + let n = b.value(HirType::I32, HirValueKind::Parameter(0)); + let entry = HirId::new(); + let mut out = Vec::new(); + let x = b.chain(&mut out, n); + let blocks = vec![block( + entry, + vec![], + out, + HirTerminator::Return { values: vec![x] }, + )]; + function("straight", n, b, blocks) +} + +fn call(entry: *const u8, n: i32) -> i32 { + // SAFETY: both functions are `fn(i32) -> i32`. + let f: extern "C" fn(i32) -> i32 = unsafe { std::mem::transmute(entry) }; + f(n) +} + +/// One test: the lazy compiler a backend installs is the process's. +#[test] +fn bodies_go_once_nothing_can_ask_for_them() { + opt_audit::enable(); + let (looped, straight) = (looped(), straight()); + let (a, b) = (looped.id, straight.id); + let mut module = HirModule::new(InternedString::new_global("body_retention")); + module.functions.insert(a, looped); + module.functions.insert(b, straight); + let mut backend = TieredBackend::new(TieredConfig::default()).expect("tiered backend"); + backend.set_emit_osr_probes(true); + backend + .compile_module_lazily(module, None, HashSet::from([a, b]), HashSet::new(), false) + .expect("module compiles"); + let (_, entry, bead) = backend.interpreter_bridge(); + let Some(bead_a) = bead(a) else { + // OSR off in this environment: no bead to ask for bodies by. + return; + }; + let mut source = backend.interpreter_body_source(); + let mut frame_exit = backend.interpreter_frame_exit_hook(); + + // A frame of `looped` starts in the interpreter on its interp body, + // and holds it while it runs. + let frame = source(a).expect("an interp body"); + let frame_tag = osr::body_tag(&frame); + // The optimised body, made before the first compile, which reads it. + let optimized = Arc::downgrade(&osr::lazy_optimized_body(bead_a).expect("an optimised body")); + assert_eq!(opt_audit::pipeline_runs(a), 1); + + // The first call compiles it: native code from here on. + assert_eq!(call(entry(a).expect("a stub"), 3), chain(chain(chain(3)))); + assert!( + optimized.upgrade().is_none(), + "the optimised body outlived the compile that was its last reader" + ); + // The frame still running finds its body by the tag it asks at. + let found = source(a).expect("the frame's body"); + assert!(Arc::ptr_eq(&found, &frame)); + assert_eq!(osr::body_tag(&found), frame_tag); + drop(found); + // Its frame returns: nothing holds the interp body any more. + let gone = Arc::downgrade(&frame); + drop(frame); + assert!( + gone.upgrade().is_none(), + "the interp body outlived its last frame" + ); + + // Asked for again, each body is made again: the optimised one once + // in its new cell, from the scratch while one still holds it. + let remade = source(a).expect("an interp body made again"); + assert_eq!(opt_audit::pipeline_runs(a), 1); + drop(remade); + let again = osr::lazy_optimized_body(bead_a).expect("an optimised body made again"); + let runs = opt_audit::pipeline_runs(a); + assert!(runs <= 2, "the pipeline ran {runs} times"); + let same = osr::lazy_optimized_body(bead_a).expect("the optimised body"); + assert!(Arc::ptr_eq(&again, &same)); + assert_eq!(opt_audit::pipeline_runs(a), runs); + assert_eq!(call(entry(a).expect("code"), 2), chain(chain(2))); + drop((again, same)); + + // `straight` runs once in the interpreter: a large body whose first + // frame returns with no native code for it goes then, once. + let body = source(b).expect("an interp body"); + let gone = Arc::downgrade(&body); + drop(body); + assert!(frame_exit(b), "a large body run once was kept"); + assert!(gone.upgrade().is_none()); + let body = source(b).expect("an interp body made again"); + let kept = Arc::downgrade(&body); + drop(body); + assert!(!frame_exit(b), "a body made again was let go again"); + assert!(kept.upgrade().is_some()); + assert_eq!(call(entry(b).expect("a stub"), 5), chain(5)); +} diff --git a/crates/zyntax_embed/src/runtime/tiered.rs b/crates/zyntax_embed/src/runtime/tiered.rs index 34075e78..327a9f71 100644 --- a/crates/zyntax_embed/src/runtime/tiered.rs +++ b/crates/zyntax_embed/src/runtime/tiered.rs @@ -724,6 +724,7 @@ impl TieredRuntime { } } interp.set_body_source(self.backend.interpreter_body_source()); + interp.set_frame_exit_hook(self.backend.interpreter_frame_exit_hook()); interp.set_address_source(self.backend.interpreter_address_source()); let (thunk, entry, bead) = self.backend.interpreter_bridge(); interp.set_native_bridge(thunk, entry, bead);