diff --git a/crates/compiler/src/beadie_adapter.rs b/crates/compiler/src/beadie_adapter.rs index 52bba0e5..7f4da7ea 100644 --- a/crates/compiler/src/beadie_adapter.rs +++ b/crates/compiler/src/beadie_adapter.rs @@ -32,7 +32,9 @@ use crate::hir::{HirFunction, HirId, HirModule}; #[derive(Clone)] pub struct ZyntaxFunctionDef { pub id: HirId, - pub function: HirFunction, + /// Shared with whatever handed it over: a tier compiles the body it + /// is given, not a copy made on the way. + pub function: std::sync::Arc, /// Module-level effect, handler, global, and callee context required when /// a single hot function is recompiled outside the initial bulk pass. pub module: std::sync::Arc, @@ -426,6 +428,7 @@ mod llvm_impl { if sites.is_empty() { return Vec::new(); } + crate::opt_audit::note_llvm_body(def.id, &def.function); self.with_lock(|backend| { backend.set_compile_tier(def.tier); backend.set_module_context(std::sync::Arc::clone(&def.module)); @@ -470,6 +473,7 @@ mod llvm_impl { // Resume points where a frame can take one: the sites frames // asked at, and an outlined region's own header. let sites = crate::osr::wanted_resume_points(def.bead_id, &def.function); + crate::opt_audit::note_llvm_body(def.id, &def.function); self.with_lock(|backend| { backend.set_compile_tier(tier); backend.set_module_context(std::sync::Arc::clone(&def.module)); diff --git a/crates/compiler/src/hir_interp.rs b/crates/compiler/src/hir_interp.rs index 2e940171..9af2c41e 100644 --- a/crates/compiler/src/hir_interp.rs +++ b/crates/compiler/src/hir_interp.rs @@ -2674,6 +2674,15 @@ impl std::error::Error for InterpError {} // Interpreter (dispatch loop) // ───────────────────────────────────────────────────────────────────────────── +/// The body a function's bytecode was made from, when the body source +/// gave it. Held weakly: the runtime decides how long it lives, and a +/// running frame of a function with loops holds it besides. +struct SourceBody { + body: std::sync::Weak, + /// The bytecode has loop headers a frame can ask resume points at. + loops: bool, +} + pub struct HirInterpreter { symbols: HashMap, pub profile: IdMap, @@ -2697,6 +2706,13 @@ pub struct HirInterpreter { /// Where a function's body comes from when the runtime keeps one /// apart from the module's: see [`Self::set_body_source`]. body_source: Option Option> + Send>>, + /// The body each function's bytecode was made from, for those the + /// body source gave: see [`SourceBody`]. + source_bodies: IdMap, + /// Told when the first frame run from a body the body source gave + /// returns; see [`Self::set_frame_exit_hook`]. + #[allow(clippy::type_complexity)] + frame_exit_hook: Option bool + Send>>, /// Compiles, or finds, the thunk that calls native code of a given /// shape: `fn(target, words, out)`. Installed by a runtime with a /// native tier; without one, calls into native code use the fixed @@ -2850,6 +2866,8 @@ impl HirInterpreter { uncompilable: IdMap::default(), tick_callbacks: IdMap::default(), body_source: None, + source_bodies: IdMap::default(), + frame_exit_hook: None, thunk_source: None, entry_source: None, bead_source: None, @@ -2980,6 +2998,14 @@ impl HirInterpreter { self.body_source = Some(source); } + /// Called with a function's id when the first frame run from a body + /// the body source gave returns. Answers whether the runtime let that + /// body go; the bytecode made from it goes too then, and the next + /// call asks the body source again. + pub fn set_frame_exit_hook(&mut self, hook: Box bool + Send>) { + self.frame_exit_hook = Some(hook); + } + /// Drop the bridge and every tick callback. They hold the native /// tiers' backends, which the runtime shuts down after them. pub fn clear_native_bridge(&mut self) { @@ -2987,6 +3013,7 @@ impl HirInterpreter { self.thunk_source = None; self.entry_source = None; self.bead_source = None; + self.frame_exit_hook = None; } /// Install the bridge to a native tier: `thunk` compiles the caller @@ -3520,6 +3547,26 @@ impl HirInterpreter { } } + // A frame of a function with loops holds the body its bytecode + // was made from, so a request for resume points it makes finds + // that body while it runs. Bytecode whose body nobody holds any + // more is made again from the body the source gives now. + let mut held: Option> = None; + let mut stale = false; + if let Some(source) = self.source_bodies.get(&func_id) + && source.loops + { + held = source.body.upgrade(); + stale = held.is_none(); + } + // Whether this frame made the bytecode from a body the source + // gave: the runtime is told when it returns. + let mut first = false; + if stale { + self.source_bodies.remove(&func_id); + self.cache.remove(&func_id); + } + // Compile-on-first-use, and refuse-once. What the interpreter // cannot run, native code runs, when there is native code. if !self.cache.contains_key(&func_id) { @@ -3551,6 +3598,21 @@ impl HirInterpreter { let taken = self.is_address_taken(module, func_id); match compile_function_with(module, &mut self.memory, func, taken) { Ok(cf) => { + // A recursive call compiles the body again while + // the frame that made the entry runs. + if let Some(body) = &shared { + let loops = !cf.osr_sites.is_empty(); + self.source_bodies.entry(func_id).or_insert_with(|| { + first = true; + SourceBody { + body: std::sync::Arc::downgrade(body), + loops, + } + }); + if loops { + held = Some(std::sync::Arc::clone(body)); + } + } self.cache.insert(func_id, cf); } Err(InterpError::UnsupportedInstruction(why)) => { @@ -3574,8 +3636,18 @@ impl HirInterpreter { // calls. The map ownership returns at the end. let cf = self.cache.remove(&func_id).unwrap(); let result = self.run(module, &cf, args, func_id, dest); - // Put the (immutable) compiled function back. - self.cache.insert(func_id, cf); + drop(held); + let released = first + && self + .frame_exit_hook + .as_mut() + .is_some_and(|hook| hook(func_id)); + if released { + self.source_bodies.remove(&func_id); + } else { + // Put the (immutable) compiled function back. + self.cache.insert(func_id, cf); + } result } diff --git a/crates/compiler/src/inline.rs b/crates/compiler/src/inline.rs index 8bbce6c1..bdbaea3e 100644 --- a/crates/compiler/src/inline.rs +++ b/crates/compiler/src/inline.rs @@ -181,6 +181,12 @@ fn count_insts(function: &HirFunction) -> usize { function.blocks.values().map(|b| b.instructions.len()).sum() } +/// Whether `function` is small enough that some call to it may be +/// inlined. +pub(crate) fn may_inline(function: &HirFunction) -> bool { + count_insts(function) <= MAX_INLINE_INSTS_MULTI_BLOCK +} + /// Inline every eligible direct call within `module`. Iterates a /// fixed-point: inlining one call can expose a now-eligible callee /// (when the now-inlined body's prior nested call shape was a diff --git a/crates/compiler/src/lib.rs b/crates/compiler/src/lib.rs index 5f28f7a5..d7c93fe8 100644 --- a/crates/compiler/src/lib.rs +++ b/crates/compiler/src/lib.rs @@ -72,6 +72,8 @@ pub mod memory_optimization; // Memory-aware optimizations pub mod memory_pass; pub mod monomorphize; pub mod move_insert; // Owning parameters become `Move` the borrow check can see +#[doc(hidden)] +pub mod opt_audit; // Per-function counts of optimiser runs, for tests pub mod optimization; pub mod parallel_dispatch; // A loop with independent iterations becomes a band dispatch pub mod parallel_safe; // Which counted loops have independent iterations @@ -1970,6 +1972,7 @@ fn run_interp_safe_opts_with( { return stats; } + opt_audit::note_pipeline(module); // Alloca → Malloc promotion runs ONCE up front, before the // fixed-point sweep. Two reasons it goes here: diff --git a/crates/compiler/src/llvm_jit_backend.rs b/crates/compiler/src/llvm_jit_backend.rs index be5d8044..e58f1379 100644 --- a/crates/compiler/src/llvm_jit_backend.rs +++ b/crates/compiler/src/llvm_jit_backend.rs @@ -107,6 +107,12 @@ pub struct LLVMJitBackend<'ctx> { /// [`Self::set_cross_tier_links`]. cross_tier_key: Option, global_resolver: Option Option + Send + Sync>>, + /// The body each tier compiles of a module function, when it is not + /// the module context's own; a callee compiled alongside a promoted + /// function takes it. See [`Self::set_body_source`]. + #[allow(clippy::type_complexity)] + body_source: + Option Option> + Send + Sync>>, /// What the module being compiled reaches across tiers, handed to /// the lowering. pending_cross_tier: HashMap, @@ -207,6 +213,7 @@ impl<'ctx> LLVMJitBackend<'ctx> { address_taken: std::collections::HashSet::new(), cross_tier_key: None, global_resolver: None, + body_source: None, pending_cross_tier: HashMap::new(), pending_shared_globals: HashMap::new(), pending_entry_abi: None, @@ -1084,6 +1091,17 @@ impl<'ctx> LLVMJitBackend<'ctx> { self.global_resolver = Some(globals); } + /// Where a callee compiled alongside a promoted function takes its + /// body from: `bodies` answers with the body every tier compiles of + /// a function, or `None` for one whose module-context body is that. + #[allow(clippy::type_complexity)] + pub fn set_body_source( + &mut self, + bodies: std::sync::Arc Option> + Send + Sync>, + ) { + self.body_source = Some(bodies); + } + /// Compile a single function, together with everything it calls. /// /// The callees come from the module context, so they are recompiled at @@ -1169,7 +1187,11 @@ impl<'ctx> LLVMJitBackend<'ctx> { if callee == id { continue; } - if let Some(f) = ctx.functions.get(&callee) { + let tier_body = self.body_source.as_ref().and_then(|body| body(callee)); + if let Some(f) = tier_body { + crate::opt_audit::note_llvm_body(callee, &f); + functions.insert(callee, (*f).clone()); + } else if let Some(f) = ctx.functions.get(&callee) { functions.insert(callee, f.clone()); } } diff --git a/crates/compiler/src/opt_audit.rs b/crates/compiler/src/opt_audit.rs new file mode 100644 index 00000000..6aca9276 --- /dev/null +++ b/crates/compiler/src/opt_audit.rs @@ -0,0 +1,102 @@ +//! Per-function counts of the optimiser's work, for tests that hold +//! each body to one run of the pipeline and every compile above the +//! baseline to that body. Off until [`enable`]; while off, each hook +//! costs one relaxed load. + +use std::collections::HashMap; +use std::sync::atomic::{AtomicBool, Ordering}; +use std::sync::{Arc, Mutex, OnceLock}; + +use crate::hir::{HirFunction, HirId, HirModule}; + +static ON: AtomicBool = AtomicBool::new(false); + +#[derive(Default)] +struct Counts { + pipeline: HashMap, + finishing: HashMap, + llvm: HashMap>, +} + +fn counts() -> &'static Mutex { + static C: OnceLock> = OnceLock::new(); + C.get_or_init(|| Mutex::new(Counts::default())) +} + +/// Start counting, for the rest of the process. +pub fn enable() { + ON.store(true, Ordering::Relaxed); +} + +fn on() -> bool { + ON.load(Ordering::Relaxed) +} + +/// A run of the whole pipeline over `module`: one for each function it +/// optimises, those not through it already. +pub(crate) fn note_pipeline(module: &HirModule) { + if !on() { + return; + } + let mut c = counts().lock().unwrap(); + for (id, f) in &module.functions { + if !f.attributes.optimized && !f.is_external { + *c.pipeline.entry(*id).or_default() += 1; + } + } +} + +/// The incremental passes run over `id`, a body that arrived optimised. +pub(crate) fn note_finishing(id: HirId) { + if on() { + *counts().lock().unwrap().finishing.entry(id).or_default() += 1; + } +} + +/// `body` handed to the LLVM tier as the HIR of `id`. +#[cfg_attr(not(feature = "llvm-backend"), allow(dead_code))] +pub(crate) fn note_llvm_body(id: HirId, body: &Arc) { + if on() { + counts() + .lock() + .unwrap() + .llvm + .entry(id) + .or_default() + .push(Arc::as_ptr(body) as usize); + } +} + +/// How many runs of the whole pipeline optimised `id`. +pub fn pipeline_runs(id: HirId) -> usize { + counts() + .lock() + .unwrap() + .pipeline + .get(&id) + .copied() + .unwrap_or(0) +} + +/// How many times the incremental passes ran over `id`. +pub fn finishing_runs(id: HirId) -> usize { + counts() + .lock() + .unwrap() + .finishing + .get(&id) + .copied() + .unwrap_or(0) +} + +/// The address of each body the LLVM tier was handed for `id`, in the +/// order it was handed them. +pub fn llvm_bodies(id: HirId) -> Vec { + counts() + .lock() + .unwrap() + .llvm + .get(&id) + .cloned() + .unwrap_or_default() +} diff --git a/crates/compiler/src/tiered_backend.rs b/crates/compiler/src/tiered_backend.rs index 6eed5572..78ac755d 100644 --- a/crates/compiler/src/tiered_backend.rs +++ b/crates/compiler/src/tiered_backend.rs @@ -241,6 +241,397 @@ struct FunctionEntry { bead_id: u64, } +/// A function's optimised body; see [`OptimizedBodies`]. +type BodyCell = Arc>>; + +/// Each lazy function's optimised body, beside the lowered one its +/// module holds. A cell is filled at most once per load or reload, by +/// the first caller to reach it while any other waits, and every compile +/// of the function at any tier reads its HIR from its cell or from the +/// body a reload swapped in. A cell is replaced whole, by a reload or to +/// let its body go, and never refilled in place. +#[derive(Default)] +struct OptimizedBodies(RwLock>); + +impl OptimizedBodies { + fn cell(&self, id: HirId) -> Option { + self.0.read().unwrap().get(&id).cloned() + } + + /// The body in `id`'s cell, if it is filled. + fn get(&self, id: HirId) -> Option> { + self.0.read().unwrap().get(&id)?.get().cloned() + } + + /// An empty cell for each of `ids` that has none; a cell already + /// there is kept, filled or not. + fn add(&self, ids: impl IntoIterator) { + let mut cells = self.0.write().unwrap(); + for id in ids { + cells.entry(id).or_default(); + } + } + + /// `cell` in place of `id`'s, handing back the one it replaced. + fn replace(&self, id: HirId, cell: BodyCell) -> Option { + self.0.write().unwrap().insert(id, cell) + } + + /// Put back what [`Self::replace`] handed back. + fn restore(&self, id: HirId, cell: Option) { + let mut cells = self.0.write().unwrap(); + match cell { + Some(cell) => cells.insert(id, cell), + None => cells.remove(&id), + }; + } + + fn filled(body: Arc) -> BodyCell { + let cell = std::sync::OnceLock::new(); + let _ = cell.set(body); + Arc::new(cell) + } +} + +/// The HIR every tier compiles of `func_id`: the body a reload swapped +/// in, else its optimised body, made now if a lazy function has none +/// yet, else the module's, which an eager function was optimised in at +/// load. +fn tier_body( + func_id: HirId, + bead_id: u64, + swapped: Option<&Arc>, + bodies: &OptimizedBodies, + lazy: bool, + module: &HirModule, +) -> Option> { + if let Some(f) = swapped { + return Some(Arc::clone(f)); + } + if let Some(f) = bodies.get(func_id) { + return Some(f); + } + if lazy && let Some(f) = osr::lazy_optimized_body(bead_id) { + return Some(f); + } + module.functions.get(&func_id).map(|f| Arc::new(f.clone())) +} + +impl OptimizedBodies { + /// An empty cell in place of `id`'s, if that cell still holds `body`: + /// a cell a reload filled since keeps the reload's body. The next + /// request for the body makes it again, once, in the new cell. + fn release(&self, id: HirId, body: &Arc) -> bool { + let mut cells = self.0.write().unwrap(); + let holds = cells + .get(&id) + .and_then(|cell| cell.get()) + .is_some_and(|held| Arc::ptr_eq(held, body)); + if holds { + cells.insert(id, BodyCell::default()); + } + holds + } +} + +/// Whether the optimised body of a function is kept once every tier +/// that compiles it has: a body small enough to be inlined is what a +/// caller optimised later inlines. +fn kept_for_inlining(body: &HirFunction) -> bool { + crate::inline::may_inline(body) +} + +/// A body at least this many instructions long whose first interpreted +/// frame returns before the function has native code is let go then: +/// such a body is mostly run once, and costs more kept than made again. +const RUN_ONCE_INSTRUCTIONS: usize = 512; + +/// What the interpreter was given of one lazy function. +#[derive(Default)] +struct FrameBodies { + /// The body made for the interpreter's first runs, while it is kept: + /// until the function has native code, or, for a large body, until + /// its first frame returns. A frame running it holds it besides. + interp: Option>, + /// The body `interp` held last, while anyone holds it. + interp_weak: std::sync::Weak, + /// The tier body last handed to the interpreter, while a frame + /// running it holds it. + tier_weak: std::sync::Weak, + /// An interp body was made; one made after it was let go is kept + /// until the function has native code. + made: bool, + remade: bool, +} + +impl FrameBodies { + /// The interp body, while anyone holds it. + fn interp(&self) -> Option> { + self.interp.clone().or_else(|| self.interp_weak.upgrade()) + } +} + +/// The interp bodies of the lazy functions, by function id. +type InterpBodies = Mutex>; + +/// The interp body of `func_id`: the one held, else one made now. +#[allow(clippy::type_complexity)] +fn interp_body( + bodies: &InterpBodies, + func_id: HirId, + bead_id: u64, + make: Option<&Arc Option> + Send + Sync>>, +) -> Option> { + if let Some(body) = bodies + .lock() + .unwrap() + .get(&func_id) + .and_then(FrameBodies::interp) + { + return Some(body); + } + let body = make?(bead_id)?; + let mut bodies = bodies.lock().unwrap(); + let entry = bodies.entry(func_id).or_default(); + if let Some(body) = entry.interp() { + return Some(body); + } + entry.remade |= entry.made; + entry.made = true; + entry.interp_weak = Arc::downgrade(&body); + entry.interp = Some(Arc::clone(&body)); + Some(body) +} + +/// Let `func_id`'s interp body go; a frame running it keeps it alive +/// for as long as it runs. +fn release_interp_body(bodies: &InterpBodies, func_id: HirId) { + if let Some(entry) = bodies.lock().unwrap().get_mut(&func_id) { + entry.interp = None; + } +} + +fn instruction_count(f: &HirFunction) -> usize { + f.blocks.values().map(|b| b.instructions.len()).sum() +} + +/// The functions `f` calls directly. +fn direct_callees(f: &HirFunction) -> impl Iterator + '_ { + use crate::hir::{HirCallable, HirInstruction, HirTerminator}; + f.blocks.values().flat_map(|b| { + let calls = b.instructions.iter().filter_map(|inst| match inst { + HirInstruction::Call { + callee: HirCallable::Function(target), + .. + } => Some(*target), + _ => None, + }); + let invoke = match &b.terminator { + HirTerminator::Invoke { + callee: HirCallable::Function(target), + .. + } => Some(*target), + _ => None, + }; + calls.chain(invoke) + }) +} + +/// A module the program's lazy functions are optimised in, one at a +/// time: the module's declarations and extern functions, and of its +/// defined functions those an optimised body reaches through direct +/// calls, each copied in when first reached. Every function in it has +/// its direct callees in it too, which is all any pass reads of a +/// function other than the one it optimises; what the passes know of +/// the module as a whole comes from [`Facts`]. +struct Scratch { + module: HirModule, + /// The functions of `module` through the pipeline in it. + done: HashSet, +} + +impl Scratch { + /// The scratch of `module` in `slots`, made if none is. + fn of<'a>(slots: &'a mut HashMap, module: &Arc) -> &'a mut Scratch { + let key = Arc::as_ptr(module) as usize; + slots.entry(key).or_insert_with(|| { + let started = web_time::Instant::now(); + let externs = module + .functions + .iter() + .filter(|(_, f)| f.is_external) + .map(|(id, f)| { + let mut f = f.clone(); + f.attributes.optimized = true; + f.attributes.deferred = true; + (*id, f) + }) + .collect(); + let scratch = HirModule { + id: module.id, + name: module.name, + functions: externs, + globals: module.globals.clone(), + types: module.types.clone(), + imports: module.imports.clone(), + exports: module.exports.clone(), + version: module.version, + dependencies: module.dependencies.clone(), + effects: module.effects.clone(), + handlers: module.handlers.clone(), + automatic_release: module.automatic_release, + }; + if std::env::var_os("ZYNTAX_TRACE_LAZY").is_some() { + eprintln!( + "[lazy] scratch module made in {:.2} ms on {}", + started.elapsed().as_secs_f64() * 1e3, + std::thread::current().name().unwrap_or("?") + ); + } + Scratch { + module: scratch, + done: HashSet::new(), + } + }) + } + + /// Copy in `roots` and every function they reach through direct + /// calls that is not in yet, each marked through the pipeline and + /// deferred, so a pass that walks the module touches the one being + /// optimised alone. A program function whose cell is filled goes in + /// as its cell holds it, so a body optimised later inlines it as + /// every tier runs it; one that arrived optimised stays as it + /// arrived, box readers and all, as the release pass reads it. + fn reach( + &mut self, + roots: impl IntoIterator, + module: &HirModule, + bodies: &OptimizedBodies, + program: &HashSet, + ) { + let mut work: Vec = roots.into_iter().collect(); + while let Some(id) = work.pop() { + if self.module.functions.contains_key(&id) { + continue; + } + let Some(lowered) = module.functions.get(&id) else { + continue; + }; + let optimized = program.contains(&id).then(|| bodies.get(id)).flatten(); + let mut f = match optimized { + Some(body) => { + self.done.insert(id); + (*body).clone() + } + None => lowered.clone(), + }; + f.attributes.optimized = true; + f.attributes.deferred = true; + work.extend(direct_callees(&f)); + self.module.functions.insert(id, f); + } + } +} + +/// The scratch modules of one lazy compiler, and the optimisations under +/// way in them. They are let go once none is under way and no compile +/// is waiting, and made again as asked. +#[derive(Default)] +struct Scratches { + slots: Mutex>, + pending: std::sync::atomic::AtomicUsize, +} + +impl Scratches { + /// Note an optimisation about to use a scratch, until the returned + /// guard drops; `quiet` answers whether no compile is waiting. + fn enter<'a>(&'a self, quiet: &'a (dyn Fn() -> bool + Send + Sync)) -> ScratchUse<'a> { + self.pending + .fetch_add(1, std::sync::atomic::Ordering::SeqCst); + ScratchUse { + scratches: self, + quiet, + } + } + + /// Let every scratch go, once the optimisation in one, if any, is + /// done. + fn clear(&self) { + let gone = std::mem::take(&mut *self.slots.lock().unwrap()); + drop(gone); + } + + /// Let every scratch go if no optimisation is under way. + fn release_if_unused(&self) { + let mut slots = self.slots.lock().unwrap(); + if self.pending.load(std::sync::atomic::Ordering::SeqCst) != 0 || slots.is_empty() { + return; + } + let gone = std::mem::take(&mut *slots); + drop(slots); + drop(gone); + } +} + +struct ScratchUse<'a> { + scratches: &'a Scratches, + quiet: &'a (dyn Fn() -> bool + Send + Sync), +} + +impl Drop for ScratchUse<'_> { + fn drop(&mut self) { + let last = self + .scratches + .pending + .fetch_sub(1, std::sync::atomic::Ordering::SeqCst) + == 1; + if last && (self.quiet)() { + self.scratches.release_if_unused(); + } + } +} + +/// What the passes know of each module as a whole, built once, by the +/// first thread that needs it: the release pass's facts, and the call +/// cycles the inliner keeps within. A linked library's functions carry +/// their facts, so the build covers the program's own. +#[derive(Default)] +struct Facts { + facts: Mutex>>, + caches: Mutex>>, +} + +impl Facts { + fn of(&self, module: &Arc) -> Arc { + let key = Arc::as_ptr(module) as usize; + let mut built = self.facts.lock().unwrap(); + Arc::clone(built.entry(key).or_insert_with(|| { + let started = web_time::Instant::now(); + let facts = Arc::new(crate::drop_insert::facts_of(module)); + if std::env::var_os("ZYNTAX_TRACE_LAZY").is_some() { + eprintln!( + "[lazy] release facts built in {:.2} ms on {}", + started.elapsed().as_secs_f64() * 1e3, + std::thread::current().name().unwrap_or("?") + ); + } + facts + })) + } + + /// The pipeline's cache for `module`, over the module as lowered: + /// inlining keeps every function's reach, so its call cycles hold + /// for the bodies optimised since. + fn cache(&self, module: &Arc) -> Arc { + let key = Arc::as_ptr(module) as usize; + if let Some(cache) = self.caches.lock().unwrap().get(&key) { + return Arc::clone(cache); + } + let cache = Arc::new(crate::OptCache::with_facts(self.of(module), module)); + Arc::clone(self.caches.lock().unwrap().entry(key).or_insert(cache)) + } +} + /// What the tier above answered a bead's promotion request with, kept /// per bead and tier so the answer is not worked out again. #[derive(Clone, Copy)] @@ -317,6 +708,15 @@ impl CompileQueue { !self.busy.load(std::sync::atomic::Ordering::Acquire) } + /// Whether the worker is in no job and no request waits for it. + fn quiet(&self) -> bool { + if !self.idle() { + return false; + } + let promote = self.promote.lock().unwrap().is_empty(); + promote && self.compile.lock().unwrap().is_empty() + } + /// Whether a compile of `bead_id` was asked for already. fn is_queued(&self, bead_id: u64) -> bool { self.queued.lock().unwrap().contains(&bead_id) @@ -491,7 +891,7 @@ type FirstCompiles = ( struct QuickBaseline { /// The program's own lazy functions; a finished one arrived optimised. lazy: HashSet, - interp_bodies: Arc>>>, + interp_bodies: Arc, #[allow(clippy::type_complexity)] make_interp_body: Option Option> + Send + Sync>>, /// What each quick baseline's call cell held when it was published: @@ -512,17 +912,12 @@ impl QuickBaseline { { return None; } - if let Some(body) = self.interp_bodies.lock().unwrap().get(&func_id) { - return Some(Arc::clone(body)); - } - let body = self.make_interp_body.as_ref()?(bead_id)?; - Some(Arc::clone( - self.interp_bodies - .lock() - .unwrap() - .entry(func_id) - .or_insert(body), - )) + interp_body( + &self.interp_bodies, + func_id, + bead_id, + self.make_interp_body.as_ref(), + ) } fn note_published(&self, bead_id: u64) { @@ -598,6 +993,11 @@ struct UndoSwap { /// nothing was published). old_published: usize, old_body: Arc, + /// The body a reload before this one swapped in, if one had. + old_function: Option>, + /// The function's optimised-body cell before the reload, if it had + /// one. + old_body_cell: Option, /// OSR resume points this reload published, to be unpublished. helper_sites: Vec<(u64, u64)>, /// Dispatch-table slots this reload patched: (slot address, value @@ -642,10 +1042,9 @@ pub struct TieredBackend { /// in their cell; the interpreter reaches their baseline the same /// way, so one compile is made of each. lazy: HashSet, - /// The body the first-call compile optimised for each function it - /// compiled, which a later tier compiles again rather than the - /// module's unoptimised one. - optimized_bodies: Arc>>>, + /// Each lazy function's optimised body, made once and compiled by + /// every tier in place of the module's unoptimised one. + optimized_bodies: Arc, /// Of the lazy functions, those that arrived optimised (a linked /// snapshot's), across every module compiled so far: the interpreter /// takes their body from the optimiser, which has little to do to @@ -657,14 +1056,19 @@ pub struct TieredBackend { /// code, and asks for the optimizing tier as soon as it has some. static_hot: HashMap, /// The optimiser for a function that is not the module's: an - /// outlined resume point, optimised in the scratch module of the - /// bead it belongs to. Installed with the lazy compiler. - optimize_extra: Option HirFunction + Send + Sync>>, + /// outlined resume point of the bead it belongs to, and whether it + /// was cut from a body through the pipeline already, which it then + /// does not go through again. Installed with the lazy compiler. + #[allow(clippy::type_complexity)] + optimize_extra: Option HirFunction + Send + Sync>>, /// The body the interpreter runs a lazy function's first calls on: /// the module's, as lowered, with its releases placed (see - /// [`Self::interpreter_body_source`]). Made at the first run and - /// kept: the sites a frame on it asks at are this body's. - interp_bodies: Arc>>>, + /// [`Self::interpreter_body_source`]). Made at the first run, and + /// found by a frame running it for as long as it runs: the sites the + /// frame asks at are this body's. + interp_bodies: Arc, + /// The current lazy compiler's scratch modules. + scratches: Arc, /// Makes an entry of `interp_bodies` for a bead. Installed with the /// lazy compiler. #[allow(clippy::type_complexity)] @@ -793,11 +1197,12 @@ impl TieredBackend { _llvm_context, functions: HashMap::new(), lazy: HashSet::new(), - optimized_bodies: Arc::new(Mutex::new(HashMap::new())), + optimized_bodies: Arc::new(OptimizedBodies::default()), finished: HashSet::new(), static_hot: HashMap::new(), optimize_extra: None, interp_bodies: Arc::new(Mutex::new(HashMap::new())), + scratches: Arc::new(Scratches::default()), make_interp_body: None, current_module: None, loaded: Vec::new(), @@ -1040,6 +1445,25 @@ impl TieredBackend { let (all_lazy, all_finished) = (self.lazy.clone(), self.finished.clone()); self.install_lazy_compiler(&all_lazy, &all_finished); } + // A promotion that compiles its callee closure takes each + // callee's tier body. + #[cfg(feature = "llvm-backend")] + if let Some(llvm) = &self.llvm { + let bodies = Arc::clone(&self.optimized_bodies); + let beads: HashMap = self + .functions + .iter() + .filter(|(id, _)| self.lazy.contains(id)) + .map(|(id, e)| (*id, e.bead_id)) + .collect(); + llvm.with_lock(|be| { + be.set_body_source(Arc::new(move |id| { + bodies + .get(id) + .or_else(|| beads.get(&id).and_then(|b| osr::lazy_optimized_body(*b))) + })) + }); + } if !joining { self.install_promotion_requester(); self.precompile_static_hot(&lazy); @@ -1519,6 +1943,8 @@ impl TieredBackend { let mut old_entry = 0usize; let mut old_body: Option> = None; + let mut old_function: Option> = None; + let mut old_body_cell: Option = None; let old_cell = crate::reload::call_target(reload_key, old_id); let old_published = osr::published_entry(bead_id); if let Some(fn_entry) = self.functions.get_mut(&old_id) { @@ -1540,7 +1966,14 @@ impl TieredBackend { { fn_entry.bound.bead().eager_install(entry_ptr as *mut ()); } - fn_entry.function = Some(Arc::new(body.clone())); + // The edit's body is what every tier compiles from now + // on, the interpreter's getter included, which reads the + // cell. + let swapped = Arc::new(body.clone()); + old_function = fn_entry.function.replace(Arc::clone(&swapped)); + old_body_cell = self + .optimized_bodies + .replace(old_id, OptimizedBodies::filled(swapped)); } // A stub whose address was kept calls the edit's code too. osr::set_published_entry(bead_id, entry_ptr); @@ -1623,6 +2056,8 @@ impl TieredBackend { old_cell, old_published, old_body: old_body.unwrap_or_else(|| Arc::new(old_fn.clone())), + old_function, + old_body_cell, helper_sites, vtable_slots, }); @@ -1724,6 +2159,8 @@ impl TieredBackend { // The promotion requester captured each function's body when it // was installed; reinstall so a later promotion compiles the // edited bodies rather than the ones it captured. + // A scratch holds the bodies it copied in before the swap. + self.scratches.clear(); self.install_promotion_requester(); // A reload that changed nothing keeps the previous undo record: @@ -1769,7 +2206,8 @@ impl TieredBackend { // restored body. fn_entry.bound.bead().reload(); } - fn_entry.function = Some(Arc::clone(&swap.old_body)); + fn_entry.function = swap.old_function.clone(); + self.optimized_bodies.restore(swap.id, swap.old_body_cell); } let cell = if swap.old_cell != 0 { swap.old_cell @@ -1805,6 +2243,7 @@ impl TieredBackend { } } + self.scratches.clear(); self.install_promotion_requester(); Ok(restored) } @@ -2401,24 +2840,23 @@ impl TieredBackend { let code = adapter.on_invoke(&bound, move |tier, bead| { let c = &*ctx; // The baseline was compiled at load; the ladder above it - // compiles anew, from the body the first-call compile - // optimised when there is one. A quick baseline has none - // until its replacement: the body is optimised now. + // compiles anew, from the function's tier body. A quick + // baseline has none until its replacement: the body is + // optimised now, once, for both. if tier == 0 { if let Some(p) = c.cranelift.with_lock(|be| be.get_function_ptr(c.func_id)) { return p as *mut (); } } - let optimized = optimized_bodies.lock().unwrap().get(&c.func_id).cloned(); - let optimized = optimized - .or_else(|| lazy.then(|| osr::lazy_optimized_body(c.bead_id)).flatten()); - let body = match (&c.swapped, optimized) { - (Some(f), _) => Arc::clone(f), - (None, Some(f)) => f, - (None, None) => match c.module.functions.get(&c.func_id) { - Some(f) => Arc::new(f.clone()), - None => return ptr::null_mut(), - }, + let Some(body) = tier_body( + c.func_id, + c.bead_id, + c.swapped.as_ref(), + &optimized_bodies, + lazy, + &c.module, + ) else { + return ptr::null_mut(); }; let entry = compile_at_tier( tier, @@ -2468,19 +2906,54 @@ impl TieredBackend { let finished = self.finished.clone(); Box::new(move |id: HirId| { let bead = beads.get(&id)?; - if let Some(body) = optimized_bodies.lock().unwrap().get(&id) { - return Some(Arc::clone(body)); - } - if finished.contains(&id) { - return osr::lazy_optimized_body(*bead); + let tier = match optimized_bodies.get(id) { + Some(body) => Some(body), + None if finished.contains(&id) => Some(osr::lazy_optimized_body(*bead)?), + None => None, + }; + if let Some(body) = tier { + interp_bodies + .lock() + .unwrap() + .entry(id) + .or_default() + .tier_weak = Arc::downgrade(&body); + return Some(body); } - if let Some(body) = interp_bodies.lock().unwrap().get(&id) { - return Some(Arc::clone(body)); + interp_body(&interp_bodies, id, *bead, make_interp_body.as_ref()) + }) + } + + /// What the interpreter tells when the first frame run from a body + /// [`Self::interpreter_body_source`] gave returns: the interp body is + /// let go once the function has native code, or then for a large + /// body the first time, which is mostly run once. Answers whether + /// it was. + pub fn interpreter_frame_exit_hook(&self) -> Box bool + Send> { + let beads: HashMap = self + .functions + .iter() + .filter(|(id, _)| self.lazy.contains(id)) + .map(|(id, e)| (*id, e.bound.clone())) + .collect(); + let interp_bodies = Arc::clone(&self.interp_bodies); + Box::new(move |id: HirId| { + let Some(bound) = beads.get(&id) else { + return false; + }; + let mut bodies = interp_bodies.lock().unwrap(); + let Some(entry) = bodies.get_mut(&id) else { + return false; + }; + let Some(body) = &entry.interp else { + return false; + }; + let native = bound.bead().compiled().is_some(); + if native || (!entry.remade && instruction_count(body) >= RUN_ONCE_INSTRUCTIONS) { + entry.interp = None; + return true; } - let body = make_interp_body.as_ref()?(*bead)?; - Some(Arc::clone( - interp_bodies.lock().unwrap().entry(id).or_insert(body), - )) + false }) } @@ -2567,8 +3040,11 @@ impl TieredBackend { None => return, }; - // Build a closure beadie can call from any tier broker thread. - let func_arc = entry.body(func_id); + // Build a closure beadie can call from any tier broker thread. The + // body is looked up when a compile needs it, not per call. + let swapped = entry.function.clone(); + let bodies = Arc::clone(&self.optimized_bodies); + let lazy = self.lazy.contains(&func_id); let module_arc = Arc::clone(&entry.module); let bead_id = entry.bead_id; let cranelift = Arc::clone(&self.cranelift); @@ -2578,6 +3054,16 @@ impl TieredBackend { let verbosity = self.config.verbosity; let closure = move |tier_idx: usize, bead: &Arc| -> *mut () { + let Some(func_arc) = tier_body( + func_id, + bead_id, + swapped.as_ref(), + &bodies, + lazy, + &module_arc, + ) else { + return ptr::null_mut(); + }; compile_at_tier( tier_idx, bead, @@ -2646,84 +3132,54 @@ impl TieredBackend { None => (HashMap::new(), HashSet::new()), }; // The program's own bodies were left as lowered. Each is - // optimised on its own at its first compile, in one scratch copy - // of the module: a body optimised earlier is what a later one + // optimised on its own at its first compile, in a scratch module + // (see [`Scratch`]): a body optimised earlier is what a later one // inlines, and what the passes know about the module as a whole // is built once, since optimising a body changes none of it. - struct Scratch { - module: HirModule, - cache: crate::OptCache, - } - // One scratch per module: a program that compiles a chunk of - // itself has bodies in more than one. - impl Scratch { - /// The scratch of `module` in `slots`, made if none is: every - /// function marked through the pipeline and deferred, so a - /// pass that walks the module touches the one being optimised - /// alone. - fn of<'a>( - slots: &'a mut HashMap, - module: &Arc, - facts: &Facts, - ) -> &'a mut Scratch { - let key = Arc::as_ptr(module) as usize; - slots.entry(key).or_insert_with(|| { - let started = web_time::Instant::now(); - let facts = facts.of(module); - let mut module: HirModule = (**module).clone(); - for f in module.functions.values_mut() { - f.attributes.optimized = true; - f.attributes.deferred = true; - } - let cache = crate::OptCache::with_facts(facts, &module); - if std::env::var_os("ZYNTAX_TRACE_LAZY").is_some() { - eprintln!( - "[lazy] scratch module made in {:.2} ms on {}", - started.elapsed().as_secs_f64() * 1e3, - std::thread::current().name().unwrap_or("?") - ); - } - Scratch { module, cache } - }) - } - } - /// What the release pass knows of each module, built once, by - /// the first thread that needs it. A linked library's functions - /// carry their facts, so the build covers the program's own. - #[derive(Default)] - struct Facts(Mutex>>); - impl Facts { - fn of(&self, module: &Arc) -> Arc { - let key = Arc::as_ptr(module) as usize; - let mut built = self.0.lock().unwrap(); - Arc::clone(built.entry(key).or_insert_with(|| { - let started = web_time::Instant::now(); - let facts = Arc::new(crate::drop_insert::facts_of(module)); - if std::env::var_os("ZYNTAX_TRACE_LAZY").is_some() { - eprintln!( - "[lazy] release facts built in {:.2} ms on {}", - started.elapsed().as_secs_f64() * 1e3, - std::thread::current().name().unwrap_or("?") - ); - } - facts - })) - } - } let facts: Arc = Arc::new(Facts::default()); - let optimized: Arc>> = Arc::new(Mutex::new(HashMap::new())); + let optimized: Arc = Arc::new(Scratches::default()); + self.scratches = Arc::clone(&optimized); let scratch_shared = Arc::clone(&optimized); + // The queue is made after the closures below, which read it + // through this cell once it exists. + let queue_for_callees: Arc>>> = Arc::new(Mutex::new(None)); + let queue_cell = Arc::clone(&queue_for_callees); + // Whether no compile waits for the worker, which would take a + // scratch again. + let quiet: Arc bool + Send + Sync> = { + let queue = Arc::clone(&queue_for_callees); + Arc::new(move || queue.lock().unwrap().as_ref().is_none_or(|q| q.quiet())) + }; // The optimised body outlives the baseline compile for the tier // above it; a ladder that ends at the baseline drops it then. #[cfg(feature = "llvm-backend")] let keeps_bodies = matches!(tier2_backend, Tier2Backend::LLVM); #[cfg(not(feature = "llvm-backend"))] let keeps_bodies = false; + // What a body that arrived optimised (a linked snapshot's) still + // needs, the module's own pass having skipped it: box readers to + // loads, then what those loads let move. Incremental; never the + // pipeline. + let finish = { + let externs = externs.clone(); + let pure_fns = pure_fns.clone(); + move |f: &mut HirFunction| { + crate::opt_audit::note_finishing(f.id); + let boxed = crate::boxes::run_function(f, &externs); + if boxed.expanded + boxed.made + boxed.released + boxed.shared > 0 { + crate::licm::run(f); + crate::cse::eliminate_with(f, &pure_fns); + } + } + }; + let finish: Arc = Arc::new(finish); // The body of a lazy function as every tier runs it: the // interpreter, the baseline and the tier above compile one body, - // so a frame in any of them can move to the next. Made once, at - // the first call or the first compile, whichever comes first, - // and kept in `optimized_bodies` from then. + // so a frame in any of them can move to the next. Made in the + // function's cell by the first call or compile to ask, whichever + // comes first; any other asking meanwhile waits for it. + self.optimized_bodies + .add(by_bead.values().map(|(id, _, _)| *id)); let optimize_body = { let by_bead: HashMap)> = by_bead .iter() @@ -2734,68 +3190,77 @@ impl TieredBackend { let facts = Arc::clone(&facts); let finished = finished.clone(); let lazy = lazy.clone(); - let externs = externs.clone(); - let pure_fns = pure_fns.clone(); + let finish = Arc::clone(&finish); + let quiet = Arc::clone(&quiet); move |bead_id: u64| -> Option> { let (func_id, module_arc) = by_bead.get(&bead_id)?; - if let Some(body) = optimized_bodies.lock().unwrap().get(func_id) { + let cell = optimized_bodies.cell(*func_id)?; + if let Some(body) = cell.get() { return Some(Arc::clone(body)); } - let body = if finished.contains(func_id) { - // Optimised with its snapshot; what the module's own - // pass would have done to it, done to it alone: box - // readers to loads, then what those loads let move. - let mut f = module_arc.functions.get(func_id)?.clone(); - f.attributes.deferred = false; - let boxed = crate::boxes::run_function(&mut f, &externs); - if boxed.expanded + boxed.made + boxed.released + boxed.shared > 0 { - crate::licm::run(&mut f); - crate::cse::eliminate_with(&mut f, &pure_fns); - } - Arc::new(f) - } else { - if !lazy.contains(func_id) { - return None; + let finished = finished.contains(func_id); + if !module_arc.functions.contains_key(func_id) + || (!finished && !lazy.contains(func_id)) + { + return None; + } + let body = cell.get_or_init(|| { + if finished { + let mut f = module_arc.functions[func_id].clone(); + f.attributes.deferred = false; + finish(&mut f); + return Arc::new(f); } - let mut optimized = optimized.lock().unwrap(); - let scratch = Scratch::of(&mut optimized, module_arc, &facts); - let f = scratch.module.functions.get_mut(func_id)?; - f.attributes.optimized = false; - f.attributes.deferred = false; - // `ZYNTAX_DUMP_HIR_DIR` gets the body before and after - // its own optimisation, as `fn--lowered.hir` and - // `fn--body.hir`. - let name = f.name.resolve_global().unwrap_or_default(); - if std::env::var_os("ZYNTAX_DUMP_HIR_DIR").is_some() { - let lowered = f.clone(); - crate::hir_dump::dump_function_to_dir( - &lowered, - &scratch.module, - &format!("{name}-lowered"), + // The scratch is held until the body is in the cell's + // hands, so no other caller sees it half made. + let _using = optimized.enter(&*quiet); + let mut slots = optimized.slots.lock().unwrap(); + let scratch = Scratch::of(&mut slots, module_arc); + scratch.reach([*func_id], module_arc, &optimized_bodies, &lazy); + if !scratch.done.contains(func_id) { + let f = scratch + .module + .functions + .get_mut(func_id) + .expect("the scratch is a copy of the function's module"); + f.attributes.optimized = false; + f.attributes.deferred = false; + // `ZYNTAX_DUMP_HIR_DIR` gets the body before and + // after its own optimisation, as + // `fn--lowered.hir` and `fn--body.hir`. + if std::env::var_os("ZYNTAX_DUMP_HIR_DIR").is_some() { + let name = f.name.resolve_global().unwrap_or_default(); + let lowered = f.clone(); + crate::hir_dump::dump_function_to_dir( + &lowered, + &scratch.module, + &format!("{name}-lowered"), + ); + } + crate::run_interp_safe_opts_cached( + &mut scratch.module, + &facts.cache(module_arc), ); + scratch.done.insert(*func_id); } - crate::run_interp_safe_opts_cached(&mut scratch.module, &scratch.cache); - let f = scratch.module.functions.get_mut(func_id)?; + let f = scratch + .module + .functions + .get_mut(func_id) + .expect("no pass removes a function"); f.attributes.optimized = true; f.attributes.deferred = true; let mut body = f.clone(); body.attributes.deferred = false; + let name = body.name.resolve_global().unwrap_or_default(); crate::hir_dump::dump_function_to_dir( &body, &scratch.module, &format!("{name}-body"), ); Arc::new(body) - }; - // Another thread may have made it meanwhile; the first - // one in stays, so every tier reads the same body. - Some(Arc::clone( - optimized_bodies - .lock() - .unwrap() - .entry(*func_id) - .or_insert(body), - )) + }); + Some(Arc::clone(body)) } }; let optimize_body: Arc Option> + Send + Sync> = @@ -2806,8 +3271,12 @@ impl TieredBackend { move |bead_id| optimize_body(bead_id) }); // A function of the module's making that is not in it, an - // outlined resume point: optimised in the same scratch, alongside - // the bodies it may inline, and taken out again. + // outlined resume point. Cut from a body already through the + // pipeline, it takes what the cut calls for alone: the finishing + // passes a body that arrived optimised takes, and the cleanup of + // the entry the cut made. Cut from a lowered body, it is + // optimised in the same scratch, alongside the bodies it may + // inline, and taken out again. self.optimize_extra = Some(Arc::new({ let by_bead: HashMap> = by_bead .iter() @@ -2815,15 +3284,30 @@ impl TieredBackend { .collect(); let optimized = Arc::clone(&optimized); let facts = Arc::clone(&facts); - move |bead_id: u64, f: HirFunction| -> HirFunction { + let optimized_bodies = Arc::clone(&optimized_bodies); + let finish = Arc::clone(&finish); + let lazy = lazy.clone(); + let quiet = Arc::clone(&quiet); + move |bead_id: u64, mut f: HirFunction, already_optimized: bool| -> HirFunction { + if already_optimized { + finish(&mut f); + crate::cfg_simplify::run(&mut f); + crate::phi_prune::run_function(&mut f); + f.attributes.optimized = false; + f.attributes.deferred = false; + return f; + } let Some(module_arc) = by_bead.get(&bead_id) else { return f; }; - let mut optimized = optimized.lock().unwrap(); - let scratch = Scratch::of(&mut optimized, module_arc, &facts); + let _using = optimized.enter(&*quiet); + let mut slots = optimized.slots.lock().unwrap(); + let scratch = Scratch::of(&mut slots, module_arc); + let callees: Vec = direct_callees(&f).collect(); + scratch.reach(callees, module_arc, &optimized_bodies, &lazy); let id = f.id; scratch.module.functions.insert(id, f); - crate::run_interp_safe_opts_cached(&mut scratch.module, &scratch.cache); + crate::run_interp_safe_opts_cached(&mut scratch.module, &facts.cache(module_arc)); // No pass removes a function, so it is there to take back. let mut f = scratch .module @@ -2867,10 +3351,6 @@ impl TieredBackend { })); // What compiling a function on its first call does, once off the // caller's stack. - // The queue is made after this closure, which reads it through - // this cell once it exists. - let queue_for_callees: Arc>>> = Arc::new(Mutex::new(None)); - let queue_cell = Arc::clone(&queue_for_callees); let bead_of: HashMap = by_bead.iter().map(|(b, (id, _, _))| (*id, *b)).collect(); let published = Arc::clone(&done); @@ -2972,7 +3452,7 @@ impl TieredBackend { } } let lazy_started = web_time::Instant::now(); - let optimized = optimized_bodies.lock().unwrap().contains_key(func_id); + let optimized = optimized_bodies.get(*func_id).is_some(); if quick && !optimized && let Some(queue) = &queue @@ -3009,6 +3489,7 @@ impl TieredBackend { quick_baseline.note_published(bead_id); } publish(entry as usize, true); + release_interp_body(&quick_baseline.interp_bodies, *func_id); queue.request_replacement(bead_id); if trace { eprintln!( @@ -3085,12 +3566,6 @@ impl TieredBackend { bound.bead().eager_install(entry); } } - // A body with a loop is kept for the resume points an - // interpreted frame of it may still ask for; without one, or - // a tier above, the compile was its last reader. - if !keeps_bodies && osr::find_loop_headers(&body).is_empty() { - optimized_bodies.lock().unwrap().remove(func_id); - } publish(entry as usize, false); // A promotion held for the quick baseline goes ahead now // that its optimised body exists. Taken after the publish, @@ -3105,6 +3580,15 @@ impl TieredBackend { if keeps_bodies && static_hot.contains(&bead_id) { promote_static_hot(bead_id, &body); } + // Calls from here on run the code. A frame still interpreting + // holds the body it runs, and finds it by that, so the interp + // body goes. A ladder that ends here had its last reader of + // the optimised body in this compile, but for a caller + // optimised later that inlines it. + release_interp_body(&quick_baseline.interp_bodies, *func_id); + if !keeps_bodies && !kept_for_inlining(&body) { + optimized_bodies.release(*func_id, &body); + } // `ZYNTAX_TRACE_LAZY=1` names each first-call compile with // the time it took, the wait for the backend included, what // kind of body it was, and the thread that did it; and each @@ -3169,7 +3653,6 @@ impl TieredBackend { .stack_size(16 << 20) .spawn(move || { ON_WARM_UP.with(|on| on.set(true)); - let mut idle_since: Option = None; loop { if stop.load(std::sync::atomic::Ordering::Acquire) { return; @@ -3177,11 +3660,9 @@ impl TieredBackend { queue.busy.store(true, std::sync::atomic::Ordering::Release); match queue.take() { Some(Job::Promote(bead_id, site)) => { - idle_since = None; queue.run_promotions(bead_id, site); } Some(Job::Compile(bead_id, count, at)) => { - idle_since = None; if queue.still_wanted(bead_id, count, at) { compile(bead_id, false); } else if std::env::var_os("ZYNTAX_TRACE_LAZY").is_some() { @@ -3191,13 +3672,10 @@ impl TieredBackend { } } None => { - // A quiet spell: the scratch modules the + // Nothing waits: the scratch modules the // program's functions are optimised in are // let go, and made again if asked. - let since = *idle_since.get_or_insert_with(web_time::Instant::now); - if since.elapsed() > std::time::Duration::from_millis(250) { - scratch_shared.lock().unwrap().clear(); - } + scratch_shared.release_if_unused(); queue .busy .store(false, std::sync::atomic::Ordering::Release); @@ -3338,6 +3816,17 @@ impl TieredBackend { ) }) .collect(); + // Of those, the functions that arrived optimised. + let finished_beads: HashSet = self + .functions + .iter() + .filter(|(id, _)| self.finished.contains(id)) + .map(|(_, e)| e.bead_id) + .collect(); + // The loops a region was outlined at, by (bead, body tag, loop + // ordinal): each is outlined, and optimised, once. + let outlined_at: Arc>> = + Arc::new(Mutex::new(HashSet::new())); // After every install at the optimizing tier, whichever path // made it, the sites frames asked at meanwhile get its resume @@ -3353,10 +3842,19 @@ impl TieredBackend { .iter() .map(|(bead, (id, _, _, module, _))| (*bead, (*id, Arc::clone(module)))) .collect(); + let bodies = Arc::clone(&self.optimized_bodies); Box::new(move |bead_id: u64, body: &Arc| { let Some((func_id, module_arc)) = modules.get(&bead_id) else { return; }; + // The top of the ladder: a loop-free body has no + // frame left to ask anything of it, but for a + // caller optimised later that inlines it. A body + // with a loop is kept for the resume points a frame + // of the baseline may still ask for. + if osr::find_loop_headers(body).is_empty() && !kept_for_inlining(body) { + bodies.release(*func_id, body); + } let sites = late.lock().unwrap().installed(bead_id); if sites.is_empty() { return; @@ -3411,9 +3909,7 @@ impl TieredBackend { if let Some(llvm) = &llvm && late.lock().unwrap().claim_if_installed(bead_id, site) { - let body = swapped - .clone() - .or_else(|| optimized_bodies.lock().unwrap().get(func_id).cloned()); + let body = swapped.clone().or_else(|| optimized_bodies.get(*func_id)); let bodies = late_resume_bodies( body, module_arc.functions.get(func_id), @@ -3475,19 +3971,19 @@ impl TieredBackend { return true; } } - // The body: the one a reload swapped in, else the one the - // first-call compile optimised, else the module's. - let optimized = optimized_bodies.lock().unwrap().get(func_id).cloned(); - let func_arc = match (swapped, optimized) { - (Some(f), _) => Arc::clone(f), - (None, Some(f)) => f, - (None, None) => match module_arc.functions.get(func_id) { - Some(f) => Arc::new(f.clone()), - None => return false, - }, - }; let module_arc = Arc::clone(module_arc); let func_id = *func_id; + // The body every tier compiles, made now if it was not yet. + let body_now = || { + tier_body( + func_id, + bead_id, + swapped.as_ref(), + &optimized_bodies, + *lazy, + &module_arc, + ) + }; // An interpreted frame can enter no code mid-loop without a // resume point, and the baseline's body has only probes. Its // resume points come first, before the body: the frame is @@ -3498,21 +3994,23 @@ impl TieredBackend { if let osr::Requester::Interpreted { site } = from { let (body_tag, ordinal, _) = osr::decode_osr_site(site); // The frame runs the body its tag names: the one the - // interpreter made for the first runs, or the module's - // as lowered, when it started before any optimised body - // existed; else that one. - let unoptimized = interp_bodies + // interpreter made for the first runs, or the module's, + // when it started before any optimised body existed; else + // the tier body. The module's is through the pipeline + // already for a function that arrived optimised. + let untiered = interp_bodies .lock() .unwrap() .get(&func_id) - .cloned() + .and_then(FrameBodies::interp) .filter(|f| osr::body_tag(f) == body_tag) + .map(|f| (f, false)) .or_else(|| { module_arc .functions .get(&func_id) .filter(|f| osr::body_tag(f) == body_tag) - .map(|f| Arc::new(f.clone())) + .map(|f| (Arc::new(f.clone()), finished_beads.contains(&bead_id))) }); // `ZYNTAX_DISABLE_OUTLINE=1` gives such a frame the // baseline's resume point instead, to bisect; safe to @@ -3520,24 +4018,34 @@ impl TieredBackend { let outline = optimize_extra .as_ref() .filter(|_| std::env::var_os("ZYNTAX_DISABLE_OUTLINE").is_none()); - match (unoptimized, outline) { - // A frame on an unoptimised body leaves into the - // region optimised on its own: the body itself was - // never optimised, and need not be for this call. - (Some(body), Some(optimize)) => { - let optimize = Arc::clone(optimize); - publish_outlined_resume_point( - &cranelift, - func_id, - bead_id, - &body, - &module_arc, - ordinal, - &*optimize, - &outlined_regions, - ); + match (untiered, outline) { + // A frame on a body other than the tier body leaves + // into the region optimised on its own: the body + // itself need not be for this call. A loop another + // request outlined already has its resume point, or + // is getting it. + (Some((body, optimized)), Some(optimize)) => { + let key = (bead_id, body_tag, ordinal); + if outlined_at.lock().unwrap().insert(key) { + let optimize = Arc::clone(optimize); + let attempted = publish_outlined_resume_point( + &cranelift, + func_id, + bead_id, + &body, + &module_arc, + ordinal, + &|f| optimize(bead_id, f, optimized), + // Kept for the tier above, when there + // is one. + optimizing.then_some(&*outlined_regions), + ); + if !attempted { + outlined_at.lock().unwrap().remove(&key); + } + } } - (Some(body), None) => { + (Some((body, _)), None) => { publish_baseline_resume_points( &cranelift, func_id, @@ -3547,16 +4055,39 @@ impl TieredBackend { ordinal, ); } - (None, _) => publish_baseline_resume_points( - &cranelift, - func_id, - bead_id, - &func_arc, - &module_arc, - ordinal, - ), + // A frame on neither runs the tier body, which is + // made by now, and which the frame holds. + (None, _) => { + let held = interp_bodies + .lock() + .unwrap() + .get(&func_id) + .and_then(|b| b.tier_weak.upgrade()) + .filter(|f| osr::body_tag(f) == body_tag); + if let Some(body) = held.or_else(body_now) { + publish_baseline_resume_points( + &cranelift, + func_id, + bead_id, + &body, + &module_arc, + ordinal, + ); + } + } } } + // A bead with its baseline, and with the tier above's answer + // in, has nothing left to ask of its body, which may have + // been let go. + if bound.bead().compiled().is_some() + && (!optimizing || outcomes.lock().unwrap().contains_key(&(bead_id, tier_idx))) + { + return true; + } + let Some(func_arc) = body_now() else { + return false; + }; // The baseline, so the promotion has something to promote and // the next call has code. if !ensure_baseline( @@ -3591,18 +4122,6 @@ impl TieredBackend { } return true; } - // The tier above compiles the body the baseline was optimised - // from, which a first-call compile just made when the request - // came from the interpreter. - let func_arc = match swapped { - Some(_) => func_arc, - None => optimized_bodies - .lock() - .unwrap() - .get(&func_id) - .cloned() - .unwrap_or(func_arc), - }; #[cfg(feature = "llvm-backend")] if !crate::abi::llvm_entry_abi_supported(&func_arc, false) { if osr::osr_trace_enabled() { @@ -3733,16 +4252,24 @@ impl TieredBackend { .get(&func_id) .ok_or_else(|| CompilerError::Backend(format!("Function {:?} not found", func_id)))?; - let func_arc = entry.body(func_id); - let module_arc = Arc::clone(&entry.module); + let lazy = self.lazy.contains(&func_id); let bead_id = entry.bead_id; + let func_arc = tier_body( + func_id, + bead_id, + entry.function.as_ref(), + &self.optimized_bodies, + lazy, + &entry.module, + ) + .ok_or_else(|| CompilerError::Backend(format!("Function {:?} has no body", func_id)))?; + let module_arc = Arc::clone(&entry.module); let cranelift = Arc::clone(&self.cranelift); #[cfg(feature = "llvm-backend")] let llvm = self.llvm.as_ref().map(Arc::clone); let tier2_backend = self.config.tier2_backend; let verbosity = self.config.verbosity; let tier_idx = target_tier.index(); - let lazy = self.lazy.contains(&func_id); if !ensure_baseline( &entry.bound, func_id, @@ -4166,7 +4693,7 @@ fn publish_late_resume_point( }; let def = ZyntaxFunctionDef { id: *id, - function: (**body).clone(), + function: Arc::clone(body), module: Arc::clone(module_arc), tier: OptimizationTier::Optimized.index(), bead_id, @@ -4297,10 +4824,12 @@ fn ensure_baseline( /// `optimize` and compiled as a function of its own (see /// [`osr::outline`]); the site gets the adapter that enters it. Where the /// region cannot stand alone, the baseline's own resume point instead. -/// The region goes into `regions` under the bead before the site is -/// published, for the tier above to give the frame resume points of its +/// The region goes into `regions`, given when there is a tier above, +/// under the bead before the site is published, for that tier to give +/// the frame resume points of its /// own (see [`promote_outlined_region`]): the frame may ask for that -/// promotion from the region the moment it is in it. +/// promotion from the region the moment it is in it. Whether the region +/// was attempted: false when there was no such loop or no frame waits. #[allow(clippy::too_many_arguments, clippy::type_complexity)] fn publish_outlined_resume_point( cranelift: &Arc, @@ -4309,12 +4838,12 @@ fn publish_outlined_resume_point( func_arc: &Arc, module_arc: &Arc, asked_at: u64, - optimize: &(dyn Fn(u64, HirFunction) -> HirFunction + Send + Sync), - regions: &Mutex)>>>, -) { + optimize: &(dyn Fn(HirFunction) -> HirFunction + Send + Sync), + regions: Option<&Mutex)>>>>, +) -> bool { let def = ZyntaxFunctionDef { id: func_id, - function: (**func_arc).clone(), + function: Arc::clone(func_arc), module: Arc::clone(module_arc), tier: OptimizationTier::Baseline.index(), bead_id, @@ -4323,7 +4852,7 @@ fn publish_outlined_resume_point( .get(asked_at as usize) .copied() else { - return; + return false; }; if !osr::frame_waiting(bead_id) { if osr::osr_trace_enabled() { @@ -4332,7 +4861,7 @@ fn publish_outlined_resume_point( func_arc.name.resolve_global().unwrap_or_default() ); } - return; + return false; } let outlined = std::thread::scope(|scope| { std::thread::Builder::new() @@ -4340,7 +4869,7 @@ fn publish_outlined_resume_point( .stack_size(16 << 20) .spawn_scoped(scope, || { cranelift - .outlined_resume_point_at(&def, header, &|f| optimize(bead_id, f)) + .outlined_resume_point_at(&def, header, optimize) .map(|out| { let sites: Vec<(u64, usize)> = out .sites @@ -4361,14 +4890,16 @@ fn publish_outlined_resume_point( ); } publish_baseline_resume_points(cranelift, func_id, bead_id, func_arc, module_arc, asked_at); - return; + return true; }; - regions - .lock() - .unwrap() - .entry(bead_id) - .or_default() - .push((region_id, Arc::new(region))); + if let Some(regions) = regions { + regions + .lock() + .unwrap() + .entry(bead_id) + .or_default() + .push((region_id, Arc::new(region))); + } for (site, code) in sites { let code = code as *mut (); if !code.is_null() && osr::helper_for(bead_id, site).is_null() { @@ -4381,6 +4912,7 @@ fn publish_outlined_resume_point( osr::publish_helper(bead_id, site, code); } } + true } /// The optimizing tier's resume points for an outlined region: the @@ -4410,7 +4942,7 @@ fn promote_outlined_region( crate::hir_dump::dump_function_to_dir(region, module_arc, &format!("{name}-tier{tier_idx}")); let def = ZyntaxFunctionDef { id: region_id, - function: (**region).clone(), + function: Arc::clone(region), module: Arc::clone(module_arc), tier: tier_idx, bead_id, @@ -4450,7 +4982,7 @@ fn publish_baseline_resume_points( ) { let def = ZyntaxFunctionDef { id: func_id, - function: (**func_arc).clone(), + function: Arc::clone(func_arc), module: Arc::clone(module_arc), tier: OptimizationTier::Baseline.index(), bead_id, @@ -4548,7 +5080,7 @@ pub fn compile_at_tier( ) -> *mut () { let def = ZyntaxFunctionDef { id: func_id, - function: (**func_arc).clone(), + function: Arc::clone(func_arc), module: Arc::clone(module_arc), tier: tier_idx, bead_id, diff --git a/crates/compiler/tests/body_retention.rs b/crates/compiler/tests/body_retention.rs new file mode 100644 index 00000000..7960e9da --- /dev/null +++ b/crates/compiler/tests/body_retention.rs @@ -0,0 +1,303 @@ +//! What the tiers keep of a lazy function's HIR. Its interp body and its +//! optimised body go once it has native code and no interpreted frame +//! runs them; a frame still running one finds it for as long as it +//! runs; a large body whose first frame returns before there is native +//! code goes then; and a body asked for again after it went is made +//! again, once. + +#![cfg(feature = "cranelift-backend")] + +use std::collections::HashSet; +use std::sync::Arc; + +use indexmap::IndexMap; +use zyntax_compiler::hir::{ + BinaryOp, HirBlock, HirConstant, HirFunction, HirFunctionSignature, HirId, HirInstruction, + HirModule, HirParam, HirPhi, HirTerminator, HirType, HirValue, HirValueKind, +}; +use zyntax_compiler::opt_audit; +use zyntax_compiler::osr; +use zyntax_compiler::tiered_backend::{TieredBackend, TieredConfig}; +use zyntax_typed_ast::InternedString; + +/// Steps of the chain each body runs: two instructions a step, more than +/// any size a body is kept at. +const STEPS: usize = 300; + +fn key(k: usize) -> i32 { + (k as i32).wrapping_mul(0x9e37) ^ 0x5a5a +} + +/// The chain, as the compiled code computes it. +fn chain(mut x: i32) -> i32 { + for k in 0..STEPS { + x = x.wrapping_mul(31) ^ key(k); + } + x +} + +struct Builder { + values: IndexMap, +} + +impl Builder { + fn value(&mut self, ty: HirType, kind: HirValueKind) -> HirId { + let id = HirId::new(); + self.values.insert( + id, + HirValue { + id, + ty, + kind, + uses: Default::default(), + span: None, + }, + ); + id + } + + fn constant(&mut self, v: i32) -> HirId { + self.value(HirType::I32, HirValueKind::Constant(HirConstant::I32(v))) + } + + fn binary( + &mut self, + out: &mut Vec, + op: BinaryOp, + left: HirId, + right: HirId, + ) -> HirId { + let ty = if matches!(op, BinaryOp::Lt) { + HirType::Bool + } else { + HirType::I32 + }; + let result = self.value(ty.clone(), HirValueKind::Instruction); + out.push(HirInstruction::Binary { + op, + result, + ty, + left, + right, + }); + result + } + + /// `x * 31 ^ key(k)` for each step, into `out`. + fn chain(&mut self, out: &mut Vec, mut x: HirId) -> HirId { + let thirty_one = self.constant(31); + for k in 0..STEPS { + let key = self.constant(key(k)); + let m = self.binary(out, BinaryOp::Mul, x, thirty_one); + x = self.binary(out, BinaryOp::Xor, m, key); + } + x + } +} + +fn block( + id: HirId, + phis: Vec, + instructions: Vec, + terminator: HirTerminator, +) -> HirBlock { + HirBlock { + id, + label: None, + phis, + instructions, + terminator, + dominance_frontier: Default::default(), + predecessors: vec![], + successors: vec![], + } +} + +fn function(name: &str, param: HirId, b: Builder, blocks: Vec) -> HirFunction { + let mut f = HirFunction::new( + InternedString::new_global(name), + HirFunctionSignature { + params: vec![HirParam { + id: param, + name: InternedString::new_global("n"), + ty: HirType::I32, + attributes: Default::default(), + ownership: Default::default(), + }], + returns: vec![HirType::I32], + type_params: vec![], + const_params: vec![], + lifetime_params: vec![], + is_variadic: false, + is_async: false, + is_fiber: false, + effects: vec![], + is_pure: true, + }, + ); + f.entry_block = blocks[0].id; + f.blocks = blocks.into_iter().map(|b| (b.id, b)).collect(); + f.values = b.values; + // Left for its first call, as the runtime leaves a program's + // functions. + f.attributes.optimized = true; + f.attributes.deferred = true; + f +} + +/// `looped(n)`: `x = n`, then `n` times `x = chain(x)`; returns `x`. +fn looped() -> HirFunction { + let mut b = Builder { + values: IndexMap::new(), + }; + let n = b.value(HirType::I32, HirValueKind::Parameter(0)); + let zero = b.constant(0); + let one = b.constant(1); + let (entry, header, body, exit) = (HirId::new(), HirId::new(), HirId::new(), HirId::new()); + let i = b.value(HirType::I32, HirValueKind::Instruction); + let x = b.value(HirType::I32, HirValueKind::Instruction); + let mut test = Vec::new(); + let cmp = b.binary(&mut test, BinaryOp::Lt, i, n); + let mut step = Vec::new(); + let x_next = b.chain(&mut step, x); + let i_next = b.binary(&mut step, BinaryOp::Add, i, one); + let blocks = vec![ + block( + entry, + vec![], + vec![], + HirTerminator::Branch { target: header }, + ), + block( + header, + vec![ + HirPhi { + result: i, + ty: HirType::I32, + incoming: vec![(zero, entry), (i_next, body)], + }, + HirPhi { + result: x, + ty: HirType::I32, + incoming: vec![(n, entry), (x_next, body)], + }, + ], + test, + HirTerminator::CondBranch { + condition: cmp, + true_target: body, + false_target: exit, + }, + ), + block(body, vec![], step, HirTerminator::Branch { target: header }), + block( + exit, + vec![], + vec![], + HirTerminator::Return { values: vec![x] }, + ), + ]; + function("looped", n, b, blocks) +} + +/// `straight(n)`: `chain(n)`, with no loop. +fn straight() -> HirFunction { + let mut b = Builder { + values: IndexMap::new(), + }; + let n = b.value(HirType::I32, HirValueKind::Parameter(0)); + let entry = HirId::new(); + let mut out = Vec::new(); + let x = b.chain(&mut out, n); + let blocks = vec![block( + entry, + vec![], + out, + HirTerminator::Return { values: vec![x] }, + )]; + function("straight", n, b, blocks) +} + +fn call(entry: *const u8, n: i32) -> i32 { + // SAFETY: both functions are `fn(i32) -> i32`. + let f: extern "C" fn(i32) -> i32 = unsafe { std::mem::transmute(entry) }; + f(n) +} + +/// One test: the lazy compiler a backend installs is the process's. +#[test] +fn bodies_go_once_nothing_can_ask_for_them() { + opt_audit::enable(); + let (looped, straight) = (looped(), straight()); + let (a, b) = (looped.id, straight.id); + let mut module = HirModule::new(InternedString::new_global("body_retention")); + module.functions.insert(a, looped); + module.functions.insert(b, straight); + let mut backend = TieredBackend::new(TieredConfig::default()).expect("tiered backend"); + backend.set_emit_osr_probes(true); + backend + .compile_module_lazily(module, None, HashSet::from([a, b]), HashSet::new(), false) + .expect("module compiles"); + let (_, entry, bead) = backend.interpreter_bridge(); + let Some(bead_a) = bead(a) else { + // OSR off in this environment: no bead to ask for bodies by. + return; + }; + let mut source = backend.interpreter_body_source(); + let mut frame_exit = backend.interpreter_frame_exit_hook(); + + // A frame of `looped` starts in the interpreter on its interp body, + // and holds it while it runs. + let frame = source(a).expect("an interp body"); + let frame_tag = osr::body_tag(&frame); + // The optimised body, made before the first compile, which reads it. + let optimized = Arc::downgrade(&osr::lazy_optimized_body(bead_a).expect("an optimised body")); + assert_eq!(opt_audit::pipeline_runs(a), 1); + + // The first call compiles it: native code from here on. + assert_eq!(call(entry(a).expect("a stub"), 3), chain(chain(chain(3)))); + assert!( + optimized.upgrade().is_none(), + "the optimised body outlived the compile that was its last reader" + ); + // The frame still running finds its body by the tag it asks at. + let found = source(a).expect("the frame's body"); + assert!(Arc::ptr_eq(&found, &frame)); + assert_eq!(osr::body_tag(&found), frame_tag); + drop(found); + // Its frame returns: nothing holds the interp body any more. + let gone = Arc::downgrade(&frame); + drop(frame); + assert!( + gone.upgrade().is_none(), + "the interp body outlived its last frame" + ); + + // Asked for again, each body is made again: the optimised one once + // in its new cell, from the scratch while one still holds it. + let remade = source(a).expect("an interp body made again"); + assert_eq!(opt_audit::pipeline_runs(a), 1); + drop(remade); + let again = osr::lazy_optimized_body(bead_a).expect("an optimised body made again"); + let runs = opt_audit::pipeline_runs(a); + assert!(runs <= 2, "the pipeline ran {runs} times"); + let same = osr::lazy_optimized_body(bead_a).expect("the optimised body"); + assert!(Arc::ptr_eq(&again, &same)); + assert_eq!(opt_audit::pipeline_runs(a), runs); + assert_eq!(call(entry(a).expect("code"), 2), chain(chain(2))); + drop((again, same)); + + // `straight` runs once in the interpreter: a large body whose first + // frame returns with no native code for it goes then, once. + let body = source(b).expect("an interp body"); + let gone = Arc::downgrade(&body); + drop(body); + assert!(frame_exit(b), "a large body run once was kept"); + assert!(gone.upgrade().is_none()); + let body = source(b).expect("an interp body made again"); + let kept = Arc::downgrade(&body); + drop(body); + assert!(!frame_exit(b), "a body made again was let go again"); + assert!(kept.upgrade().is_some()); + assert_eq!(call(entry(b).expect("a stub"), 5), chain(5)); +} diff --git a/crates/compiler/tests/optimized_once.rs b/crates/compiler/tests/optimized_once.rs new file mode 100644 index 00000000..fe30f907 --- /dev/null +++ b/crates/compiler/tests/optimized_once.rs @@ -0,0 +1,114 @@ +//! A lazy function's HIR goes through the pipeline once, and every tier +//! above the interpreter compiles that body: the baseline, the LLVM tier +//! through the promotion requester, and the LLVM tier through an +//! explicit `optimize_function`. + +#![cfg(all(feature = "llvm-backend", feature = "cranelift-backend"))] + +mod common; + +use std::collections::HashSet; +use std::sync::Arc; + +use zyntax_compiler::hir::{HirFunction, HirId, HirModule}; +use zyntax_compiler::opt_audit; +use zyntax_compiler::osr; +use zyntax_compiler::tiered_backend::{ + OptimizationTier, Tier2Backend, TieredBackend, TieredConfig, +}; +use zyntax_typed_ast::InternedString; + +/// `count_to`, left for its first call as the runtime leaves a +/// program's functions. +fn lazy_loop() -> HirFunction { + let (mut function, _) = common::counted_loop(); + function.attributes.optimized = true; + function.attributes.deferred = true; + function +} + +fn call(entry: *const u8, n: i32) -> i32 { + // SAFETY: the entry is `count_to`'s code, `fn(i32) -> i32`. + let f: extern "C" fn(i32) -> i32 = unsafe { std::mem::transmute(entry) }; + f(n) +} + +/// Waits for the LLVM tier to have been handed `id`'s HIR `n` times. +fn llvm_compiles(id: HirId, n: usize) -> Vec { + for _ in 0..1000 { + let bodies = opt_audit::llvm_bodies(id); + if bodies.len() >= n { + return bodies; + } + std::thread::sleep(std::time::Duration::from_millis(10)); + } + opt_audit::llvm_bodies(id) +} + +/// One test: the lazy compiler and the requester a backend installs are +/// the process's. +#[test] +fn every_tier_compiles_the_one_optimised_body() { + opt_audit::enable(); + let (by_requester, by_request) = (lazy_loop(), lazy_loop()); + let (a, b) = (by_requester.id, by_request.id); + let mut module = HirModule::new(InternedString::new_global("optimized_once")); + module.functions.insert(a, by_requester); + module.functions.insert(b, by_request); + let config = TieredConfig { + tier2_backend: Tier2Backend::LLVM, + enable_osr: true, + ..TieredConfig::default() + }; + let mut backend = TieredBackend::new(config).expect("tiered backend"); + backend.set_emit_osr_probes(true); + backend + .compile_module_lazily(module, None, HashSet::from([a, b]), HashSet::new(), false) + .expect("module compiles"); + + // Each is called through its stub, which optimises the body and + // compiles it at the baseline; a call too short for its loop to ask + // for the tier above. + let (_, entry, bead) = backend.interpreter_bridge(); + for id in [a, b] { + assert_eq!(call(entry(id).expect("a stub"), 5), 10); + } + let (bead_a, bead_b) = (bead(a).expect("a bead"), bead(b).expect("a bead")); + let stored = |bead| osr::lazy_optimized_body(bead).expect("an optimised body"); + let (body_a, body_b) = (stored(bead_a), stored(bead_b)); + + // The requester, as a compiled frame asks. + assert!(osr::run_promotion( + bead_a, + osr::Requester::Compiled { site: osr::NO_SITE } + )); + let handed = llvm_compiles(a, 1); + assert_eq!( + handed, + vec![Arc::as_ptr(&body_a) as usize], + "the requester's LLVM compile took a body other than the optimised one" + ); + + // An explicit promotion. + backend + .optimize_function(b, OptimizationTier::Optimized) + .expect("promotion"); + let handed = llvm_compiles(b, 1); + assert_eq!( + handed, + vec![Arc::as_ptr(&body_b) as usize], + "optimize_function's LLVM compile took a body other than the optimised one" + ); + + for id in [a, b] { + assert_eq!( + opt_audit::pipeline_runs(id), + 1, + "the pipeline ran more than once over one function" + ); + } + // The interpreter runs the same body. + let mut source = backend.interpreter_body_source(); + let interp = source(a).expect("a body for the interpreter"); + assert!(Arc::ptr_eq(&interp, &body_a)); +} diff --git a/crates/compiler/tests/optimized_once_race.rs b/crates/compiler/tests/optimized_once_race.rs new file mode 100644 index 00000000..fc6e1689 --- /dev/null +++ b/crates/compiler/tests/optimized_once_race.rs @@ -0,0 +1,87 @@ +//! Callers asking for a lazy function's optimised body at once, while +//! the compile worker makes it too, all get one body, made by one run of +//! the pipeline. + +#![cfg(feature = "cranelift-backend")] + +mod common; + +use std::collections::HashSet; +use std::sync::{Arc, Barrier}; + +use zyntax_compiler::hir::{ + HirConstant, HirInstruction, HirModule, HirType, HirValue, HirValueKind, +}; +use zyntax_compiler::opt_audit; +use zyntax_compiler::osr; +use zyntax_compiler::tiered_backend::{TieredBackend, TieredConfig}; +use zyntax_typed_ast::InternedString; + +const ASKERS: usize = 8; + +/// One test: the lazy optimizer a backend installs is the process's. +#[test] +fn racing_callers_share_one_run_of_the_pipeline() { + opt_audit::enable(); + for _ in 0..200 { + // A loop hot before its first call, so the worker is asked to + // compile it as the module loads. + let (mut function, header) = common::counted_loop(); + let bound = zyntax_compiler::hir::HirId::new(); + function.values.insert( + bound, + HirValue { + id: bound, + ty: HirType::I32, + kind: HirValueKind::Constant(HirConstant::I32(5000)), + uses: Default::default(), + span: None, + }, + ); + for inst in &mut function.blocks.get_mut(&header).unwrap().instructions { + if let HirInstruction::Binary { right, .. } = inst { + *right = bound; + } + } + function.attributes.optimized = true; + function.attributes.deferred = true; + let id = function.id; + let mut module = HirModule::new(InternedString::new_global("optimized_once_race")); + module.functions.insert(id, function); + let config = TieredConfig { + enable_osr: true, + ..TieredConfig::default() + }; + let mut backend = TieredBackend::new(config).expect("tiered backend"); + backend + .compile_module_lazily(module, None, HashSet::from([id]), HashSet::new(), false) + .expect("module compiles"); + let (_, _, bead) = backend.interpreter_bridge(); + let bead = bead(id).expect("a bead"); + + let barrier = Arc::new(Barrier::new(ASKERS)); + let askers: Vec<_> = (0..ASKERS) + .map(|_| { + let barrier = Arc::clone(&barrier); + std::thread::spawn(move || { + barrier.wait(); + osr::lazy_optimized_body(bead).expect("an optimised body") + }) + }) + .collect(); + let bodies: Vec<_> = askers + .into_iter() + .map(|t| t.join().expect("asker")) + .collect(); + assert!( + bodies.iter().all(|b| Arc::ptr_eq(b, &bodies[0])), + "callers were handed different bodies" + ); + drop(backend); + assert_eq!( + opt_audit::pipeline_runs(id), + 1, + "the pipeline ran more than once over one function" + ); + } +} diff --git a/crates/zyntax_embed/src/runtime/tiered.rs b/crates/zyntax_embed/src/runtime/tiered.rs index 34075e78..327a9f71 100644 --- a/crates/zyntax_embed/src/runtime/tiered.rs +++ b/crates/zyntax_embed/src/runtime/tiered.rs @@ -724,6 +724,7 @@ impl TieredRuntime { } } interp.set_body_source(self.backend.interpreter_body_source()); + interp.set_frame_exit_hook(self.backend.interpreter_frame_exit_hook()); interp.set_address_source(self.backend.interpreter_address_source()); let (thunk, entry, bead) = self.backend.interpreter_bridge(); interp.set_native_bridge(thunk, entry, bead);