Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
6 changes: 5 additions & 1 deletion crates/compiler/src/beadie_adapter.rs
Original file line number Diff line number Diff line change
Expand Up @@ -32,7 +32,9 @@ use crate::hir::{HirFunction, HirId, HirModule};
#[derive(Clone)]
pub struct ZyntaxFunctionDef {
pub id: HirId,
pub function: HirFunction,
/// Shared with whatever handed it over: a tier compiles the body it
/// is given, not a copy made on the way.
pub function: std::sync::Arc<HirFunction>,
/// Module-level effect, handler, global, and callee context required when
/// a single hot function is recompiled outside the initial bulk pass.
pub module: std::sync::Arc<HirModule>,
Expand Down Expand Up @@ -426,6 +428,7 @@ mod llvm_impl {
if sites.is_empty() {
return Vec::new();
}
crate::opt_audit::note_llvm_body(def.id, &def.function);
self.with_lock(|backend| {
backend.set_compile_tier(def.tier);
backend.set_module_context(std::sync::Arc::clone(&def.module));
Expand Down Expand Up @@ -470,6 +473,7 @@ mod llvm_impl {
// Resume points where a frame can take one: the sites frames
// asked at, and an outlined region's own header.
let sites = crate::osr::wanted_resume_points(def.bead_id, &def.function);
crate::opt_audit::note_llvm_body(def.id, &def.function);
self.with_lock(|backend| {
backend.set_compile_tier(tier);
backend.set_module_context(std::sync::Arc::clone(&def.module));
Expand Down
76 changes: 74 additions & 2 deletions crates/compiler/src/hir_interp.rs
Original file line number Diff line number Diff line change
Expand Up @@ -2674,6 +2674,15 @@ impl std::error::Error for InterpError {}
// Interpreter (dispatch loop)
// ─────────────────────────────────────────────────────────────────────────────

/// The body a function's bytecode was made from, when the body source
/// gave it. Held weakly: the runtime decides how long it lives, and a
/// running frame of a function with loops holds it besides.
struct SourceBody {
body: std::sync::Weak<HirFunction>,
/// The bytecode has loop headers a frame can ask resume points at.
loops: bool,
}

pub struct HirInterpreter {
symbols: HashMap<String, SymbolEntry>,
pub profile: IdMap<ProfileSample>,
Expand All @@ -2697,6 +2706,13 @@ pub struct HirInterpreter {
/// Where a function's body comes from when the runtime keeps one
/// apart from the module's: see [`Self::set_body_source`].
body_source: Option<Box<dyn FnMut(HirId) -> Option<std::sync::Arc<HirFunction>> + Send>>,
/// The body each function's bytecode was made from, for those the
/// body source gave: see [`SourceBody`].
source_bodies: IdMap<SourceBody>,
/// Told when the first frame run from a body the body source gave
/// returns; see [`Self::set_frame_exit_hook`].
#[allow(clippy::type_complexity)]
frame_exit_hook: Option<Box<dyn FnMut(HirId) -> bool + Send>>,
/// Compiles, or finds, the thunk that calls native code of a given
/// shape: `fn(target, words, out)`. Installed by a runtime with a
/// native tier; without one, calls into native code use the fixed
Expand Down Expand Up @@ -2850,6 +2866,8 @@ impl HirInterpreter {
uncompilable: IdMap::default(),
tick_callbacks: IdMap::default(),
body_source: None,
source_bodies: IdMap::default(),
frame_exit_hook: None,
thunk_source: None,
entry_source: None,
bead_source: None,
Expand Down Expand Up @@ -2980,13 +2998,22 @@ impl HirInterpreter {
self.body_source = Some(source);
}

/// Called with a function's id when the first frame run from a body
/// the body source gave returns. Answers whether the runtime let that
/// body go; the bytecode made from it goes too then, and the next
/// call asks the body source again.
pub fn set_frame_exit_hook(&mut self, hook: Box<dyn FnMut(HirId) -> bool + Send>) {
self.frame_exit_hook = Some(hook);
}

/// Drop the bridge and every tick callback. They hold the native
/// tiers' backends, which the runtime shuts down after them.
pub fn clear_native_bridge(&mut self) {
self.tick_callbacks = IdMap::default();
self.thunk_source = None;
self.entry_source = None;
self.bead_source = None;
self.frame_exit_hook = None;
}

/// Install the bridge to a native tier: `thunk` compiles the caller
Expand Down Expand Up @@ -3520,6 +3547,26 @@ impl HirInterpreter {
}
}

// A frame of a function with loops holds the body its bytecode
// was made from, so a request for resume points it makes finds
// that body while it runs. Bytecode whose body nobody holds any
// more is made again from the body the source gives now.
let mut held: Option<std::sync::Arc<HirFunction>> = None;
let mut stale = false;
if let Some(source) = self.source_bodies.get(&func_id)
&& source.loops
{
held = source.body.upgrade();
stale = held.is_none();
}
// Whether this frame made the bytecode from a body the source
// gave: the runtime is told when it returns.
let mut first = false;
if stale {
self.source_bodies.remove(&func_id);
self.cache.remove(&func_id);
}

// Compile-on-first-use, and refuse-once. What the interpreter
// cannot run, native code runs, when there is native code.
if !self.cache.contains_key(&func_id) {
Expand Down Expand Up @@ -3551,6 +3598,21 @@ impl HirInterpreter {
let taken = self.is_address_taken(module, func_id);
match compile_function_with(module, &mut self.memory, func, taken) {
Ok(cf) => {
// A recursive call compiles the body again while
// the frame that made the entry runs.
if let Some(body) = &shared {
let loops = !cf.osr_sites.is_empty();
self.source_bodies.entry(func_id).or_insert_with(|| {
first = true;
SourceBody {
body: std::sync::Arc::downgrade(body),
loops,
}
});
if loops {
held = Some(std::sync::Arc::clone(body));
}
}
self.cache.insert(func_id, cf);
}
Err(InterpError::UnsupportedInstruction(why)) => {
Expand All @@ -3574,8 +3636,18 @@ impl HirInterpreter {
// calls. The map ownership returns at the end.
let cf = self.cache.remove(&func_id).unwrap();
let result = self.run(module, &cf, args, func_id, dest);
// Put the (immutable) compiled function back.
self.cache.insert(func_id, cf);
drop(held);
let released = first
&& self
.frame_exit_hook
.as_mut()
.is_some_and(|hook| hook(func_id));
if released {
self.source_bodies.remove(&func_id);
} else {
// Put the (immutable) compiled function back.
self.cache.insert(func_id, cf);
}
result
}

Expand Down
6 changes: 6 additions & 0 deletions crates/compiler/src/inline.rs
Original file line number Diff line number Diff line change
Expand Up @@ -181,6 +181,12 @@ fn count_insts(function: &HirFunction) -> usize {
function.blocks.values().map(|b| b.instructions.len()).sum()
}

/// Whether `function` is small enough that some call to it may be
/// inlined.
pub(crate) fn may_inline(function: &HirFunction) -> bool {
count_insts(function) <= MAX_INLINE_INSTS_MULTI_BLOCK
}

/// Inline every eligible direct call within `module`. Iterates a
/// fixed-point: inlining one call can expose a now-eligible callee
/// (when the now-inlined body's prior nested call shape was a
Expand Down
3 changes: 3 additions & 0 deletions crates/compiler/src/lib.rs
Original file line number Diff line number Diff line change
Expand Up @@ -72,6 +72,8 @@ pub mod memory_optimization; // Memory-aware optimizations
pub mod memory_pass;
pub mod monomorphize;
pub mod move_insert; // Owning parameters become `Move` the borrow check can see
#[doc(hidden)]
pub mod opt_audit; // Per-function counts of optimiser runs, for tests
pub mod optimization;
pub mod parallel_dispatch; // A loop with independent iterations becomes a band dispatch
pub mod parallel_safe; // Which counted loops have independent iterations
Expand Down Expand Up @@ -1970,6 +1972,7 @@ fn run_interp_safe_opts_with(
{
return stats;
}
opt_audit::note_pipeline(module);

// Alloca → Malloc promotion runs ONCE up front, before the
// fixed-point sweep. Two reasons it goes here:
Expand Down
24 changes: 23 additions & 1 deletion crates/compiler/src/llvm_jit_backend.rs
Original file line number Diff line number Diff line change
Expand Up @@ -107,6 +107,12 @@ pub struct LLVMJitBackend<'ctx> {
/// [`Self::set_cross_tier_links`].
cross_tier_key: Option<u64>,
global_resolver: Option<std::sync::Arc<dyn Fn(HirId) -> Option<usize> + Send + Sync>>,
/// The body each tier compiles of a module function, when it is not
/// the module context's own; a callee compiled alongside a promoted
/// function takes it. See [`Self::set_body_source`].
#[allow(clippy::type_complexity)]
body_source:
Option<std::sync::Arc<dyn Fn(HirId) -> Option<std::sync::Arc<HirFunction>> + Send + Sync>>,
/// What the module being compiled reaches across tiers, handed to
/// the lowering.
pending_cross_tier: HashMap<HirId, crate::llvm_backend::CrossTierCallee>,
Expand Down Expand Up @@ -207,6 +213,7 @@ impl<'ctx> LLVMJitBackend<'ctx> {
address_taken: std::collections::HashSet::new(),
cross_tier_key: None,
global_resolver: None,
body_source: None,
pending_cross_tier: HashMap::new(),
pending_shared_globals: HashMap::new(),
pending_entry_abi: None,
Expand Down Expand Up @@ -1084,6 +1091,17 @@ impl<'ctx> LLVMJitBackend<'ctx> {
self.global_resolver = Some(globals);
}

/// Where a callee compiled alongside a promoted function takes its
/// body from: `bodies` answers with the body every tier compiles of
/// a function, or `None` for one whose module-context body is that.
#[allow(clippy::type_complexity)]
pub fn set_body_source(
&mut self,
bodies: std::sync::Arc<dyn Fn(HirId) -> Option<std::sync::Arc<HirFunction>> + Send + Sync>,
) {
self.body_source = Some(bodies);
}

/// Compile a single function, together with everything it calls.
///
/// The callees come from the module context, so they are recompiled at
Expand Down Expand Up @@ -1169,7 +1187,11 @@ impl<'ctx> LLVMJitBackend<'ctx> {
if callee == id {
continue;
}
if let Some(f) = ctx.functions.get(&callee) {
let tier_body = self.body_source.as_ref().and_then(|body| body(callee));
if let Some(f) = tier_body {
crate::opt_audit::note_llvm_body(callee, &f);
functions.insert(callee, (*f).clone());
} else if let Some(f) = ctx.functions.get(&callee) {
functions.insert(callee, f.clone());
}
}
Expand Down
102 changes: 102 additions & 0 deletions crates/compiler/src/opt_audit.rs
Original file line number Diff line number Diff line change
@@ -0,0 +1,102 @@
//! Per-function counts of the optimiser's work, for tests that hold
//! each body to one run of the pipeline and every compile above the
//! baseline to that body. Off until [`enable`]; while off, each hook
//! costs one relaxed load.

use std::collections::HashMap;
use std::sync::atomic::{AtomicBool, Ordering};
use std::sync::{Arc, Mutex, OnceLock};

use crate::hir::{HirFunction, HirId, HirModule};

static ON: AtomicBool = AtomicBool::new(false);

#[derive(Default)]
struct Counts {
pipeline: HashMap<HirId, usize>,
finishing: HashMap<HirId, usize>,
llvm: HashMap<HirId, Vec<usize>>,
}

fn counts() -> &'static Mutex<Counts> {
static C: OnceLock<Mutex<Counts>> = OnceLock::new();
C.get_or_init(|| Mutex::new(Counts::default()))
}

/// Start counting, for the rest of the process.
pub fn enable() {
ON.store(true, Ordering::Relaxed);
}

fn on() -> bool {
ON.load(Ordering::Relaxed)
}

/// A run of the whole pipeline over `module`: one for each function it
/// optimises, those not through it already.
pub(crate) fn note_pipeline(module: &HirModule) {
if !on() {
return;
}
let mut c = counts().lock().unwrap();
for (id, f) in &module.functions {
if !f.attributes.optimized && !f.is_external {
*c.pipeline.entry(*id).or_default() += 1;
}
}
}

/// The incremental passes run over `id`, a body that arrived optimised.
pub(crate) fn note_finishing(id: HirId) {
if on() {
*counts().lock().unwrap().finishing.entry(id).or_default() += 1;
}
}

/// `body` handed to the LLVM tier as the HIR of `id`.
#[cfg_attr(not(feature = "llvm-backend"), allow(dead_code))]
pub(crate) fn note_llvm_body(id: HirId, body: &Arc<HirFunction>) {
if on() {
counts()
.lock()
.unwrap()
.llvm
.entry(id)
.or_default()
.push(Arc::as_ptr(body) as usize);
}
}

/// How many runs of the whole pipeline optimised `id`.
pub fn pipeline_runs(id: HirId) -> usize {
counts()
.lock()
.unwrap()
.pipeline
.get(&id)
.copied()
.unwrap_or(0)
}

/// How many times the incremental passes ran over `id`.
pub fn finishing_runs(id: HirId) -> usize {
counts()
.lock()
.unwrap()
.finishing
.get(&id)
.copied()
.unwrap_or(0)
}

/// The address of each body the LLVM tier was handed for `id`, in the
/// order it was handed them.
pub fn llvm_bodies(id: HirId) -> Vec<usize> {
counts()
.lock()
.unwrap()
.llvm
.get(&id)
.cloned()
.unwrap_or_default()
}
Loading
Loading