perf(core): promote per region, promote BEGIN loops, keep loops off the memory stack

Promotion was all-or-nothing per word, so one `.` or one host call put the
whole body -- hot loops included -- on the memory data stack, where a
loop-carried add costs 2.2 ns/iteration instead of 0.31. The stack simulator
now runs over each promotable stretch of a word; BEGIN/UNTIL, BEGIN/AGAIN and
BEGIN/WHILE/REPEAT join DO/LOOP as promotable when the construct is provably
stack-neutral; and the inliner no longer moves a loop-bearing callee into a
caller that can never be promoted.

Fixes a bug the BEGIN work uncovered, present since promotion was introduced
and shipped in 0.2.6: the loop fixup and the IF join copied locals one slot at
a time in index order, so a body that permutes the stack lost a value --
`: C 3 4 2 0 DO SWAP LOOP . . ;` printed `4 4` where gforth prints `4 3`.

Four of five benchmarks now beat sf64: Factorial 0.29x, Collatz 0.30x,
NestedLoops 0.27x, GCD 0.67x. Only Fibonacci is behind, at 1.24x. Also scale
GCD, Factorial and NestedLoops, which ran in 14-51 us where scatter and fixed
costs dominated -- that is what exposed GCD as a loss and pointed at BEGIN.
WS-014, WS-015, WS-016, WS-019.
This commit is contained in:
Oleksandr Kozachuk
2026-08-09 17:25:21 +02:00
parent fc34bd9b24
commit b8dcc021a2
6 changed files with 581 additions and 52 deletions
+64 -3
View File
@@ -53,7 +53,12 @@ pub fn optimize(
// Phase 2: inline then simplify again
if config.inline {
ir = inline(ir, bodies, 8);
// A caller that can never leave the memory data stack would drag an
// inlined loop down with it, so leave those callees where they are:
// as their own word the loop keeps its registers, and one call is far
// cheaper than a loop's worth of memory traffic.
let keep_loops_out = !crate::codegen::promotable_modulo_calls(&ir);
ir = inline(ir, bodies, 8, keep_loops_out);
}
if config.peephole {
ir = peephole(ir);
@@ -496,7 +501,12 @@ fn dce(ops: Vec<IrOp>) -> Vec<IrOp> {
/// Inline small word bodies: replaces `Call(id)` with the word's IR body
/// if the body is small enough and not recursive.
fn inline(ops: Vec<IrOp>, bodies: &HashMap<WordId, Vec<IrOp>>, max_size: usize) -> Vec<IrOp> {
fn inline(
ops: Vec<IrOp>,
bodies: &HashMap<WordId, Vec<IrOp>>,
max_size: usize,
keep_loops_out: bool,
) -> Vec<IrOp> {
let mut out = Vec::new();
for op in ops {
match &op {
@@ -505,6 +515,7 @@ fn inline(ops: Vec<IrOp>, bodies: &HashMap<WordId, Vec<IrOp>>, max_size: usize)
&& body.len() <= max_size
&& !contains_call_to(body, *id)
&& !contains_exit(body)
&& !(keep_loops_out && crate::codegen::contains_loop(body))
{
// Inline the body, recursively converting TailCall back to Call
// (tail position in the callee is not tail position in the caller).
@@ -517,7 +528,7 @@ fn inline(ops: Vec<IrOp>, bodies: &HashMap<WordId, Vec<IrOp>>, max_size: usize)
}
_ => {
out.push(apply_to_bodies(op, &|inner| {
inline(inner, bodies, max_size)
inline(inner, bodies, max_size, keep_loops_out)
}));
}
}
@@ -1012,4 +1023,54 @@ mod tests {
let result = optimize(vec![IrOp::Call(WordId(5))], &config, &bodies);
assert_eq!(result, vec![IrOp::Call(WordId(5))]);
}
#[test]
fn keeps_a_loop_out_of_a_caller_stuck_on_the_memory_stack() {
// The caller has a `.`, so it can never leave the memory data stack.
// Inlining the loop would drag it down too; as its own word the loop
// keeps its registers and the caller just pays one call.
let mut bodies = HashMap::new();
bodies.insert(
WordId(5),
vec![IrOp::DoLoop {
body: vec![IrOp::PushI32(1), IrOp::Add],
is_plus_loop: false,
}],
);
let result = opt_with_inline(vec![IrOp::Call(WordId(5)), IrOp::Dot], &bodies);
assert!(
matches!(result.first(), Some(IrOp::Call(WordId(5)))),
"loop should not have been inlined, got {result:?}"
);
}
#[test]
fn still_inlines_a_loop_into_a_caller_that_can_be_promoted() {
let mut bodies = HashMap::new();
bodies.insert(
WordId(5),
vec![IrOp::DoLoop {
body: vec![IrOp::PushI32(1), IrOp::Add],
is_plus_loop: false,
}],
);
let result = opt_with_inline(vec![IrOp::Call(WordId(5)), IrOp::Dup], &bodies);
assert!(
!result.iter().any(|op| matches!(op, IrOp::Call(_))),
"loop should have been inlined, got {result:?}"
);
}
#[test]
fn still_inlines_straight_line_words_anywhere() {
// Only loops are held back; a small straight-line word is still
// better off inlined even into an unpromotable caller.
let mut bodies = HashMap::new();
bodies.insert(WordId(5), vec![IrOp::Dup, IrOp::Mul]);
let result = opt_with_inline(vec![IrOp::Call(WordId(5)), IrOp::Dot], &bodies);
assert!(
!result.iter().any(|op| matches!(op, IrOp::Call(_))),
"straight-line word should still inline, got {result:?}"
);
}
}