From 311f4273b473b8bfd21795df55a120d67e2033d4 Mon Sep 17 00:00:00 2001 From: nella Date: Fri, 25 Sep 2026 21:12:37 +0200 Subject: [PATCH 01/12] Load the indirect call target last. --- src/backend/pass/emit/x86/x86_lower_call.cpp | 10 ++++++---- .../test/correctness/custom/call_indirect_x87_arg.c | 12 ++++++++++++ 2 files changed, 18 insertions(+), 4 deletions(-) create mode 100644 src/compiler/test/correctness/custom/call_indirect_x87_arg.c diff --git a/src/backend/pass/emit/x86/x86_lower_call.cpp b/src/backend/pass/emit/x86/x86_lower_call.cpp index 8fd4533..0263380 100644 --- a/src/backend/pass/emit/x86/x86_lower_call.cpp +++ b/src/backend/pass/emit/x86/x86_lower_call.cpp @@ -167,10 +167,9 @@ namespace rat { call.isCall = true; call.clobbers = callerSavedClobbers(); - if(c->isIndirect()) { - VReg t = gpValue(c->getTarget()); - mov(R11, t); - } + VReg target = kNoVReg; + if(c->isIndirect()) + target = gpValue(c->getTarget()); // classify and materialize every argument up front X86ArgAssigner as(*conv); @@ -233,6 +232,9 @@ namespace rat { if(al.reg < 0) call.uses.push_back(al.val); call.imm = (I64)as.stackBytes; + // R11 last: argument setup may use it as encoder scratch (x87 constants) + if(c->isIndirect()) + mov(R11, target); emit(std::move(call)); // return value diff --git a/src/compiler/test/correctness/custom/call_indirect_x87_arg.c b/src/compiler/test/correctness/custom/call_indirect_x87_arg.c new file mode 100644 index 0000000..6e24a8c --- /dev/null +++ b/src/compiler/test/correctness/custom/call_indirect_x87_arg.c @@ -0,0 +1,12 @@ +// expect: 0 +// passes: + +long double g; +__attribute__((noinline)) void take(long double x) { g = x; } +void (*volatile fp)(long double) = take; + +int main(void) { + void (*f)(long double) = fp; + f(1.5L); + return g == 1.5L ? 0 : 1; +} From 3281be3325334e5cd7d89b4c41140e54e05e4ad6 Mon Sep 17 00:00:00 2001 From: nella Date: Fri, 25 Sep 2026 21:58:04 +0200 Subject: [PATCH 02/12] Never allocate scratch registers. --- src/backend/codegen/reg_alloc_base.cpp | 35 +------------------ src/backend/codegen/reg_alloc_base.h | 1 - .../custom/regalloc_stackrestore_r10.c | 30 ++++++++++++++++ 3 files changed, 31 insertions(+), 35 deletions(-) create mode 100644 src/compiler/test/correctness/custom/regalloc_stackrestore_r10.c diff --git a/src/backend/codegen/reg_alloc_base.cpp b/src/backend/codegen/reg_alloc_base.cpp index b4382d8..e86246a 100644 --- a/src/backend/codegen/reg_alloc_base.cpp +++ b/src/backend/codegen/reg_alloc_base.cpp @@ -301,13 +301,6 @@ namespace rat { return false; } - B32 RegAllocBase::isAllocatable(const RegClass& rc, PhysReg p) { - for(PhysReg c : rc.allocatable) - if(c == p) - return true; - return false; - } - void RegAllocBase::collectRematDefs() { if(!hooks->isRemat) return; @@ -533,33 +526,7 @@ namespace rat { collectCopyHints(); collectRematDefs(); - // optimistic first pass: spill-scratch regs join the allocatable pool; if - // nothing spills keep them, else re-solve with scratch reserved for rewrite - const RegisterInfo* realRi = ri; - RegisterInfo wide = *ri; - B32 widened = false; - for(RegClass& rc : wide.classes) - for(PhysReg p : rc.scratch) - if(!isAllocatable(rc, p)) { - rc.allocatable.push_back(p); - widened = true; - } - U32 savedFrameBytes = fn->frameBytes; - if(widened) { - ri = &wide; - solve(); - ri = realRi; - if(anySpilled()) { - // roll back and re-solve with the normal register set - fn->frameBytes = savedFrameBytes; - slotPool.clear(); - usedCallee.clear(); - resetState(); - solve(); - } - } else { - solve(); - } + solve(); assignPieces(); rewrite(); diff --git a/src/backend/codegen/reg_alloc_base.h b/src/backend/codegen/reg_alloc_base.h index 8cfb5de..05832fb 100644 --- a/src/backend/codegen/reg_alloc_base.h +++ b/src/backend/codegen/reg_alloc_base.h @@ -119,7 +119,6 @@ namespace rat { U32 classOf(VReg v) const; const RegClass& regClass(U32 cls) const; static B32 isCalleeSaved(const RegClass& rc, PhysReg p); - static B32 isAllocatable(const RegClass& rc, PhysReg p); // a slot just written from a register nothing has touched since struct Memo { B32 on = false; diff --git a/src/compiler/test/correctness/custom/regalloc_stackrestore_r10.c b/src/compiler/test/correctness/custom/regalloc_stackrestore_r10.c new file mode 100644 index 0000000..45e61b1 --- /dev/null +++ b/src/compiler/test/correctness/custom/regalloc_stackrestore_r10.c @@ -0,0 +1,30 @@ +// expect: 0 +// output: +//| 2420 +// passes: + +long f(long n, long a) { + long x1, x2, x3, x4, x5, x6, x7, x8, x9, x10, x11; + { + long v[n]; + __asm__ volatile("" ::"r"(v) : "memory"); + x1 = a * 3 + 1; + x2 = a * 4 + 2; + x3 = a * 5 + 3; + x4 = a * 6 + 4; + x5 = a * 7 + 5; + x6 = a * 8 + 6; + x7 = a * 9 + 7; + x8 = a * 10 + 8; + x9 = a * 11 + 9; + x10 = a * 12 + 10; + x11 = a * 13 + 11; + } + return x1 * 1 + x2 * 2 + x3 * 3 + x4 * 4 + x5 * 5 + x6 * 6 + x7 * 7 + x8 * 8 + x9 * 9 + x10 * 10 + + x11 * 11; +} + +int main(void) { + __builtin_printf("%ld\n", f(7, 3)); + return 0; +} From aa5cce78e2948f636e32a7b09c96a58c86dbcc48 Mon Sep 17 00:00:00 2001 From: nella Date: Fri, 25 Sep 2026 22:41:19 +0200 Subject: [PATCH 03/12] New RA. --- src/backend/codegen/reg_alloc.cpp | 329 ++++++++++++++++++++++++++++++ src/backend/codegen/reg_alloc.h | 70 +++++++ 2 files changed, 399 insertions(+) create mode 100644 src/backend/codegen/reg_alloc.cpp create mode 100644 src/backend/codegen/reg_alloc.h diff --git a/src/backend/codegen/reg_alloc.cpp b/src/backend/codegen/reg_alloc.cpp new file mode 100644 index 0000000..43d116f --- /dev/null +++ b/src/backend/codegen/reg_alloc.cpp @@ -0,0 +1,329 @@ +#include "codegen/reg_alloc.h" + +#include + +#include "codegen/machine_module.h" +#include "ir/module.h" +#include "target/target.h" + +namespace rat { + namespace detail { + // per-loop-level operand weight + constexpr U32 kLoopUseWeight = 3; + constexpr U32 kMaxUseWeight = 100000; + + PhysReg firstFree(const List& regs, U64 blocked) { + for(PhysReg p : regs) + if(!((blocked >> p) & 1)) + return p; + return kNoReg; + } + } // namespace detail + + detail::RegAllocFunc::RegAllocFunc(MachineFunc& f, const RegisterInfo& r, const RegAllocHooks& h) + : fn(f), + ri(r), + hooks(h), + nv(f.nextVReg) {} + + void detail::RegAllocFunc::run() { + for(const RegClass& rc : ri.classes) { + for(PhysReg p : rc.allocatable) + allocMask[rc.id] |= (U64)1 << p; + for(PhysReg p : rc.calleeSaved) + calleeMask |= (U64)1 << p; + } + number(); + liveness(); + buildIntervals(); + assignRegs(); + rewrite(); + fn.usedCalleeSaved.clear(); + for(U64 m = usedCallee; m; m &= m - 1) + fn.usedCalleeSaved.push_back((PhysReg)countTrailingZeros64(m)); + } + + B32 detail::RegAllocFunc::isCopy(const MachineInstr& in) const { + return hooks.isCopy && in.defs.size() == 1 && in.uses.size() == 1 && hooks.isCopy(in); + } + + void detail::RegAllocFunc::number() { + blockFirst.assign(1, 0); + for(const MachineBlock& blk : fn.blocks) + blockFirst.push_back(blockFirst.back() + (U32)blk.insts.size()); + busy.assign(2 * (U64)blockFirst.back(), 0); + for(U32 b = 0; b < fn.blocks.size(); ++b) + pinFixed(b); + } + + // a fixed register is busy from its def to its last use in the block (call argument + // windows, div/shift operands, incoming arguments up to their copy), plus clobbers + void detail::RegAllocFunc::pinFixed(U32 b) { + const List& insts = fn.blocks[b].insts; + U64 live = 0; + for(U32 k = (U32)insts.size(); k-- > 0;) { + const MachineInstr& in = insts[k]; + U64 u = 2 * (U64)(blockFirst[b] + k); + U64 defs = 0; + U64 uses = 0; + U64 clob = 0; + for(const MachineOperand& o : in.defs) + if(o.isPhys()) + defs |= (U64)1 << o.phys; + for(const MachineOperand& o : in.uses) + if(o.isPhys()) + uses |= (U64)1 << o.phys; + for(PhysReg p : in.clobbers) + clob |= (U64)1 << p; + assert((in.isCall || !(clob & live)) && "clobber inside a fixed-register window"); + busy[u + 1] |= live | defs | clob; + live &= ~(defs | clob); + busy[u] |= live | uses | clob; + if(!isCopy(in)) + busy[u + 1] |= uses; + live |= uses; + } + } + + void detail::RegAllocFunc::liveness() { + U32 nb = (U32)fn.blocks.size(); + List defStamp(nv, 0); + List ueStamp(nv, 0); + List> defBlocks(nv); + List> ueBlocks(nv); + for(U32 b = 0; b < nb; ++b) + for(const MachineInstr& in : fn.blocks[b].insts) { + for(const MachineOperand& o : in.uses) + if(o.isVReg() && defStamp[o.vreg] != b + 1 && ueStamp[o.vreg] != b + 1) { + ueStamp[o.vreg] = b + 1; + ueBlocks[o.vreg].push_back(b); + } + for(const MachineOperand& o : in.defs) + if(o.isVReg() && defStamp[o.vreg] != b + 1) { + defStamp[o.vreg] = b + 1; + defBlocks[o.vreg].push_back(b); + } + } + List defIn(nb, kNoVReg); + List liveIn(nb, kNoVReg); + List outStamp(nb, kNoVReg); + List work; + liveOut.assign(nb, {}); + for(VReg v = 1; v < nv; ++v) { + for(U32 b : defBlocks[v]) + defIn[b] = v; + for(U32 b : ueBlocks[v]) { + liveIn[b] = v; + work.push_back(b); + } + while(!work.empty()) { + U32 x = work.back(); + work.pop_back(); + for(I32 p : fn.blocks[x].preds) { + if(outStamp[p] != v) { + outStamp[p] = v; + liveOut[p].push_back(v); + } + if(defIn[p] != v && liveIn[p] != v) { + liveIn[p] = v; + work.push_back(p); + } + } + } + } + } + + // segments come in descending order per vreg, merge touching ones + void detail::RegAllocFunc::addSeg(VReg v, I32 start, I32 end) { + List& segs = iv[v].segs; + if(!segs.empty() && end + 1 >= segs.back().first) + segs.back().first = std::min(segs.back().first, start); + else + segs.emplace_back(start, end); + } + + // backward walk per block from its live-out set, blocks in reverse + void detail::RegAllocFunc::buildIntervals() { + iv.resize(nv); + List live(nv, 0); + List segEnd(nv, 0); + List liveList; + for(U32 b = (U32)fn.blocks.size(); b-- > 0;) { + const MachineBlock& blk = fn.blocks[b]; + if(blk.insts.empty()) + continue; + for(VReg v : liveOut[b]) { + live[v] = 1; + segEnd[v] = 2 * (I32)blockFirst[b + 1] - 1; + liveList.push_back(v); + } + U32 weight = 1; + for(I32 d = blk.loopDepth; d > 0 && weight < detail::kMaxUseWeight; --d) + weight *= detail::kLoopUseWeight; + for(U32 k = (U32)blk.insts.size(); k-- > 0;) { + const MachineInstr& in = blk.insts[k]; + I32 u = 2 * (I32)(blockFirst[b] + k); + for(const MachineOperand& o : in.defs) { + if(!o.isVReg()) + continue; + iv[o.vreg].weight += (F32)weight; + I32 end = u + 1; // a dead def still takes its slot + if(live[o.vreg]) + end = segEnd[o.vreg]; + addSeg(o.vreg, u + 1, end); + live[o.vreg] = 0; + } + I32 useEnd = u + 1; + if(isCopy(in)) + useEnd = u; + for(const MachineOperand& o : in.uses) { + if(!o.isVReg()) + continue; + iv[o.vreg].weight += (F32)weight; + if(live[o.vreg]) + continue; + live[o.vreg] = 1; + segEnd[o.vreg] = useEnd; + liveList.push_back(o.vreg); + } + } + for(VReg v : liveList) + if(live[v]) { + addSeg(v, 2 * (I32)blockFirst[b], segEnd[v]); + live[v] = 0; + } + liveList.clear(); + } + for(RaInterval& t : iv) { + std::reverse(t.segs.begin(), t.segs.end()); + I32 len = 0; + for(const auto& [start, end] : t.segs) + len += end - start + 1; + if(len) + t.weight /= std::sqrt((F32)len); + } + } + + PhysReg detail::RegAllocFunc::pick(VReg v) const { + const RegClass& rc = ri.classes[fn.vregClass[v]]; + U64 blocked = ~allocMask[rc.id]; + for(const auto& [start, end] : iv[v].segs) + for(I32 s = start; s <= end && blocked != ~0ull; ++s) + blocked |= busy[(U64)s]; + return detail::firstFree(rc.allocatable, blocked); + } + + void detail::RegAllocFunc::assignRegs() { + List> order; // (-weight, vreg) + for(VReg v = 1; v < nv; ++v) + if(!iv[v].segs.empty()) + order.emplace_back(-iv[v].weight, v); + std::sort(order.begin(), order.end()); + for(const auto& [negWeight, v] : order) { + RaInterval& t = iv[v]; + assert(!ri.classes[fn.vregClass[v]].scratch.empty() && "vreg of a class with no scratch"); + t.reg = pick(v); + if(t.reg == kNoReg) { + const RegClass& rc = ri.classes[fn.vregClass[v]]; + U32 bytes = rc.spillBytes; + if(!bytes) + bytes = ri.spillSlotBytes; + t.slot = hooks.allocSlot(fn, rc.id, bytes); + continue; + } + usedCallee |= ((U64)1 << t.reg) & calleeMask; + for(const auto& [start, end] : t.segs) + for(I32 s = start; s <= end; ++s) + busy[(U64)s] |= (U64)1 << t.reg; + } + } + + PhysReg detail::RegAllocFunc::pickTemp(U32 cls, U64 hard, U64 soft) { + const RegClass& rc = ri.classes[cls]; + PhysReg p = detail::firstFree(rc.scratch, hard | soft); + if(p == kNoReg) + p = detail::firstFree(rc.allocatable, hard | soft); + if(p == kNoReg) + p = detail::firstFree(rc.scratch, hard); + assert(p != kNoReg && "no free register for a spilled operand"); + usedCallee |= ((U64)1 << p) & calleeMask; + return p; + } + + PhysReg + detail::RegAllocFunc::spillReg(List& out, const MachineOperand& o, U32 i, B32 use) { + for(const auto& [v, r] : temps) + if(v == o.vreg) + return r; + U32 cls = fn.vregClass[o.vreg]; + U64 hard = (busy[2 * (U64)i] | busy[2 * (U64)i + 1]) & ~own; + if(use) + hard |= taken; + PhysReg r = pickTemp(cls, hard, own | taken); + if(use) + out.push_back(hooks.makeReload(r, iv[o.vreg].slot, cls, o.width)); + taken |= (U64)1 << r; + temps.emplace_back(o.vreg, r); + return r; + } + + void detail::RegAllocFunc::rewriteInstr(List& out, MachineInstr& in, U32 i) { + temps.clear(); + taken = 0; + own = 0; + for(PhysReg p : in.clobbers) + own |= (U64)1 << p; + stores.clear(); + for(MachineOperand& o : in.uses) { + if(!o.isVReg()) + continue; + PhysReg r = iv[o.vreg].reg; + if(r == kNoReg && in.isCall) { // the call reads the slot itself + o = MachineOperand::frameSlot(iv[o.vreg].slot, o.width); + continue; + } + if(r == kNoReg) + r = spillReg(out, o, i, true); + o = MachineOperand::fixed(r, o.width); + } + for(MachineOperand& o : in.defs) { + if(!o.isVReg()) + continue; + const RaInterval& t = iv[o.vreg]; + PhysReg r = t.reg; + if(r == kNoReg) { + r = spillReg(out, o, i, false); + stores.push_back(hooks.makeSpill(t.slot, r, fn.vregClass[o.vreg], o.width)); + } + o = MachineOperand::fixed(r, o.width); + } + out.push_back(std::move(in)); + for(MachineInstr& s : stores) + out.push_back(std::move(s)); + } + + void detail::RegAllocFunc::rewrite() { + List out; + for(U32 b = 0; b < fn.blocks.size(); ++b) { + List& insts = fn.blocks[b].insts; + out.clear(); + out.reserve(insts.size()); + for(U32 k = 0; k < insts.size(); ++k) { + MachineInstr& in = insts[k]; + U32 i = blockFirst[b] + k; + rewriteInstr(out, in, i); + } + insts.swap(out); + } + } + + B32 RegAllocPass::run(Module& module, MachineModule& mm, const TargetInfo& target) { + B32 changed = false; + RegAllocHooks hooks = target.regAllocHooks(); + for(const Function* f : module) { + detail::RegAllocFunc(mm.get(f), *target.registers(), hooks).run(); + changed = true; + } + return changed; + } +} // namespace rat diff --git a/src/backend/codegen/reg_alloc.h b/src/backend/codegen/reg_alloc.h new file mode 100644 index 0000000..3666c6f --- /dev/null +++ b/src/backend/codegen/reg_alloc.h @@ -0,0 +1,70 @@ +#ifndef RAT_CODEGEN_REGALLOC_H +#define RAT_CODEGEN_REGALLOC_H + +#include "core.h" + +#include "codegen/machine_function.h" +#include "pass/pass.h" + +namespace rat { + namespace detail { + PhysReg firstFree(const List& regs, U64 blocked); + + using RaSeg = Pair; + + struct RaInterval { + List segs; + F32 weight = 0; + PhysReg reg = kNoReg; + I32 slot = 0; + }; + + struct RegAllocFunc { + RegAllocFunc(MachineFunc& f, const RegisterInfo& r, const RegAllocHooks& h); + void run(); + private: + // intervals + void number(); + void pinFixed(U32 b); + void liveness(); + void buildIntervals(); + void addSeg(VReg v, I32 start, I32 end); + // assignment + void assignRegs(); + PhysReg pick(VReg v) const; + // rewrite + void rewrite(); + void rewriteInstr(List& out, MachineInstr& in, U32 i); + PhysReg spillReg(List& out, const MachineOperand& o, U32 i, B32 use); + PhysReg pickTemp(U32 cls, U64 hard, U64 soft); + // queries + B32 isCopy(const MachineInstr& in) const; + private: + MachineFunc& fn; + const RegisterInfo& ri; + const RegAllocHooks& hooks; + U32 nv; + U64 allocMask[kMaxRegClasses] = {}; + U64 calleeMask = 0; + // numbering + List blockFirst; // block -> first instruction, one past the end at the back + List busy; // slot -> busy physical registers + U64 usedCallee = 0; + // liveness + List> liveOut; // block -> live-out vregs + List iv; + // rewrite of the current instruction + List> temps; // spilled vreg -> its temp + U64 taken = 0; // temps + U64 own = 0; // clobbers + List stores; + }; + } // namespace detail + + struct RegAllocPass : MachinePass { + const C8* name() const override { return "regalloc"; } + B32 run(Module& module, MachineModule& mm, const TargetInfo& target) override; + }; +} // namespace rat + +#endif From a6940c448ddbf5b6e65f4abd66e9325257c06972 Mon Sep 17 00:00:00 2001 From: nella Date: Fri, 25 Sep 2026 23:27:53 +0200 Subject: [PATCH 04/12] Replace old RA. --- src/backend/codegen/linear_scan_reg_alloc.cpp | 514 ---------------- src/backend/codegen/linear_scan_reg_alloc.h | 116 ---- src/backend/codegen/reg_alloc_base.cpp | 552 ------------------ src/backend/codegen/reg_alloc_base.h | 176 ------ src/backend/pass/pass_registry.cpp | 4 +- src/backend/rat.h | 2 +- src/compiler/compile.cpp | 2 +- 7 files changed, 4 insertions(+), 1362 deletions(-) delete mode 100644 src/backend/codegen/linear_scan_reg_alloc.cpp delete mode 100644 src/backend/codegen/linear_scan_reg_alloc.h delete mode 100644 src/backend/codegen/reg_alloc_base.cpp delete mode 100644 src/backend/codegen/reg_alloc_base.h diff --git a/src/backend/codegen/linear_scan_reg_alloc.cpp b/src/backend/codegen/linear_scan_reg_alloc.cpp deleted file mode 100644 index 74971d0..0000000 --- a/src/backend/codegen/linear_scan_reg_alloc.cpp +++ /dev/null @@ -1,514 +0,0 @@ -#include "codegen/linear_scan_reg_alloc.h" - -#include - -#include "target/target.h" - -namespace rat { - void LinearScanRegAllocPass::solve() { - pinsByPoint.clear(); - for(U32 pt = 0; pt < (U32)fixedAt.size(); ++pt) - if(fixedAt[pt]) - pinsByPoint.emplace_back((I32)pt, fixedAt[pt]); - - buildIntervals(); - assignRegs(); - assignSpillSlots(); - } - - RegAllocBase::Assignment LinearScanRegAllocPass::assignmentOf(VReg v) { - if(const Interval* iv = ivFind(v)) - return {iv->assigned, iv->spillSlot, iv->cls, iv->spilled}; - return {kNoReg, 0, 0, false}; - } - - void LinearScanRegAllocPass::buildIntervals() { - liveness(); - - if(intervals.size() < fn->nextVReg) - intervals.resize(fn->nextVReg); - - auto ivFor = [&](VReg v) -> Interval& { - Interval& iv = intervals[v]; - if(iv.vreg == kNoVReg) { - iv.vreg = v; - iv.cls = classOf(v); - } - return iv; - }; - - // backward walk per block - segEnd.assign(fn->nextVReg, 0); - live.resetAll(fn->nextVReg); - for(U32 b = 0; b < fn->blocks.size(); ++b) { - if(blkPts[b].empty()) - continue; - I32 first = (I32)blkPts[b].front(); - I32 last = (I32)blkPts[b].back(); - - if(liveIsDense) { - live.copyFrom(denseLive.out[b]); - live.forEach([&](VReg v) { segEnd[v] = last; }); - } else { - for(VReg v : sparseLive.out[b]) { - live.set(v); - segEnd[v] = last; - } - } - - // an inner-loop reference costs more than a straight-line one - U32 weight = 1; - for(I32 d = fn->blocks[b].loopDepth; d > 0 && weight < kMaxUseWeight; --d) - weight *= kLoopUseWeight; - - for(I32 i = (I32)fn->blocks[b].insts.size() - 1; i >= 0; --i) { - I32 pt = (I32)blkPts[b][(U32)i]; - const MachineInstr& in = fn->blocks[b].insts[(U32)i]; - for(const MachineOperand& d : in.defs) - if(d.isVReg()) { - ivFor(d.vreg).uses += weight; - if(live.test(d.vreg)) { - ivFor(d.vreg).segs.push_back({pt, segEnd[d.vreg]}); - live.reset(d.vreg); - } else { - ivFor(d.vreg).segs.push_back({pt, pt}); // dead def still occupies its point - } - } - for(const MachineOperand& u : in.uses) - if(u.isVReg()) { - ivFor(u.vreg).uses += weight; - if(!live.test(u.vreg)) { - live.set(u.vreg); - segEnd[u.vreg] = pt; - } - } - } - if(liveIsDense) - live.forEach([&](VReg v) { ivFor(v).segs.push_back({first, segEnd[v]}); }); - else - for(VReg v : sparseLive.in[b]) { - ivFor(v).segs.push_back({first, segEnd[v]}); - live.reset(v); // leaves the set empty for the next block - } - } - - for(U32 v = 1; v < fn->nextVReg; ++v) { - Interval& iv = intervals[v]; - if(!iv.live()) - continue; - coalesceSegs(iv); - U32 c = 0; - for(const Seg& sg : iv.segs) { - while(c < callPts.size() && callPts[c] <= sg.start) - ++c; - if(c < callPts.size() && callPts[c] < sg.end) { - iv.crossesCall = true; - break; - } - } - } - } - - void LinearScanRegAllocPass::coalesceSegs(Interval& iv) { - if(iv.segs.empty()) - return; - std::sort(iv.segs.begin(), iv.segs.end(), [](const Seg& a, const Seg& b) { - return a.start != b.start ? a.start < b.start : a.end < b.end; - }); - // merge in place - U32 w = 0; - for(U32 r = 0; r < iv.segs.size(); ++r) { - const Seg sg = iv.segs[r]; - if(w && sg.start <= iv.segs[w - 1].end + 1) - iv.segs[w - 1].end = std::max(iv.segs[w - 1].end, sg.end); - else - iv.segs[w++] = sg; - } - iv.segs.resize(w); - iv.start = iv.segs.front().start; - iv.end = iv.segs.back().end; - } - - B32 LinearScanRegAllocPass::overlapOnlyAt(const Interval& a, - const Interval& b, - const List& pts) { - U32 i = 0, j = 0; - while(i < (U32)a.segs.size() && j < (U32)b.segs.size()) { - I32 lo = std::max(a.segs[i].start, b.segs[j].start); - I32 hi = std::min(a.segs[i].end, b.segs[j].end); - if(lo <= hi) { - if(lo != hi) - return false; // a real overlapping range, not a single touch point - B32 isCopyPt = false; - for(I32 p : pts) - if(p == lo) - isCopyPt = true; - if(!isCopyPt) - return false; - } - if(a.segs[i].end < b.segs[j].end) - ++i; - else - ++j; - } - return true; - } - - U64 LinearScanRegAllocPass::forbidden(const Interval& iv) const { - U64 bad = 0; - // one exemption lookup per vreg; most vregs have no copy pins at all - auto vt = copyPinAt.find(iv.vreg); - const Map* exempt = vt == copyPinAt.end() ? nullptr : &vt->second; - for(const Seg& sg : iv.segs) { - auto lo = std::lower_bound(pinsByPoint.begin(), - pinsByPoint.end(), - sg.start, - [](const auto& a, I32 pt) { return a.first < pt; }); - for(auto it = lo; it != pinsByPoint.end() && it->first <= sg.end; ++it) { - U64 m = it->second; - if(exempt) { - auto pi = exempt->find(it->first); - if(pi != exempt->end()) - m &= ~((U64)1 << pi->second); - } - bad |= m; - } - } - return bad; - } - - void LinearScanRegAllocPass::assignRegs() { - List sorted; - sorted.reserve(fn->nextVReg); - for(U32 v = 1; v < fn->nextVReg; ++v) - if(intervals[v].live()) - sorted.push_back(&intervals[v]); - std::sort(sorted.begin(), sorted.end(), [](const Interval* a, const Interval* b) { - return a->start != b->start ? a->start < b->start : a->vreg < b->vreg; - }); - - List active; - U32 maxCls = 0; - for(const RegClass& rc : ri->classes) - maxCls = std::max(maxCls, rc.id); - freeRegs.assign(maxCls + 1, 0); - for(const RegClass& rc : ri->classes) - for(PhysReg p : rc.allocatable) - freeRegs[rc.id] |= (U64)1 << p; - - auto expire = [&](I32 start) { - // compact in place - expiredBuf.clear(); - U32 w = 0; - for(U32 r = 0; r < active.size(); ++r) { - Interval* a = active[r]; - if(a->end < start) - expiredBuf.push_back(a); - else - active[w++] = a; - } - active.resize(w); - // move partners may share a register, it becomes free only when its last active holder - // expires - for(const Interval* e : expiredBuf) { - if(e->spilled || e->assigned == kNoReg) - continue; - B32 stillHeld = false; - for(const Interval* a : active) - if(a->assigned == e->assigned && !a->spilled) { - stillHeld = true; - break; - } - if(!stillHeld) - freeRegs[e->cls] |= (U64)1 << e->assigned; - } - }; - - for(Interval* iv : sorted) { - expire(iv->start); - - const RegClass& rc = regClass(iv->cls); - U64& pool = freeRegs[iv->cls]; - U64 bad = forbidden(*iv); - - // biased pick - PhysReg pick = kNoReg; - auto colorOf = [&](VReg p) { - const Interval* pi = ivFind(p); - return pi ? pi->assigned : kNoReg; - }; - for(PhysReg h : hintedRegs(iv->vreg, colorOf)) { - if((bad >> h) & 1) - continue; - if(iv->crossesCall && !isCalleeSaved(rc, h)) - continue; - if(isCalleeSaved(rc, h) && !usedCallee.count(h) && !iv->crossesCall) - continue; - if((pool >> h) & 1) { - pick = h; - break; - } - B32 shareable = false; - for(const Interval* a : active) - if(a->assigned == h && !a->spilled) { - shareable = - a->cls == iv->cls && overlapOnlyAt(*iv, *a, copyPointsBetween(iv->vreg, a->vreg)); - if(!shareable) - break; - } - if(shareable) { - pick = h; - break; - } - } - - if(pick == kNoReg) { - if(iv->crossesCall) { - for(PhysReg p : rc.calleeSaved) - if(((pool >> p) & 1) && !((bad >> p) & 1)) { - pick = p; - break; - } - } else { - // short-lived value - for(PhysReg p : rc.allocatable) - if(((pool >> p) & 1) && !((bad >> p) & 1)) { - pick = p; - break; - } - } - } - - if(pick != kNoReg) { - pool &= ~((U64)1 << pick); - iv->assigned = pick; - if(isCalleeSaved(rc, pick)) - usedCallee.insert(pick); - active.push_back(iv); - } else { - spillAt(iv, active); - } - } - } - - void LinearScanRegAllocPass::spillAt(Interval* cur, List& active) { - const RegClass& rc = regClass(cur->cls); - U64 bad = forbidden(*cur); - - // spill cost ~ memory ops added: one reload per use; remat values reload - // without a store so cost far less - auto costOf = [&](const Interval* iv) -> F64 { - F64 c = (F64)(iv->uses ? iv->uses : 1); - if(rematDef.find(iv->vreg) != rematDef.end()) - c *= 0.3; - I32 span = 0; - for(const Seg& sg : iv->segs) - span += sg.end - sg.start + 1; - // a dense temp still wins its register, but a long-lived hot value (loop state) is - // not written off just for being long - return c / std::sqrt((F64)(span > 0 ? span : 1)); - }; - - Interval* victim = nullptr; - for(Interval* a : active) - if(a->cls == cur->cls && !a->spilled) { - if(cur->crossesCall && !isCalleeSaved(rc, a->assigned)) - continue; - if((bad >> a->assigned) & 1) - continue; - B32 shared = false; - for(const Interval* o : active) - if(o != a && o->assigned == a->assigned && !o->spilled) { - shared = true; - break; - } - if(shared) - continue; - if(!victim || costOf(a) < costOf(victim) || - (costOf(a) == costOf(victim) && a->end > victim->end)) - victim = a; - } - - if(victim && costOf(victim) < costOf(cur)) { - cur->assigned = victim->assigned; - if(isCalleeSaved(rc, cur->assigned)) - usedCallee.insert(cur->assigned); - victim->assigned = kNoReg; - victim->spilled = true; - for(Interval*& a : active) - if(a == victim) - a = cur; - } else { - cur->spilled = true; - } - } - - void LinearScanRegAllocPass::assignPieces() { - if(!anySpilled()) - return; - List busy(fixedAt.begin(), fixedAt.end()); - for(U32 v = 1; v < fn->nextVReg; ++v) { - const Interval& iv = intervals[v]; - if(!iv.live() || iv.spilled || iv.assigned == kNoReg) - continue; - for(const Seg& sg : iv.segs) - for(I32 p = sg.start; p <= sg.end; ++p) - busy[(U32)p] |= (U64)1 << iv.assigned; - } - // every operand of a spilled vreg, sorted by vreg then program order - List touches; - for(U32 b = 0; b < fn->blocks.size(); ++b) - for(U32 i = 0; i < fn->blocks[b].insts.size(); ++i) { - const MachineInstr& in = fn->blocks[b].insts[i]; - if(in.isCall) - continue; // reads its slot directly - I32 pt = (I32)blkPts[b][i]; - for(const MachineOperand& o : in.uses) - if(o.isVReg() && intervals[o.vreg].spilled) - touches.emplace_back(o.vreg, pt); - for(const MachineOperand& o : in.defs) - if(o.isVReg() && intervals[o.vreg].spilled) - touches.emplace_back(o.vreg, pt); - } - std::sort(touches.begin(), touches.end()); - for(U32 i = 0; i < touches.size();) { - U32 j = i; - while(j < touches.size() && touches[j].first == touches[i].first) - ++j; - cutPieces(&touches[i], j - i, busy); - i = j; - } - } - - void LinearScanRegAllocPass::cutPieces(const Touch* pts, U32 count, List& busy) { - VReg v = pts[0].first; - U32 cls = intervals[v].cls; - U32 i = 0; - while(i < count) { - I32 start = pts[i].second; - I32 end = start; - U32 block = order[(U32)start].block; - U32 j = i + 1; - while(j < count && order[(U32)pts[j].second].block == block) { - auto call = std::upper_bound(callPts.begin(), callPts.end(), end); - if(call != callPts.end() && *call < pts[j].second) - break; - end = pts[j].second; - ++j; - } - i = j; - if(end == start) - continue; - PhysReg reg = pickPieceReg(cls, start, end, busy); - if(reg != kNoReg) - pieces.push_back({v, start, end, reg}); - } - } - - PhysReg LinearScanRegAllocPass::pickPieceReg(U32 cls, I32 start, I32 end, List& busy) { - const RegClass& rc = regClass(cls); - U64 used = 0; - for(I32 p = start; p <= end; ++p) - used |= busy[(U32)p]; - for(PhysReg p : rc.allocatable) { - if((used >> p) & 1) - continue; - if(isCalleeSaved(rc, p) && !usedCallee.count(p)) - continue; - for(I32 q = start; q <= end; ++q) - busy[(U32)q] |= (U64)1 << p; - return p; - } - return kNoReg; - } - - B32 LinearScanRegAllocPass::pieceAfter(const Touch& key, const Piece& pc) { - return key.first < pc.vreg || (key.first == pc.vreg && key.second < pc.start); - } - - RegAllocBase::Piece* LinearScanRegAllocPass::pieceAt(VReg v, I32 pt) { - auto after = std::upper_bound(pieces.begin(), pieces.end(), Touch{v, pt}, pieceAfter); - if(after == pieces.begin()) - return nullptr; - Piece& pc = *(after - 1); - return pc.vreg == v && pt <= pc.end ? &pc : nullptr; - } - - VReg LinearScanRegAllocPass::webFind(VReg v) { - while(webParent[v] != v) - v = webParent[v] = webParent[webParent[v]]; - return v; - } - - void LinearScanRegAllocPass::webUnion(VReg x, VReg y) { - VReg rx = webFind(x), ry = webFind(y); - if(rx == ry) - return; - // safe only when no two members are live at once - List connecting = copyPointsBetween(x, y); - List none; - for(VReg a : webMembers[rx]) - for(VReg b : webMembers[ry]) { - B32 isEdge = (a == x && b == y) || (a == y && b == x); - if(!overlapOnlyAt(ivAt(a), ivAt(b), isEdge ? connecting : none)) - return; - } - webParent[ry] = rx; - List& into = webMembers[rx]; - for(VReg m : webMembers[ry]) - into.push_back(m); - webMembers.erase(ry); - } - - void LinearScanRegAllocPass::buildSpillWebs(const List& spilled) { - webParent.assign(fn->nextVReg, kNoVReg); - webMembers.clear(); - for(const Interval* iv : spilled) { - webParent[iv->vreg] = iv->vreg; - webMembers[iv->vreg].push_back(iv->vreg); - } - for(const Interval* iv : spilled) { - // a spilled remat def never stores its slot, so - // sharing would drop its copies as self-moves and read a stale slot - if(rematDef.find(iv->vreg) != rematDef.end()) - continue; - auto it = copyHints.find(iv->vreg); - if(it == copyHints.end()) - continue; - for(const CopyHint& h : it->second) { - const Interval* p = ivFind(h.partner); - if(p && p->spilled && rematDef.find(h.partner) == rematDef.end()) - webUnion(iv->vreg, h.partner); - } - } - } - - void LinearScanRegAllocPass::assignSpillSlots() { - // pack - List spilled; - for(U32 v = 1; v < fn->nextVReg; ++v) - if(intervals[v].live() && intervals[v].spilled) - spilled.push_back(&intervals[v]); - std::sort(spilled.begin(), spilled.end(), [](const Interval* a, const Interval* b) { - return a->start != b->start ? a->start < b->start : a->vreg < b->vreg; - }); - buildSpillWebs(spilled); - Map webSlot; - for(Interval* iv : spilled) { - VReg root = webFind(iv->vreg); - auto it = webSlot.find(root); - if(it != webSlot.end()) { - iv->spillSlot = it->second; - continue; - } - I32 start = iv->start, end = iv->end; - for(VReg m : webMembers[root]) { - start = std::min(start, ivAt(m).start); - end = std::max(end, ivAt(m).end); - } - iv->spillSlot = takeSpillSlot(iv->cls, start, end); - webSlot.emplace(root, iv->spillSlot); - } - } - -} // namespace rat diff --git a/src/backend/codegen/linear_scan_reg_alloc.h b/src/backend/codegen/linear_scan_reg_alloc.h deleted file mode 100644 index cdd3b0c..0000000 --- a/src/backend/codegen/linear_scan_reg_alloc.h +++ /dev/null @@ -1,116 +0,0 @@ -// linear-scan register allocation over the target-independent machine IR. virtual registers -// are numbered along a linearized instruction order and live ranges are computed as segment -// lists (hole-aware: the gaps between segments are provably off every def-use path, so -// fixed-register pins inside a hole don't constrain the value and call clobbers inside a -// hole don't force a callee-saved register). assignment scans ranges in start order per -// register class; copy hints bias the choice so coalescable moves become elided self-moves, -// and move partners whose ranges meet only at their connecting copies may share a register. -// under pressure a value is spilled to a frame slot. the allocator is fully -// backend-agnostic: it reads register classes from a RegisterInfo and builds -// spill/reload/slot constructs through RegAllocHooks callbacks, so it never names a single -// target opcode -// -// references: -// - M. Poletto and V. Sarkar, "Linear Scan Register Allocation", ACM TOPLAS, 1999 -// - O. Traub, G. Holloway, M. D. Smith, "Quality and Speed in Linear-scan Register -// Allocation", PLDI, 1998 - -#ifndef RAT_CODEGEN_LINEARSCANREGALLOC_H -#define RAT_CODEGEN_LINEARSCANREGALLOC_H - -#include "core.h" - -#include "codegen/reg_alloc_base.h" - -namespace rat { - struct LinearScanRegAllocPass : RegAllocBase { - const C8* name() const override { return "linear-scan-regalloc"; } - private: - // per-loop-level use weight - static constexpr U32 kLoopUseWeight = 3; - static constexpr U32 kMaxUseWeight = 100000; - - using Touch = Pair; // a spilled vreg and a point reading or writing it - - // closed [start, end] - struct Seg { - I32 start; - I32 end; - }; - - struct Interval { - VReg vreg = kNoVReg; - U32 cls = 0; - I32 start = -1; // hull: segs.front().start - I32 end = -1; // hull: segs.back().end - List segs; - PhysReg assigned = kNoReg; - I32 spillSlot = 0; - B32 spilled = false; - B32 crossesCall = false; - U32 uses = 0; // operand occurrences - - void reset() { - vreg = kNoVReg; - cls = 0; - start = -1; - end = -1; - segs.clear(); - assigned = kNoReg; - spillSlot = 0; - spilled = false; - crossesCall = false; - uses = 0; - } - B32 live() const { return vreg != kNoVReg; } - }; - - B32 anySpilled() const override { - for(U32 v = 1; v < fn->nextVReg; ++v) - if(intervals[v].live() && intervals[v].spilled) - return true; - return false; - } - - void resetState() override { - pieces.clear(); - U32 nv = std::min((U32)intervals.size(), fn->nextVReg); - for(U32 v = 0; v < nv; ++v) - intervals[v].reset(); - } - void solve() override; - Assignment assignmentOf(VReg v) override; - void buildIntervals(); - static void coalesceSegs(Interval& iv); - static B32 overlapOnlyAt(const Interval& a, const Interval& b, const List& pts); - U64 forbidden(const Interval& iv) const; - void assignRegs(); - void assignSpillSlots(); - void assignPieces() override; - void cutPieces(const Touch* pts, U32 count, List& busy); - static B32 pieceAfter(const Touch& key, const Piece& pc); - PhysReg pickPieceReg(U32 cls, I32 start, I32 end, List& busy); - Piece* pieceAt(VReg v, I32 pt) override; - void spillAt(Interval* cur, List& active); - void buildSpillWebs(const List& spilled); - VReg webFind(VReg v); - void webUnion(VReg x, VReg y); - private: - Interval& ivAt(VReg v) { return intervals[v]; } - const Interval* ivFind(VReg v) const { - return v < intervals.size() && intervals[v].live() ? &intervals[v] : nullptr; - } - - List intervals; - List pieces; // sorted by (vreg, start) - List webParent; - Map> webMembers; // union-find root -> members - List> pinsByPoint; - List expiredBuf; - List segEnd; - VRegSet live; - List freeRegs; - }; -} // namespace rat - -#endif diff --git a/src/backend/codegen/reg_alloc_base.cpp b/src/backend/codegen/reg_alloc_base.cpp deleted file mode 100644 index e86246a..0000000 --- a/src/backend/codegen/reg_alloc_base.cpp +++ /dev/null @@ -1,552 +0,0 @@ -#include "codegen/reg_alloc_base.h" - -#include "codegen/machine_module.h" -#include "ir/module.h" -#include "target/target.h" - -namespace rat { - void RegAllocBase::number() { - blkPts.assign(fn->blocks.size(), {}); - for(U32 b = 0; b < fn->blocks.size(); ++b) - for(U32 i = 0; i < fn->blocks[b].insts.size(); ++i) { - I32 pt = (I32)order.size(); - blkPts[b].push_back((U32)pt); - order.push_back({b, i}); - fixedAt.push_back(0); - const MachineInstr& in = fn->blocks[b].insts[i]; - if(in.isCall) - callPts.push_back(pt); - for(const MachineOperand& o : in.uses) - if(o.isPhys()) - fixedAt[pt] |= (U64)1 << o.phys; - for(const MachineOperand& o : in.defs) - if(o.isPhys()) - fixedAt[pt] |= (U64)1 << o.phys; - for(PhysReg p : in.clobbers) - fixedAt[pt] |= (U64)1 << p; - } - pinFixedArgWindows(); - } - - void RegAllocBase::pinFixedArgWindows() { - for(I32 c : callPts) { - U32 b = order[(U32)c].block; - U32 callIdx = order[(U32)c].inst; - const List& pts = blkPts[b]; - const MachineInstr& call = fn->blocks[b].insts[callIdx]; - for(const MachineOperand& u : call.uses) { - if(!u.isPhys()) - continue; - PhysReg p = u.phys; - for(I32 i = (I32)callIdx - 1; i >= 0; --i) { - I32 pt = (I32)pts[(U32)i]; - const MachineInstr& in = fn->blocks[b].insts[(U32)i]; - B32 defsP = false; - for(const MachineOperand& d : in.defs) - if(d.isPhys() && d.phys == p) { - defsP = true; - break; - } - fixedAt[pt] |= (U64)1 << p; - if(defsP) - break; - } - } - } - } - - void RegAllocBase::collectCopyHints() { - if(!hooks->isCopy) - return; - for(U32 b = 0; b < (U32)fn->blocks.size(); ++b) - for(U32 i = 0; i < (U32)fn->blocks[b].insts.size(); ++i) { - const MachineInstr& in = fn->blocks[b].insts[i]; - if(!hooks->isCopy(in) || in.defs.size() != 1 || in.uses.size() != 1) - continue; - const MachineOperand& d = in.defs[0]; - const MachineOperand& u = in.uses[0]; - I32 pt = (I32)blkPts[b][i]; - if(d.isVReg() && u.isVReg()) { - if(classOf(d.vreg) != classOf(u.vreg)) - continue; - copyHints[d.vreg].push_back({u.vreg, pt}); - copyHints[u.vreg].push_back({d.vreg, pt}); - } else if(d.isVReg() && u.isPhys()) { - physHints[d.vreg].push_back(u.phys); - copyPinAt[d.vreg].emplace(pt, u.phys); - } else if(d.isPhys() && u.isVReg()) { - physHints[u.vreg].push_back(d.phys); - copyPinAt[u.vreg].emplace(pt, d.phys); - } - } - } - - List RegAllocBase::hintedRegs(VReg v, const Delegate& colorOf) const { - List hints; - if(auto it = physHints.find(v); it != physHints.end()) - for(PhysReg p : it->second) - hints.push_back(p); - if(auto it = copyHints.find(v); it != copyHints.end()) - for(const CopyHint& h : it->second) - if(PhysReg p = colorOf(h.partner); p != kNoReg) - hints.push_back(p); - return hints; - } - - List RegAllocBase::copyPointsBetween(VReg a, VReg b) const { - List pts; - if(auto it = copyHints.find(a); it != copyHints.end()) - for(const CopyHint& h : it->second) - if(h.partner == b) - pts.push_back(h.pt); - return pts; - } - - I32 RegAllocBase::takeSpillSlot(U32 cls, I32 start, I32 end) { - for(PooledSlot& ps : slotPool[cls]) - if(ps.freeEnd < start) { - ps.freeEnd = end; - return ps.slot; - } - U32 bytes = ri->spillSlotBytes; - for(const RegClass& rc : ri->classes) - if(rc.id == cls && rc.spillBytes) - bytes = rc.spillBytes; - I32 slot = hooks->allocSlot(*fn, cls, bytes); - slotPool[cls].push_back({slot, end}); - return slot; - } - - void DenseLive::prep(U32 nb, U32 nv) { - for(List* v : {&in, &out, &use, &def}) { - if(v->size() < nb) - v->resize(nb); - for(U32 i = 0; i < nb; ++i) - (*v)[i].resetAll(nv); - } - } - - void SparseLive::prep(U32 nb) { - for(List* v : {&in, &out, &use, &def}) { - if(v->size() < nb) - v->resize(nb); - for(U32 i = 0; i < nb; ++i) - (*v)[i].clear(); - } - } - - void detail::vregUnion(const VRegList& a, const VRegList& b, VRegList& dst) { - dst.clear(); - U32 i = 0, j = 0; - while(i < a.size() && j < b.size()) { - if(a[i] < b[j]) - dst.push_back(a[i++]); - else if(b[j] < a[i]) - dst.push_back(b[j++]); - else { - dst.push_back(a[i++]); - ++j; - } - } - while(i < a.size()) - dst.push_back(a[i++]); - while(j < b.size()) - dst.push_back(b[j++]); - } - - void detail::vregUnionMasked(const VRegList& a, - const VRegList& b, - const VRegList& mask, - VRegList& dst) { - dst.clear(); - U32 i = 0, m = 0; - for(U32 j = 0; j < b.size(); ++j) { - while(m < mask.size() && mask[m] < b[j]) - ++m; - if(m < mask.size() && mask[m] == b[j]) // killed by a def in this block - continue; - while(i < a.size() && a[i] < b[j]) - dst.push_back(a[i++]); - if(i < a.size() && a[i] == b[j]) - ++i; - dst.push_back(b[j]); - } - while(i < a.size()) - dst.push_back(a[i++]); - } - - void RegAllocBase::blockUseDefsSparse() { - VRegSet used(fn->nextVReg), defd(fn->nextVReg); - for(U32 b = 0; b < (U32)fn->blocks.size(); ++b) { - VRegList& use = sparseLive.use[b]; - VRegList& def = sparseLive.def[b]; - for(const MachineInstr& in : fn->blocks[b].insts) { - for(const MachineOperand& u : in.uses) - if(u.isVReg() && !defd.test(u.vreg) && !used.test(u.vreg)) { - used.set(u.vreg); - use.push_back(u.vreg); - } - for(const MachineOperand& d : in.defs) - if(d.isVReg() && !defd.test(d.vreg)) { - defd.set(d.vreg); - def.push_back(d.vreg); - } - } - for(VReg v : use) - used.reset(v); - for(VReg v : def) - defd.reset(v); - std::sort(use.begin(), use.end()); - std::sort(def.begin(), def.end()); - } - } - - void RegAllocBase::liveOutOf(U32 b, VRegList& out, VRegList& tmp) { - const List& succs = fn->blocks[b].succs; - if(succs.empty()) { - out.clear(); - return; - } - out = sparseLive.in[(U32)succs[0]]; - for(U32 i = 1; i < (U32)succs.size(); ++i) { - detail::vregUnion(out, sparseLive.in[(U32)succs[i]], tmp); - out.swap(tmp); - } - } - - constexpr U64 kDenseLivenessBytes = 64ull << 20; - - B32 RegAllocBase::denseLivenessFits() const { - U64 words = (fn->nextVReg + 63) / 64; - U64 perBlock = words * 8 * 4; // in, out, use and def, eight bytes to the word - return (U64)fn->blocks.size() * perBlock <= kDenseLivenessBytes; - } - - void RegAllocBase::liveness() { - liveIsDense = denseLivenessFits(); - if(liveIsDense) - livenessDense(); - else - livenessSparse(); - } - - void RegAllocBase::blockUseDefsDense() { - for(U32 b = 0; b < (U32)fn->blocks.size(); ++b) - for(const MachineInstr& in : fn->blocks[b].insts) { - for(const MachineOperand& u : in.uses) - if(u.isVReg() && !denseLive.def[b].test(u.vreg)) - denseLive.use[b].set(u.vreg); - for(const MachineOperand& d : in.defs) - if(d.isVReg()) - denseLive.def[b].set(d.vreg); - } - } - - void RegAllocBase::livenessDense() { - U32 nb = (U32)fn->blocks.size(); - U32 nv = fn->nextVReg; - denseLive.prep(nb, nv); - blockUseDefsDense(); - B32 changed = true; - VRegSet out, in; - out.resetAll(nv); - in.resetAll(nv); - while(changed) { - changed = false; - for(I32 b = (I32)nb - 1; b >= 0; --b) { - out.resetAll(nv); - for(I32 s : fn->blocks[b].succs) - out.orWith(denseLive.in[s]); - in.assignUnionMasked(denseLive.use[b], out, denseLive.def[b]); // use | (out & ~def) - if(!(in == denseLive.in[b]) || !(out == denseLive.out[b])) { - changed = true; - denseLive.in[b].copyFrom(in); - denseLive.out[b].copyFrom(out); - } - } - } - } - - void RegAllocBase::livenessSparse() { - U32 nb = (U32)fn->blocks.size(); - sparseLive.prep(nb); - blockUseDefsSparse(); - B32 changed = true; - VRegList out, in, tmp; - while(changed) { - changed = false; - for(I32 b = (I32)nb - 1; b >= 0; --b) { - liveOutOf((U32)b, out, tmp); - // use | (out & ~def) - detail::vregUnionMasked(sparseLive.use[b], out, sparseLive.def[b], in); - if(in != sparseLive.in[b] || out != sparseLive.out[b]) { - changed = true; - sparseLive.in[b] = in; - sparseLive.out[b] = out; - } - } - } - } - - U32 RegAllocBase::classOf(VReg v) const { - return v < fn->vregClass.size() ? fn->vregClass[v] : 0; - } - - const RegClass& RegAllocBase::regClass(U32 cls) const { return ri->classes[cls]; } - - B32 RegAllocBase::isCalleeSaved(const RegClass& rc, PhysReg p) { - for(PhysReg c : rc.calleeSaved) - if(c == p) - return true; - return false; - } - - void RegAllocBase::collectRematDefs() { - if(!hooks->isRemat) - return; - Map defCount; - for(U32 b = 0; b < fn->blocks.size(); ++b) - for(const MachineInstr& in : fn->blocks[b].insts) - for(const MachineOperand& d : in.defs) - if(d.isVReg()) - ++defCount[d.vreg]; - for(U32 b = 0; b < fn->blocks.size(); ++b) - for(const MachineInstr& in : fn->blocks[b].insts) { - if(in.defs.size() != 1 || !in.defs[0].isVReg() || !hooks->isRemat(in)) - continue; - VReg v = in.defs[0].vreg; - if(defCount[v] == 1) - rematDef[v] = in; - } - } - - B32 RegAllocBase::dropsRematDef(const MachineInstr& in) { - if(in.defs.size() != 1 || !in.defs[0].isVReg()) - return false; - VReg v = in.defs[0].vreg; - return rematDef.count(v) && assignmentOf(v).spilled && !slotReadByCall.count(v); - } - - void RegAllocBase::emitReload(List& out, - PhysReg dst, - const MachineOperand& u, - const Assignment& a) { - auto rt = rematDef.find(u.vreg); - if(rt == rematDef.end()) { - out.push_back(hooks->makeReload(dst, a.spillSlot, a.cls, u.width)); - return; - } - MachineInstr m = rt->second; - m.defs[0] = MachineOperand::fixed(dst, u.width); - out.push_back(std::move(m)); - } - - void RegAllocBase::emitStore(List& out, I32 slot, PhysReg src, U32 cls, U32 width) { - out.push_back(hooks->makeSpill(slot, src, cls, width)); - memo = {true, slot, src, cls, width}; - } - - // the register that just stored its slot or target after a reload into it - PhysReg RegAllocBase::sourceOf(List& out, - const MachineOperand& u, - const Assignment& a, - PhysReg target) { - if(memo.on && memo.cls == a.cls && memo.slot == a.spillSlot && memo.width == u.width) - return memo.reg; - emitReload(out, target, u, a); - return target; - } - - B32 RegAllocBase::rewriteCopy(List& out, MachineInstr& in) { - MachineOperand& d = in.defs[0]; - MachineOperand& u = in.uses[0]; - Assignment da = d.isVReg() ? assignmentOf(d.vreg) : Assignment{}; - Assignment ua = u.isVReg() ? assignmentOf(u.vreg) : Assignment{}; - if(!da.spilled && !ua.spilled) - return false; - if(da.spilled && ua.spilled && da.spillSlot == ua.spillSlot) - return true; // slot self-copy - PhysReg dst = kNoReg; // where the value must land - if(d.isPhys()) - dst = d.phys; - else if(!da.spilled) - dst = da.reg; - PhysReg src; - if(ua.spilled) - src = sourceOf(out, u, ua, dst != kNoReg ? dst : scratchAt(ua.cls, 0)); - else - src = u.isPhys() ? u.phys : ua.reg; - if(da.spilled) { - emitStore(out, da.spillSlot, src, da.cls, d.width); - return true; - } - if(src != dst) { - d = MachineOperand::fixed(dst, d.width); - u = MachineOperand::fixed(src, u.width); - out.push_back(in); - } - memo.on = false; - return true; - } - - void RegAllocBase::rewriteInstr(List& out, MachineInstr& in, I32 pt) { - U32 useScratch[kMaxRegClasses] = {0}; - for(MachineOperand& u : in.uses) { - if(!u.isVReg()) - continue; - Assignment a = assignmentOf(u.vreg); - if(!a.spilled) { - u = MachineOperand::fixed(a.reg, u.width); - continue; - } - if(in.isCall) { - u = MachineOperand::frameSlot(a.spillSlot, u.width); - continue; - } - if(Piece* pc = pieceAt(u.vreg, pt)) { - if(!pc->loaded) - emitReload(out, pc->reg, u, a); - pc->loaded = true; - u = MachineOperand::fixed(pc->reg, u.width); - continue; - } - // the previous instruction just stored this slot from a register nothing has - // touched since, reuse - B32 tied = false; - for(const MachineOperand& d : in.defs) - if(d.isVReg() && d.vreg == u.vreg) - tied = true; - if(memo.on && !tied && useScratch[a.cls] == 0 && memo.cls == a.cls && - memo.slot == a.spillSlot && memo.width == u.width) { - u = MachineOperand::fixed(memo.reg, u.width); - ++useScratch[a.cls]; // reserve index 0 in case memo.reg is scratch 0 - continue; - } - PhysReg sc = scratchAt(a.cls, useScratch[a.cls]++); - emitReload(out, sc, u, a); - u = MachineOperand::fixed(sc, u.width); - } - - U32 defScratch[kMaxRegClasses] = {0}; - List stores; - for(MachineOperand& d : in.defs) { - if(!d.isVReg()) - continue; - Assignment a = assignmentOf(d.vreg); - if(!a.spilled) { - d = MachineOperand::fixed(a.reg, d.width); - continue; - } - PhysReg sc; - if(Piece* pc = pieceAt(d.vreg, pt)) { - sc = pc->reg; - pc->loaded = true; - } else { - sc = scratchAt(a.cls, defScratch[a.cls]++); - } - stores.push_back(hooks->makeSpill(a.spillSlot, sc, a.cls, d.width)); - d = MachineOperand::fixed(sc, d.width); - } - - memo.on = false; - out.push_back(in); - for(MachineInstr& s : stores) - out.push_back(std::move(s)); - if(stores.empty() || in.isCall) - return; - // uses[0] = frame slot, uses[1] = source register - const MachineInstr& last = out.back(); - if(last.uses.size() == 2 && last.uses[0].kind == MachineOperand::Kind::FrameSlot && - last.uses[1].kind == MachineOperand::Kind::Phys) - memo = {true, last.uses[0].slot, last.uses[1].phys, last.regClass, last.uses[1].width}; - } - - void RegAllocBase::rewrite() { - slotReadByCall.clear(); - for(U32 b = 0; b < fn->blocks.size(); ++b) - for(const MachineInstr& in : fn->blocks[b].insts) - if(in.isCall) - for(const MachineOperand& u : in.uses) - if(u.isVReg()) - slotReadByCall.insert(u.vreg); - - for(U32 b = 0; b < fn->blocks.size(); ++b) { - List out; - memo.on = false; - for(U32 i = 0; i < fn->blocks[b].insts.size(); ++i) { - MachineInstr& in = fn->blocks[b].insts[i]; - I32 pt = (I32)blkPts[b][i]; - if(dropsRematDef(in)) { // every use remats instead - memo.on = false; - continue; - } - B32 copy = hooks->isCopy && hooks->isCopy(in) && in.defs.size() == 1 && in.uses.size() == 1; - if(copy && rewriteCopy(out, in)) - continue; - rewriteInstr(out, in, pt); - } - fn->blocks[b].insts = std::move(out); - } - } - - PhysReg RegAllocBase::scratchAt(U32 cls, U32 idx) { - const RegClass& rc = ri->classes[cls]; - if(rc.scratch.empty()) { - ok = false; - return kNoReg; - } - if(idx >= rc.scratch.size()) { - ok = false; - idx = (U32)rc.scratch.size() - 1; - } - return rc.scratch[idx]; - } - - B32 RegAllocBase::allocate(MachineFunc& f, - const RegisterInfo& r, - const RegAllocHooks& h, - List* usedCalleeSaved) { - fn = &f; - ri = &r; - hooks = &h; - order.clear(); - blkPts.clear(); - callPts.clear(); - fixedAt.clear(); - usedCallee.clear(); - copyHints.clear(); - physHints.clear(); - copyPinAt.clear(); - slotPool.clear(); - rematDef.clear(); - ok = true; - resetState(); - - number(); - collectCopyHints(); - collectRematDefs(); - - solve(); - assignPieces(); - rewrite(); - - if(usedCalleeSaved) { - usedCalleeSaved->assign(usedCallee.begin(), usedCallee.end()); - std::sort(usedCalleeSaved->begin(), usedCalleeSaved->end()); - } - return ok; - } - - B32 RegAllocBase::run(Module& module, MachineModule& mm, const TargetInfo& target) { - U32 changed = 0; - for(const Function* fn : module) { - MachineFunc& mf = mm.get(fn); - B32 allocated = - allocate(mf, *target.registers(), target.regAllocHooks(), &mf.usedCalleeSaved); - assert(allocated && "register allocation ran out of scratch registers"); - (void)allocated; - ++changed; - } - return changed != 0; - } -} // namespace rat diff --git a/src/backend/codegen/reg_alloc_base.h b/src/backend/codegen/reg_alloc_base.h deleted file mode 100644 index 05832fb..0000000 --- a/src/backend/codegen/reg_alloc_base.h +++ /dev/null @@ -1,176 +0,0 @@ -#ifndef RAT_CODEGEN_REGALLOCBASE_H -#define RAT_CODEGEN_REGALLOCBASE_H - -#include "core.h" - -#include "codegen/machine_function.h" -#include "pass/pass.h" - -namespace rat { - // dense bitset over vreg numbers - struct VRegSet { - explicit VRegSet(U32 bits = 0) - : words((bits + 63) / 64, 0) {} - - B32 test(VReg v) const { return (U32)((words[v >> 6] >> (v & 63)) & 1); } - void set(VReg v) { words[v >> 6] |= (U64)1 << (v & 63); } - void reset(VReg v) { words[v >> 6] &= ~((U64)1 << (v & 63)); } - - B32 operator==(const VRegSet& o) const { return words == o.words; } - - void resetAll(U32 bits) { words.assign((bits + 63) / 64, 0); } - void copyFrom(const VRegSet& o) { words = o.words; } - - void orWith(const VRegSet& o) { - for(U32 i = 0; i < (U32)words.size(); ++i) - words[i] |= o.words[i]; - } - - // this = a | (b & ~mask) - void assignUnionMasked(const VRegSet& a, const VRegSet& b, const VRegSet& mask) { - for(U32 i = 0; i < (U32)words.size(); ++i) - words[i] = a.words[i] | (b.words[i] & ~mask.words[i]); - } - - template void forEach(F f) const { // ascending vreg order - for(U32 wi = 0; wi < (U32)words.size(); ++wi) - for(U64 w = words[wi]; w; w &= w - 1) - f((VReg)(wi * 64 + (U32)countTrailingZeros64(w))); - } - private: - List words; - }; - - using VRegList = List; - - namespace detail { - void vregUnion(const VRegList& a, const VRegList& b, VRegList& dst); // dst = a | b - void vregUnionMasked(const VRegList& a, - const VRegList& b, - const VRegList& mask, - VRegList& dst); // dst = a | (b & ~mask) - } // namespace detail - - struct DenseLive { - List in, out, use, def; - void prep(U32 nb, U32 nv); - }; - - struct SparseLive { - List in, out, use, def; - void prep(U32 nb); - }; - - struct RegAllocBase : MachinePass { - B32 run(Module& module, MachineModule& mm, const TargetInfo& target) override; - protected: - struct Loc { - U32 block; - U32 inst; - }; - - // where a vreg ended up: a physical register, or a frame slot when spilled - struct Assignment { - PhysReg reg = kNoReg; - I32 spillSlot = 0; - U32 cls = 0; - B32 spilled = false; - }; - - // one side of a coalescable copy and the point of the copy itself - struct CopyHint { - VReg partner; - I32 pt; - }; - - virtual void resetState() = 0; // clear solver state between functions - virtual void solve() = 0; // compute assignments (number() already ran) - virtual Assignment assignmentOf(VReg v) = 0; // result lookup used by rewrite() - virtual B32 anySpilled() const = 0; // did solve() spill anything? - virtual void assignPieces() = 0; - struct Piece { - VReg vreg; - I32 start; - I32 end; - PhysReg reg; - B32 loaded = false; // reloaded once at its first use - }; - virtual Piece* pieceAt(VReg v, I32 pt) = 0; // the piece holding v at pt, if any - - B32 allocate(MachineFunc& fn, - const RegisterInfo& ri, - const RegAllocHooks& hooks, - List* usedCalleeSaved); - - void number(); - void pinFixedArgWindows(); - void collectCopyHints(); - void collectRematDefs(); - void liveness(); - - // allocation preferences derived from copies - List hintedRegs(VReg v, const Delegate& colorOf) const; - - // program points of copies connecting two move partners - List copyPointsBetween(VReg a, VReg b) const; - - // spill-slot pool - I32 takeSpillSlot(U32 cls, I32 start, I32 end); - U32 classOf(VReg v) const; - const RegClass& regClass(U32 cls) const; - static B32 isCalleeSaved(const RegClass& rc, PhysReg p); - // a slot just written from a register nothing has touched since - struct Memo { - B32 on = false; - I32 slot = 0; - PhysReg reg = kNoReg; - U32 cls = 0; - U32 width = 0; - }; - - void rewrite(); - B32 dropsRematDef(const MachineInstr& in); - void - emitReload(List& out, PhysReg dst, const MachineOperand& u, const Assignment& a); - void emitStore(List& out, I32 slot, PhysReg src, U32 cls, U32 width); - PhysReg - sourceOf(List& out, const MachineOperand& u, const Assignment& a, PhysReg target); - B32 rewriteCopy(List& out, MachineInstr& in); - void rewriteInstr(List& out, MachineInstr& in, I32 pt); - PhysReg scratchAt(U32 cls, U32 idx); - protected: - MachineFunc* fn = nullptr; - const RegisterInfo* ri = nullptr; - const RegAllocHooks* hooks = nullptr; - List order; // linear point -> (block, instruction) - List> blkPts; // block -> its linear points - List callPts; // points that are calls - List fixedAt; // physical registers pinned at each point - Set usedCallee; - Map> copyHints; // vreg <-> vreg move affinities - Map> physHints; // vreg <-> fixed-register move affinities - Map> copyPinAt; // pin exemptions - Map rematDef; // single pure def per remat vreg - B32 ok = true; - Memo memo; - Set slotReadByCall; - B32 liveIsDense = false; - DenseLive denseLive; - SparseLive sparseLive; - private: - B32 denseLivenessFits() const; - void livenessDense(); - void livenessSparse(); - void blockUseDefsDense(); - void blockUseDefsSparse(); - void liveOutOf(U32 b, VRegList& out, VRegList& tmp); - - struct PooledSlot { - I32 slot; - I32 freeEnd; // last point of the current occupant's live range - }; - Map> slotPool; // per register class - }; -} // namespace rat - -#endif diff --git a/src/backend/pass/pass_registry.cpp b/src/backend/pass/pass_registry.cpp index 5093a84..c3f2611 100644 --- a/src/backend/pass/pass_registry.cpp +++ b/src/backend/pass/pass_registry.cpp @@ -7,7 +7,7 @@ #include "pass/verify.h" -#include "codegen/linear_scan_reg_alloc.h" +#include "codegen/reg_alloc.h" #include "pass/emit/graph_emitter.h" #include "pass/emit/text_emitter.h" #include "pass/emit/x86/x86_encode.h" @@ -62,7 +62,7 @@ namespace rat { // the default x86 machine pipeline, in order constexpr MachineEntry kMachinePasses[] = { {"x86-lower", "lower IR to x86 machine instructions", &mk}, - {"regalloc", "linear-scan register allocation", &mk}, + {"regalloc", "priority bin-packing register allocation", &mk}, {"x86-peephole", "post-RA copy and spill-slot cleanup", &mk}, {"x86-layout", "block ordering and frame layout", &mk}, {"x86-encode", "instruction encoding and object emission", &mk}, diff --git a/src/backend/rat.h b/src/backend/rat.h index fd1154b..c830100 100644 --- a/src/backend/rat.h +++ b/src/backend/rat.h @@ -3,7 +3,7 @@ #include "ir/module.h" -#include "codegen/linear_scan_reg_alloc.h" +#include "codegen/reg_alloc.h" #include "target/target.h" #include "pass/pass_manager.h" diff --git a/src/compiler/compile.cpp b/src/compiler/compile.cpp index 1c891d6..10c314d 100644 --- a/src/compiler/compile.cpp +++ b/src/compiler/compile.cpp @@ -22,7 +22,7 @@ namespace rat::cc { pm.add(std::move(p)); } else { pm.add(); - pm.add(); + pm.add(); pm.add(); pm.add(); pm.add(out); From b02bbe1b1637c9fd5cd5a79fec85df1d0549cf1f Mon Sep 17 00:00:00 2001 From: nella Date: Sat, 26 Sep 2026 00:16:40 +0200 Subject: [PATCH 05/12] Coalesce copies into bundles. --- src/backend/codegen/reg_alloc.cpp | 98 +++++++++++++++++++++++++++---- src/backend/codegen/reg_alloc.h | 13 +++- 2 files changed, 98 insertions(+), 13 deletions(-) diff --git a/src/backend/codegen/reg_alloc.cpp b/src/backend/codegen/reg_alloc.cpp index 43d116f..9c7c947 100644 --- a/src/backend/codegen/reg_alloc.cpp +++ b/src/backend/codegen/reg_alloc.cpp @@ -36,6 +36,7 @@ namespace rat { number(); liveness(); buildIntervals(); + coalesce(); assignRegs(); rewrite(); fn.usedCalleeSaved.clear(); @@ -47,6 +48,12 @@ namespace rat { return hooks.isCopy && in.defs.size() == 1 && in.uses.size() == 1 && hooks.isCopy(in); } + const detail::RaInterval& detail::RegAllocFunc::bundle(VReg v) const { return iv[iv[v].root]; } + + B32 detail::RegAllocFunc::sameBundle(const MachineOperand& a, const MachineOperand& b) const { + return a.isVReg() && b.isVReg() && iv[a.vreg].root == iv[b.vreg].root; + } + void detail::RegAllocFunc::number() { blockFirst.assign(1, 0); for(const MachineBlock& blk : fn.blocks) @@ -142,9 +149,18 @@ namespace rat { segs.emplace_back(start, end); } + void detail::RegAllocFunc::noteCopy(const MachineInstr& in, U32 weight) { + const MachineOperand& d = in.defs[0]; + const MachineOperand& s = in.uses[0]; + if(d.isVReg() && s.isVReg() && fn.vregClass[d.vreg] == fn.vregClass[s.vreg]) + copies.push_back({~0u - weight, {d.vreg, s.vreg}}); + } + // backward walk per block from its live-out set, blocks in reverse void detail::RegAllocFunc::buildIntervals() { iv.resize(nv); + for(VReg v = 0; v < nv; ++v) + iv[v].root = v; List live(nv, 0); List segEnd(nv, 0); List liveList; @@ -174,8 +190,10 @@ namespace rat { live[o.vreg] = 0; } I32 useEnd = u + 1; - if(isCopy(in)) + if(isCopy(in)) { useEnd = u; + noteCopy(in, weight); + } for(const MachineOperand& o : in.uses) { if(!o.isVReg()) continue; @@ -194,13 +212,67 @@ namespace rat { } liveList.clear(); } - for(RaInterval& t : iv) { + for(RaInterval& t : iv) std::reverse(t.segs.begin(), t.segs.end()); + } + + // path halving + VReg detail::RegAllocFunc::find(VReg v) { + while(iv[v].root != v) + v = iv[v].root = iv[iv[v].root].root; + return v; + } + + B32 detail::RegAllocFunc::overlaps(VReg a, VReg b) const { + const List& x = iv[a].segs; + const List& y = iv[b].segs; + U32 i = 0; + U32 j = 0; + while(i < x.size() && j < y.size()) { + if(x[i].second < y[j].first) + ++i; + else if(y[j].second < x[i].first) + ++j; + else + return true; + } + return false; + } + + // b joins a, their segments do not overlap + void detail::RegAllocFunc::merge(VReg a, VReg b) { + RaInterval& t = iv[a]; + RaInterval& o = iv[b]; + List segs(t.segs.size() + o.segs.size()); + std::merge(t.segs.begin(), t.segs.end(), o.segs.begin(), o.segs.end(), segs.begin()); + t.segs.clear(); + for(const auto& [start, end] : segs) // join touching segments + if(!t.segs.empty() && t.segs.back().second + 1 == start) + t.segs.back().second = end; + else + t.segs.emplace_back(start, end); + o.segs = {}; + t.weight += o.weight; + o.root = a; + } + + // copy-related vregs whose ranges do not overlap share one register or slot, hottest + // copies first + void detail::RegAllocFunc::coalesce() { + std::sort(copies.begin(), copies.end()); + for(const auto& [cold, pair] : copies) { + VReg a = find(pair.first); + VReg b = find(pair.second); + if(a != b && !overlaps(a, b)) + merge(a, b); + } + for(VReg v = 1; v < nv; ++v) { + iv[v].root = find(v); I32 len = 0; - for(const auto& [start, end] : t.segs) + for(const auto& [start, end] : iv[v].segs) len += end - start + 1; if(len) - t.weight /= std::sqrt((F32)len); + iv[v].weight /= std::sqrt((F32)len); } } @@ -214,7 +286,7 @@ namespace rat { } void detail::RegAllocFunc::assignRegs() { - List> order; // (-weight, vreg) + List> order; // (-weight, bundle) for(VReg v = 1; v < nv; ++v) if(!iv[v].segs.empty()) order.emplace_back(-iv[v].weight, v); @@ -252,8 +324,9 @@ namespace rat { PhysReg detail::RegAllocFunc::spillReg(List& out, const MachineOperand& o, U32 i, B32 use) { + VReg root = iv[o.vreg].root; for(const auto& [v, r] : temps) - if(v == o.vreg) + if(v == root) return r; U32 cls = fn.vregClass[o.vreg]; U64 hard = (busy[2 * (U64)i] | busy[2 * (U64)i + 1]) & ~own; @@ -261,9 +334,9 @@ namespace rat { hard |= taken; PhysReg r = pickTemp(cls, hard, own | taken); if(use) - out.push_back(hooks.makeReload(r, iv[o.vreg].slot, cls, o.width)); + out.push_back(hooks.makeReload(r, iv[root].slot, cls, o.width)); taken |= (U64)1 << r; - temps.emplace_back(o.vreg, r); + temps.emplace_back(root, r); return r; } @@ -277,9 +350,9 @@ namespace rat { for(MachineOperand& o : in.uses) { if(!o.isVReg()) continue; - PhysReg r = iv[o.vreg].reg; + PhysReg r = bundle(o.vreg).reg; if(r == kNoReg && in.isCall) { // the call reads the slot itself - o = MachineOperand::frameSlot(iv[o.vreg].slot, o.width); + o = MachineOperand::frameSlot(bundle(o.vreg).slot, o.width); continue; } if(r == kNoReg) @@ -289,7 +362,7 @@ namespace rat { for(MachineOperand& o : in.defs) { if(!o.isVReg()) continue; - const RaInterval& t = iv[o.vreg]; + const RaInterval& t = bundle(o.vreg); PhysReg r = t.reg; if(r == kNoReg) { r = spillReg(out, o, i, false); @@ -302,6 +375,7 @@ namespace rat { out.push_back(std::move(s)); } + // copies inside a bundle vanish void detail::RegAllocFunc::rewrite() { List out; for(U32 b = 0; b < fn.blocks.size(); ++b) { @@ -311,6 +385,8 @@ namespace rat { for(U32 k = 0; k < insts.size(); ++k) { MachineInstr& in = insts[k]; U32 i = blockFirst[b] + k; + if(isCopy(in) && sameBundle(in.defs[0], in.uses[0])) + continue; rewriteInstr(out, in, i); } insts.swap(out); diff --git a/src/backend/codegen/reg_alloc.h b/src/backend/codegen/reg_alloc.h index 3666c6f..ef0b60f 100644 --- a/src/backend/codegen/reg_alloc.h +++ b/src/backend/codegen/reg_alloc.h @@ -15,6 +15,7 @@ namespace rat { struct RaInterval { List segs; F32 weight = 0; + VReg root = kNoVReg; PhysReg reg = kNoReg; I32 slot = 0; }; @@ -29,7 +30,12 @@ namespace rat { void liveness(); void buildIntervals(); void addSeg(VReg v, I32 start, I32 end); + void noteCopy(const MachineInstr& in, U32 weight); // assignment + void coalesce(); + VReg find(VReg v); + B32 overlaps(VReg a, VReg b) const; + void merge(VReg a, VReg b); void assignRegs(); PhysReg pick(VReg v) const; // rewrite @@ -39,6 +45,8 @@ namespace rat { PhysReg pickTemp(U32 cls, U64 hard, U64 soft); // queries B32 isCopy(const MachineInstr& in) const; + const RaInterval& bundle(VReg v) const; + B32 sameBundle(const MachineOperand& a, const MachineOperand& b) const; private: MachineFunc& fn; const RegisterInfo& ri; @@ -50,11 +58,12 @@ namespace rat { List blockFirst; // block -> first instruction, one past the end at the back List busy; // slot -> busy physical registers U64 usedCallee = 0; - // liveness + // liveness and bundles List> liveOut; // block -> live-out vregs List iv; + List>> copies; // (~weight, (def, source)) // rewrite of the current instruction - List> temps; // spilled vreg -> its temp + List> temps; // spilled bundle -> its temp U64 taken = 0; // temps U64 own = 0; // clobbers List stores; From 6d632f5ebc3d06d361e45b3588576678d820da31 Mon Sep 17 00:00:00 2001 From: nella Date: Sat, 26 Sep 2026 00:53:08 +0200 Subject: [PATCH 06/12] Cap bundle size. --- src/backend/codegen/reg_alloc.cpp | 5 ++++- 1 file changed, 4 insertions(+), 1 deletion(-) diff --git a/src/backend/codegen/reg_alloc.cpp b/src/backend/codegen/reg_alloc.cpp index 9c7c947..999b442 100644 --- a/src/backend/codegen/reg_alloc.cpp +++ b/src/backend/codegen/reg_alloc.cpp @@ -11,6 +11,8 @@ namespace rat { // per-loop-level operand weight constexpr U32 kLoopUseWeight = 3; constexpr U32 kMaxUseWeight = 100000; + // a bundle stops growing past this many segments, so merging stays cheap + constexpr U32 kMaxBundleSegs = 256; PhysReg firstFree(const List& regs, U64 blocked) { for(PhysReg p : regs) @@ -263,7 +265,8 @@ namespace rat { for(const auto& [cold, pair] : copies) { VReg a = find(pair.first); VReg b = find(pair.second); - if(a != b && !overlaps(a, b)) + if(a != b && iv[a].segs.size() + iv[b].segs.size() <= detail::kMaxBundleSegs && + !overlaps(a, b)) merge(a, b); } for(VReg v = 1; v < nv; ++v) { From 0da3e4d53c0dc18b63cc76d1b5ae7d1a5eab8d69 Mon Sep 17 00:00:00 2001 From: nella Date: Sat, 26 Sep 2026 01:44:26 +0200 Subject: [PATCH 07/12] Hint fixed registers. --- src/backend/codegen/reg_alloc.cpp | 9 +++++++++ src/backend/codegen/reg_alloc.h | 1 + 2 files changed, 10 insertions(+) diff --git a/src/backend/codegen/reg_alloc.cpp b/src/backend/codegen/reg_alloc.cpp index 999b442..e6502f2 100644 --- a/src/backend/codegen/reg_alloc.cpp +++ b/src/backend/codegen/reg_alloc.cpp @@ -156,6 +156,10 @@ namespace rat { const MachineOperand& s = in.uses[0]; if(d.isVReg() && s.isVReg() && fn.vregClass[d.vreg] == fn.vregClass[s.vreg]) copies.push_back({~0u - weight, {d.vreg, s.vreg}}); + else if(d.isVReg() && s.isPhys()) + iv[d.vreg].hint = s.phys; + else if(d.isPhys() && s.isVReg()) + iv[s.vreg].hint = d.phys; } // backward walk per block from its live-out set, blocks in reverse @@ -255,6 +259,8 @@ namespace rat { t.segs.emplace_back(start, end); o.segs = {}; t.weight += o.weight; + if(t.hint == kNoReg) + t.hint = o.hint; o.root = a; } @@ -285,6 +291,9 @@ namespace rat { for(const auto& [start, end] : iv[v].segs) for(I32 s = start; s <= end && blocked != ~0ull; ++s) blocked |= busy[(U64)s]; + PhysReg hint = iv[v].hint; + if(hint != kNoReg && !((blocked >> hint) & 1)) + return hint; return detail::firstFree(rc.allocatable, blocked); } diff --git a/src/backend/codegen/reg_alloc.h b/src/backend/codegen/reg_alloc.h index ef0b60f..683687a 100644 --- a/src/backend/codegen/reg_alloc.h +++ b/src/backend/codegen/reg_alloc.h @@ -16,6 +16,7 @@ namespace rat { List segs; F32 weight = 0; VReg root = kNoVReg; + PhysReg hint = kNoReg; PhysReg reg = kNoReg; I32 slot = 0; }; From 5b52e5bdda1e06d305b78e85d6e28349105b25e1 Mon Sep 17 00:00:00 2001 From: nella Date: Sat, 26 Sep 2026 02:31:51 +0200 Subject: [PATCH 08/12] Reload and spill at copies. --- src/backend/codegen/reg_alloc.cpp | 29 ++++++++++++++++++++++++++++- src/backend/codegen/reg_alloc.h | 2 ++ 2 files changed, 30 insertions(+), 1 deletion(-) diff --git a/src/backend/codegen/reg_alloc.cpp b/src/backend/codegen/reg_alloc.cpp index e6502f2..fc86adc 100644 --- a/src/backend/codegen/reg_alloc.cpp +++ b/src/backend/codegen/reg_alloc.cpp @@ -387,6 +387,33 @@ namespace rat { out.push_back(std::move(s)); } + // a copy between a register and a spilled bundle is its reload or its spill + B32 detail::RegAllocFunc::rewriteCopy(List& out, const MachineInstr& in, U32 i) { + const MachineOperand& d = in.defs[0]; + const MachineOperand& s = in.uses[0]; + if(!(d.isVReg() || d.isPhys()) || !(s.isVReg() || s.isPhys())) + return false; + PhysReg dr = regOf(d); + PhysReg sr = regOf(s); + if(dr != kNoReg && sr != kNoReg) + return false; + if(sr == kNoReg) { + if(dr == kNoReg) + dr = pickTemp(fn.vregClass[s.vreg], busy[2 * (U64)i] | busy[2 * (U64)i + 1], 0); + out.push_back(hooks.makeReload(dr, bundle(s.vreg).slot, fn.vregClass[s.vreg], s.width)); + sr = dr; + } + if(d.isVReg() && bundle(d.vreg).reg == kNoReg) + out.push_back(hooks.makeSpill(bundle(d.vreg).slot, sr, fn.vregClass[d.vreg], d.width)); + return true; + } + + PhysReg detail::RegAllocFunc::regOf(const MachineOperand& o) const { + if(o.isPhys()) + return o.phys; + return bundle(o.vreg).reg; + } + // copies inside a bundle vanish void detail::RegAllocFunc::rewrite() { List out; @@ -397,7 +424,7 @@ namespace rat { for(U32 k = 0; k < insts.size(); ++k) { MachineInstr& in = insts[k]; U32 i = blockFirst[b] + k; - if(isCopy(in) && sameBundle(in.defs[0], in.uses[0])) + if(isCopy(in) && (sameBundle(in.defs[0], in.uses[0]) || rewriteCopy(out, in, i))) continue; rewriteInstr(out, in, i); } diff --git a/src/backend/codegen/reg_alloc.h b/src/backend/codegen/reg_alloc.h index 683687a..f7641fe 100644 --- a/src/backend/codegen/reg_alloc.h +++ b/src/backend/codegen/reg_alloc.h @@ -41,6 +41,8 @@ namespace rat { PhysReg pick(VReg v) const; // rewrite void rewrite(); + B32 rewriteCopy(List& out, const MachineInstr& in, U32 i); + PhysReg regOf(const MachineOperand& o) const; void rewriteInstr(List& out, MachineInstr& in, U32 i); PhysReg spillReg(List& out, const MachineOperand& o, U32 i, B32 use); PhysReg pickTemp(U32 cls, U64 hard, U64 soft); From fb37f42b5ab0001798d5587329a71b13a076caf7 Mon Sep 17 00:00:00 2001 From: nella Date: Sat, 26 Sep 2026 03:38:12 +0200 Subject: [PATCH 09/12] Reuse spill slots. --- src/backend/codegen/reg_alloc.cpp | 33 ++++++++++++++++++++++++++----- src/backend/codegen/reg_alloc.h | 1 + 2 files changed, 29 insertions(+), 5 deletions(-) diff --git a/src/backend/codegen/reg_alloc.cpp b/src/backend/codegen/reg_alloc.cpp index fc86adc..3c9190b 100644 --- a/src/backend/codegen/reg_alloc.cpp +++ b/src/backend/codegen/reg_alloc.cpp @@ -303,16 +303,13 @@ namespace rat { if(!iv[v].segs.empty()) order.emplace_back(-iv[v].weight, v); std::sort(order.begin(), order.end()); + List> spilled; // (start, bundle) for(const auto& [negWeight, v] : order) { RaInterval& t = iv[v]; assert(!ri.classes[fn.vregClass[v]].scratch.empty() && "vreg of a class with no scratch"); t.reg = pick(v); if(t.reg == kNoReg) { - const RegClass& rc = ri.classes[fn.vregClass[v]]; - U32 bytes = rc.spillBytes; - if(!bytes) - bytes = ri.spillSlotBytes; - t.slot = hooks.allocSlot(fn, rc.id, bytes); + spilled.emplace_back(t.segs.front().first, v); continue; } usedCallee |= ((U64)1 << t.reg) & calleeMask; @@ -320,6 +317,32 @@ namespace rat { for(I32 s = start; s <= end; ++s) busy[(U64)s] |= (U64)1 << t.reg; } + assignSlots(spilled); + } + + void detail::RegAllocFunc::assignSlots(List>& spilled) { + std::sort(spilled.begin(), spilled.end()); + List> pool[kMaxRegClasses]; // (slot, end of its last holder) + for(const auto& [start, v] : spilled) { + RaInterval& t = iv[v]; + U32 cls = fn.vregClass[v]; + I32 end = t.segs.back().second; + B32 reused = false; + for(auto& [slot, freeAt] : pool[cls]) + if(freeAt < start) { + t.slot = slot; + freeAt = end; + reused = true; + break; + } + if(reused) + continue; + U32 bytes = ri.classes[cls].spillBytes; + if(!bytes) + bytes = ri.spillSlotBytes; + t.slot = hooks.allocSlot(fn, cls, bytes); + pool[cls].emplace_back(t.slot, end); + } } PhysReg detail::RegAllocFunc::pickTemp(U32 cls, U64 hard, U64 soft) { diff --git a/src/backend/codegen/reg_alloc.h b/src/backend/codegen/reg_alloc.h index f7641fe..b008824 100644 --- a/src/backend/codegen/reg_alloc.h +++ b/src/backend/codegen/reg_alloc.h @@ -38,6 +38,7 @@ namespace rat { B32 overlaps(VReg a, VReg b) const; void merge(VReg a, VReg b); void assignRegs(); + void assignSlots(List>& spilled); PhysReg pick(VReg v) const; // rewrite void rewrite(); From 3cc33bb4a102258ce960d71e146cf94f05a06d2f Mon Sep 17 00:00:00 2001 From: nella Date: Sat, 26 Sep 2026 04:22:45 +0200 Subject: [PATCH 10/12] Scan busy slots by chunk. --- src/backend/codegen/reg_alloc.cpp | 17 ++++++++++++++--- src/backend/codegen/reg_alloc.h | 1 + 2 files changed, 15 insertions(+), 3 deletions(-) diff --git a/src/backend/codegen/reg_alloc.cpp b/src/backend/codegen/reg_alloc.cpp index 3c9190b..cf7abe4 100644 --- a/src/backend/codegen/reg_alloc.cpp +++ b/src/backend/codegen/reg_alloc.cpp @@ -63,6 +63,9 @@ namespace rat { busy.assign(2 * (U64)blockFirst.back(), 0); for(U32 b = 0; b < fn.blocks.size(); ++b) pinFixed(b); + chunk.assign((busy.size() + 63) / 64, 0); + for(U64 s = 0; s < busy.size(); ++s) + chunk[s >> 6] |= busy[s]; } // a fixed register is busy from its def to its last use in the block (call argument @@ -289,8 +292,14 @@ namespace rat { const RegClass& rc = ri.classes[fn.vregClass[v]]; U64 blocked = ~allocMask[rc.id]; for(const auto& [start, end] : iv[v].segs) - for(I32 s = start; s <= end && blocked != ~0ull; ++s) - blocked |= busy[(U64)s]; + for(I32 s = start; s <= end && blocked != ~0ull;) { + if((s & 63) == 0 && s + 63 <= end) { + blocked |= chunk[(U64)s >> 6]; + s += 64; + } else { + blocked |= busy[(U64)s++]; + } + } PhysReg hint = iv[v].hint; if(hint != kNoReg && !((blocked >> hint) & 1)) return hint; @@ -314,8 +323,10 @@ namespace rat { } usedCallee |= ((U64)1 << t.reg) & calleeMask; for(const auto& [start, end] : t.segs) - for(I32 s = start; s <= end; ++s) + for(I32 s = start; s <= end; ++s) { busy[(U64)s] |= (U64)1 << t.reg; + chunk[(U64)s >> 6] |= (U64)1 << t.reg; + } } assignSlots(spilled); } diff --git a/src/backend/codegen/reg_alloc.h b/src/backend/codegen/reg_alloc.h index b008824..21d0a5c 100644 --- a/src/backend/codegen/reg_alloc.h +++ b/src/backend/codegen/reg_alloc.h @@ -61,6 +61,7 @@ namespace rat { // numbering List blockFirst; // block -> first instruction, one past the end at the back List busy; // slot -> busy physical registers + List chunk; // 64 slots -> busy anywhere in them U64 usedCallee = 0; // liveness and bundles List> liveOut; // block -> live-out vregs From 521ec0ea5250279ae03d131351d4fe83436d2758 Mon Sep 17 00:00:00 2001 From: nella Date: Sat, 26 Sep 2026 05:47:30 +0200 Subject: [PATCH 11/12] Simplify. --- src/backend/codegen/reg_alloc.cpp | 145 +++++++++++++++--------------- src/backend/codegen/reg_alloc.h | 116 ++++++++++++------------ 2 files changed, 128 insertions(+), 133 deletions(-) diff --git a/src/backend/codegen/reg_alloc.cpp b/src/backend/codegen/reg_alloc.cpp index cf7abe4..6559a88 100644 --- a/src/backend/codegen/reg_alloc.cpp +++ b/src/backend/codegen/reg_alloc.cpp @@ -22,46 +22,39 @@ namespace rat { } } // namespace detail - detail::RegAllocFunc::RegAllocFunc(MachineFunc& f, const RegisterInfo& r, const RegAllocHooks& h) - : fn(f), - ri(r), - hooks(h), - nv(f.nextVReg) {} - - void detail::RegAllocFunc::run() { - for(const RegClass& rc : ri.classes) { - for(PhysReg p : rc.allocatable) - allocMask[rc.id] |= (U64)1 << p; - for(PhysReg p : rc.calleeSaved) - calleeMask |= (U64)1 << p; - } + void RegAllocPass::allocate(MachineFunc& f) { + fn = &f; + nv = f.nextVReg; + usedCallee = 0; + copies.clear(); + iv.clear(); number(); liveness(); buildIntervals(); coalesce(); assignRegs(); rewrite(); - fn.usedCalleeSaved.clear(); + fn->usedCalleeSaved.clear(); for(U64 m = usedCallee; m; m &= m - 1) - fn.usedCalleeSaved.push_back((PhysReg)countTrailingZeros64(m)); + fn->usedCalleeSaved.push_back((PhysReg)countTrailingZeros64(m)); } - B32 detail::RegAllocFunc::isCopy(const MachineInstr& in) const { + B32 RegAllocPass::isCopy(const MachineInstr& in) const { return hooks.isCopy && in.defs.size() == 1 && in.uses.size() == 1 && hooks.isCopy(in); } - const detail::RaInterval& detail::RegAllocFunc::bundle(VReg v) const { return iv[iv[v].root]; } + const RegAllocPass::Interval& RegAllocPass::bundle(VReg v) const { return iv[iv[v].root]; } - B32 detail::RegAllocFunc::sameBundle(const MachineOperand& a, const MachineOperand& b) const { + B32 RegAllocPass::sameBundle(const MachineOperand& a, const MachineOperand& b) const { return a.isVReg() && b.isVReg() && iv[a.vreg].root == iv[b.vreg].root; } - void detail::RegAllocFunc::number() { + void RegAllocPass::number() { blockFirst.assign(1, 0); - for(const MachineBlock& blk : fn.blocks) + for(const MachineBlock& blk : fn->blocks) blockFirst.push_back(blockFirst.back() + (U32)blk.insts.size()); busy.assign(2 * (U64)blockFirst.back(), 0); - for(U32 b = 0; b < fn.blocks.size(); ++b) + for(U32 b = 0; b < fn->blocks.size(); ++b) pinFixed(b); chunk.assign((busy.size() + 63) / 64, 0); for(U64 s = 0; s < busy.size(); ++s) @@ -70,8 +63,8 @@ namespace rat { // a fixed register is busy from its def to its last use in the block (call argument // windows, div/shift operands, incoming arguments up to their copy), plus clobbers - void detail::RegAllocFunc::pinFixed(U32 b) { - const List& insts = fn.blocks[b].insts; + void RegAllocPass::pinFixed(U32 b) { + const List& insts = fn->blocks[b].insts; U64 live = 0; for(U32 k = (U32)insts.size(); k-- > 0;) { const MachineInstr& in = insts[k]; @@ -97,14 +90,14 @@ namespace rat { } } - void detail::RegAllocFunc::liveness() { - U32 nb = (U32)fn.blocks.size(); + void RegAllocPass::liveness() { + U32 nb = (U32)fn->blocks.size(); List defStamp(nv, 0); List ueStamp(nv, 0); List> defBlocks(nv); List> ueBlocks(nv); for(U32 b = 0; b < nb; ++b) - for(const MachineInstr& in : fn.blocks[b].insts) { + for(const MachineInstr& in : fn->blocks[b].insts) { for(const MachineOperand& o : in.uses) if(o.isVReg() && defStamp[o.vreg] != b + 1 && ueStamp[o.vreg] != b + 1) { ueStamp[o.vreg] = b + 1; @@ -131,7 +124,7 @@ namespace rat { while(!work.empty()) { U32 x = work.back(); work.pop_back(); - for(I32 p : fn.blocks[x].preds) { + for(I32 p : fn->blocks[x].preds) { if(outStamp[p] != v) { outStamp[p] = v; liveOut[p].push_back(v); @@ -146,18 +139,18 @@ namespace rat { } // segments come in descending order per vreg, merge touching ones - void detail::RegAllocFunc::addSeg(VReg v, I32 start, I32 end) { - List& segs = iv[v].segs; + void RegAllocPass::addSeg(VReg v, I32 start, I32 end) { + List& segs = iv[v].segs; if(!segs.empty() && end + 1 >= segs.back().first) segs.back().first = std::min(segs.back().first, start); else segs.emplace_back(start, end); } - void detail::RegAllocFunc::noteCopy(const MachineInstr& in, U32 weight) { + void RegAllocPass::noteCopy(const MachineInstr& in, U32 weight) { const MachineOperand& d = in.defs[0]; const MachineOperand& s = in.uses[0]; - if(d.isVReg() && s.isVReg() && fn.vregClass[d.vreg] == fn.vregClass[s.vreg]) + if(d.isVReg() && s.isVReg() && fn->vregClass[d.vreg] == fn->vregClass[s.vreg]) copies.push_back({~0u - weight, {d.vreg, s.vreg}}); else if(d.isVReg() && s.isPhys()) iv[d.vreg].hint = s.phys; @@ -166,15 +159,15 @@ namespace rat { } // backward walk per block from its live-out set, blocks in reverse - void detail::RegAllocFunc::buildIntervals() { + void RegAllocPass::buildIntervals() { iv.resize(nv); for(VReg v = 0; v < nv; ++v) iv[v].root = v; List live(nv, 0); List segEnd(nv, 0); List liveList; - for(U32 b = (U32)fn.blocks.size(); b-- > 0;) { - const MachineBlock& blk = fn.blocks[b]; + for(U32 b = (U32)fn->blocks.size(); b-- > 0;) { + const MachineBlock& blk = fn->blocks[b]; if(blk.insts.empty()) continue; for(VReg v : liveOut[b]) { @@ -221,20 +214,20 @@ namespace rat { } liveList.clear(); } - for(RaInterval& t : iv) + for(Interval& t : iv) std::reverse(t.segs.begin(), t.segs.end()); } // path halving - VReg detail::RegAllocFunc::find(VReg v) { + VReg RegAllocPass::find(VReg v) { while(iv[v].root != v) v = iv[v].root = iv[iv[v].root].root; return v; } - B32 detail::RegAllocFunc::overlaps(VReg a, VReg b) const { - const List& x = iv[a].segs; - const List& y = iv[b].segs; + B32 RegAllocPass::overlaps(VReg a, VReg b) const { + const List& x = iv[a].segs; + const List& y = iv[b].segs; U32 i = 0; U32 j = 0; while(i < x.size() && j < y.size()) { @@ -249,10 +242,10 @@ namespace rat { } // b joins a, their segments do not overlap - void detail::RegAllocFunc::merge(VReg a, VReg b) { - RaInterval& t = iv[a]; - RaInterval& o = iv[b]; - List segs(t.segs.size() + o.segs.size()); + void RegAllocPass::merge(VReg a, VReg b) { + Interval& t = iv[a]; + Interval& o = iv[b]; + List segs(t.segs.size() + o.segs.size()); std::merge(t.segs.begin(), t.segs.end(), o.segs.begin(), o.segs.end(), segs.begin()); t.segs.clear(); for(const auto& [start, end] : segs) // join touching segments @@ -269,7 +262,7 @@ namespace rat { // copy-related vregs whose ranges do not overlap share one register or slot, hottest // copies first - void detail::RegAllocFunc::coalesce() { + void RegAllocPass::coalesce() { std::sort(copies.begin(), copies.end()); for(const auto& [cold, pair] : copies) { VReg a = find(pair.first); @@ -288,8 +281,8 @@ namespace rat { } } - PhysReg detail::RegAllocFunc::pick(VReg v) const { - const RegClass& rc = ri.classes[fn.vregClass[v]]; + PhysReg RegAllocPass::pick(VReg v) const { + const RegClass& rc = ri->classes[fn->vregClass[v]]; U64 blocked = ~allocMask[rc.id]; for(const auto& [start, end] : iv[v].segs) for(I32 s = start; s <= end && blocked != ~0ull;) { @@ -306,7 +299,7 @@ namespace rat { return detail::firstFree(rc.allocatable, blocked); } - void detail::RegAllocFunc::assignRegs() { + void RegAllocPass::assignRegs() { List> order; // (-weight, bundle) for(VReg v = 1; v < nv; ++v) if(!iv[v].segs.empty()) @@ -314,8 +307,8 @@ namespace rat { std::sort(order.begin(), order.end()); List> spilled; // (start, bundle) for(const auto& [negWeight, v] : order) { - RaInterval& t = iv[v]; - assert(!ri.classes[fn.vregClass[v]].scratch.empty() && "vreg of a class with no scratch"); + Interval& t = iv[v]; + assert(!ri->classes[fn->vregClass[v]].scratch.empty() && "vreg of a class with no scratch"); t.reg = pick(v); if(t.reg == kNoReg) { spilled.emplace_back(t.segs.front().first, v); @@ -331,12 +324,12 @@ namespace rat { assignSlots(spilled); } - void detail::RegAllocFunc::assignSlots(List>& spilled) { + void RegAllocPass::assignSlots(List>& spilled) { std::sort(spilled.begin(), spilled.end()); List> pool[kMaxRegClasses]; // (slot, end of its last holder) for(const auto& [start, v] : spilled) { - RaInterval& t = iv[v]; - U32 cls = fn.vregClass[v]; + Interval& t = iv[v]; + U32 cls = fn->vregClass[v]; I32 end = t.segs.back().second; B32 reused = false; for(auto& [slot, freeAt] : pool[cls]) @@ -348,16 +341,16 @@ namespace rat { } if(reused) continue; - U32 bytes = ri.classes[cls].spillBytes; + U32 bytes = ri->classes[cls].spillBytes; if(!bytes) - bytes = ri.spillSlotBytes; - t.slot = hooks.allocSlot(fn, cls, bytes); + bytes = ri->spillSlotBytes; + t.slot = hooks.allocSlot(*fn, cls, bytes); pool[cls].emplace_back(t.slot, end); } } - PhysReg detail::RegAllocFunc::pickTemp(U32 cls, U64 hard, U64 soft) { - const RegClass& rc = ri.classes[cls]; + PhysReg RegAllocPass::pickTemp(U32 cls, U64 hard, U64 soft) { + const RegClass& rc = ri->classes[cls]; PhysReg p = detail::firstFree(rc.scratch, hard | soft); if(p == kNoReg) p = detail::firstFree(rc.allocatable, hard | soft); @@ -368,13 +361,12 @@ namespace rat { return p; } - PhysReg - detail::RegAllocFunc::spillReg(List& out, const MachineOperand& o, U32 i, B32 use) { + PhysReg RegAllocPass::spillReg(List& out, const MachineOperand& o, U32 i, B32 use) { VReg root = iv[o.vreg].root; for(const auto& [v, r] : temps) if(v == root) return r; - U32 cls = fn.vregClass[o.vreg]; + U32 cls = fn->vregClass[o.vreg]; U64 hard = (busy[2 * (U64)i] | busy[2 * (U64)i + 1]) & ~own; if(use) hard |= taken; @@ -386,7 +378,7 @@ namespace rat { return r; } - void detail::RegAllocFunc::rewriteInstr(List& out, MachineInstr& in, U32 i) { + void RegAllocPass::rewriteInstr(List& out, MachineInstr& in, U32 i) { temps.clear(); taken = 0; own = 0; @@ -408,11 +400,11 @@ namespace rat { for(MachineOperand& o : in.defs) { if(!o.isVReg()) continue; - const RaInterval& t = bundle(o.vreg); + const Interval& t = bundle(o.vreg); PhysReg r = t.reg; if(r == kNoReg) { r = spillReg(out, o, i, false); - stores.push_back(hooks.makeSpill(t.slot, r, fn.vregClass[o.vreg], o.width)); + stores.push_back(hooks.makeSpill(t.slot, r, fn->vregClass[o.vreg], o.width)); } o = MachineOperand::fixed(r, o.width); } @@ -422,7 +414,7 @@ namespace rat { } // a copy between a register and a spilled bundle is its reload or its spill - B32 detail::RegAllocFunc::rewriteCopy(List& out, const MachineInstr& in, U32 i) { + B32 RegAllocPass::rewriteCopy(List& out, const MachineInstr& in, U32 i) { const MachineOperand& d = in.defs[0]; const MachineOperand& s = in.uses[0]; if(!(d.isVReg() || d.isPhys()) || !(s.isVReg() || s.isPhys())) @@ -433,26 +425,26 @@ namespace rat { return false; if(sr == kNoReg) { if(dr == kNoReg) - dr = pickTemp(fn.vregClass[s.vreg], busy[2 * (U64)i] | busy[2 * (U64)i + 1], 0); - out.push_back(hooks.makeReload(dr, bundle(s.vreg).slot, fn.vregClass[s.vreg], s.width)); + dr = pickTemp(fn->vregClass[s.vreg], busy[2 * (U64)i] | busy[2 * (U64)i + 1], 0); + out.push_back(hooks.makeReload(dr, bundle(s.vreg).slot, fn->vregClass[s.vreg], s.width)); sr = dr; } if(d.isVReg() && bundle(d.vreg).reg == kNoReg) - out.push_back(hooks.makeSpill(bundle(d.vreg).slot, sr, fn.vregClass[d.vreg], d.width)); + out.push_back(hooks.makeSpill(bundle(d.vreg).slot, sr, fn->vregClass[d.vreg], d.width)); return true; } - PhysReg detail::RegAllocFunc::regOf(const MachineOperand& o) const { + PhysReg RegAllocPass::regOf(const MachineOperand& o) const { if(o.isPhys()) return o.phys; return bundle(o.vreg).reg; } // copies inside a bundle vanish - void detail::RegAllocFunc::rewrite() { + void RegAllocPass::rewrite() { List out; - for(U32 b = 0; b < fn.blocks.size(); ++b) { - List& insts = fn.blocks[b].insts; + for(U32 b = 0; b < fn->blocks.size(); ++b) { + List& insts = fn->blocks[b].insts; out.clear(); out.reserve(insts.size()); for(U32 k = 0; k < insts.size(); ++k) { @@ -467,10 +459,17 @@ namespace rat { } B32 RegAllocPass::run(Module& module, MachineModule& mm, const TargetInfo& target) { + ri = target.registers(); + hooks = target.regAllocHooks(); + for(const RegClass& rc : ri->classes) { + for(PhysReg p : rc.allocatable) + allocMask[rc.id] |= (U64)1 << p; + for(PhysReg p : rc.calleeSaved) + calleeMask |= (U64)1 << p; + } B32 changed = false; - RegAllocHooks hooks = target.regAllocHooks(); for(const Function* f : module) { - detail::RegAllocFunc(mm.get(f), *target.registers(), hooks).run(); + allocate(mm.get(f)); changed = true; } return changed; diff --git a/src/backend/codegen/reg_alloc.h b/src/backend/codegen/reg_alloc.h index 21d0a5c..2c85479 100644 --- a/src/backend/codegen/reg_alloc.h +++ b/src/backend/codegen/reg_alloc.h @@ -9,11 +9,16 @@ namespace rat { namespace detail { PhysReg firstFree(const List& regs, U64 blocked); + } // namespace detail - using RaSeg = Pair; + struct RegAllocPass : MachinePass { + const C8* name() const override { return "regalloc"; } + B32 run(Module& module, MachineModule& mm, const TargetInfo& target) override; + private: + using Seg = Pair; - struct RaInterval { - List segs; + struct Interval { + List segs; F32 weight = 0; VReg root = kNoVReg; PhysReg hint = kNoReg; @@ -21,63 +26,54 @@ namespace rat { I32 slot = 0; }; - struct RegAllocFunc { - RegAllocFunc(MachineFunc& f, const RegisterInfo& r, const RegAllocHooks& h); - void run(); - private: - // intervals - void number(); - void pinFixed(U32 b); - void liveness(); - void buildIntervals(); - void addSeg(VReg v, I32 start, I32 end); - void noteCopy(const MachineInstr& in, U32 weight); - // assignment - void coalesce(); - VReg find(VReg v); - B32 overlaps(VReg a, VReg b) const; - void merge(VReg a, VReg b); - void assignRegs(); - void assignSlots(List>& spilled); - PhysReg pick(VReg v) const; - // rewrite - void rewrite(); - B32 rewriteCopy(List& out, const MachineInstr& in, U32 i); - PhysReg regOf(const MachineOperand& o) const; - void rewriteInstr(List& out, MachineInstr& in, U32 i); - PhysReg spillReg(List& out, const MachineOperand& o, U32 i, B32 use); - PhysReg pickTemp(U32 cls, U64 hard, U64 soft); - // queries - B32 isCopy(const MachineInstr& in) const; - const RaInterval& bundle(VReg v) const; - B32 sameBundle(const MachineOperand& a, const MachineOperand& b) const; - private: - MachineFunc& fn; - const RegisterInfo& ri; - const RegAllocHooks& hooks; - U32 nv; - U64 allocMask[kMaxRegClasses] = {}; - U64 calleeMask = 0; - // numbering - List blockFirst; // block -> first instruction, one past the end at the back - List busy; // slot -> busy physical registers - List chunk; // 64 slots -> busy anywhere in them - U64 usedCallee = 0; - // liveness and bundles - List> liveOut; // block -> live-out vregs - List iv; - List>> copies; // (~weight, (def, source)) - // rewrite of the current instruction - List> temps; // spilled bundle -> its temp - U64 taken = 0; // temps - U64 own = 0; // clobbers - List stores; - }; - } // namespace detail - - struct RegAllocPass : MachinePass { - const C8* name() const override { return "regalloc"; } - B32 run(Module& module, MachineModule& mm, const TargetInfo& target) override; + void allocate(MachineFunc& f); + // intervals + void number(); + void pinFixed(U32 b); + void liveness(); + void buildIntervals(); + void addSeg(VReg v, I32 start, I32 end); + void noteCopy(const MachineInstr& in, U32 weight); + // assignment + void coalesce(); + VReg find(VReg v); + B32 overlaps(VReg a, VReg b) const; + void merge(VReg a, VReg b); + void assignRegs(); + void assignSlots(List>& spilled); + PhysReg pick(VReg v) const; + // rewrite + void rewrite(); + B32 rewriteCopy(List& out, const MachineInstr& in, U32 i); + PhysReg regOf(const MachineOperand& o) const; + void rewriteInstr(List& out, MachineInstr& in, U32 i); + PhysReg spillReg(List& out, const MachineOperand& o, U32 i, B32 use); + PhysReg pickTemp(U32 cls, U64 hard, U64 soft); + // queries + B32 isCopy(const MachineInstr& in) const; + const Interval& bundle(VReg v) const; + B32 sameBundle(const MachineOperand& a, const MachineOperand& b) const; + private: + MachineFunc* fn = nullptr; + const RegisterInfo* ri = nullptr; + RegAllocHooks hooks; + U32 nv = 0; + U64 allocMask[kMaxRegClasses] = {}; + U64 calleeMask = 0; + // numbering + List blockFirst; // block -> first instruction, one past the end at the back + List busy; // slot -> busy physical registers + List chunk; // 64 slots -> busy anywhere in them + U64 usedCallee = 0; + // liveness and bundles + List> liveOut; // block -> live-out vregs + List iv; + List>> copies; // (~weight, (def, source)) + // rewrite of the current instruction + List> temps; // spilled bundle -> its temp + U64 taken = 0; // temps + U64 own = 0; // clobbers + List stores; }; } // namespace rat From 6d38f007d42c1df2a48bd93a71518591bf201b37 Mon Sep 17 00:00:00 2001 From: nella Date: Sat, 26 Sep 2026 06:49:17 +0200 Subject: [PATCH 12/12] Docs. --- src/backend/README.md | 7 +++++-- 1 file changed, 5 insertions(+), 2 deletions(-) diff --git a/src/backend/README.md b/src/backend/README.md index 2fe61da..e353e3a 100644 --- a/src/backend/README.md +++ b/src/backend/README.md @@ -114,8 +114,11 @@ Code generation turns the floating graph back into linear code in three stages: [`CodeGen/MachineFunction.h:`](./codegen/machine_function.h) a minimal, target-independent instruction form: blocks of `MachineInstr`s with defs, uses, clobbers, a register class, and backend-defined immediates. Operands are virtual registers, physical registers, immediates, frame slots, symbols, or block references. Calls are flagged so allocators apply clobbers and bound live ranges correctly. ## register allocation -The allocator builds on [`CodeGen/RegAllocBase.h`](./codegen/reg_alloc_base.h): per-block liveness over dense vreg bitsets, a linearized instruction order, copy-hint collection, and a common rewrite step that patches assignments in and inserts spills/reloads. It is fully backend-agnostic (register classes come from the target's `RegisterInfo`, and spill/reload/slot construction goes through `RegAllocHooks` callbacks), so the allocator never names a single target opcode. -- [**linear scan:**](./codegen/linear_scan_reg_alloc.h) Live ranges are hole-aware segment lists: the gaps between segments are provably off every def-use path, so fixed-register pins inside a hole don't constrain the value and call clobbers inside a hole don't force a callee-saved register. Assignment scans ranges in start order per class. Copy hints bias the choice so coalescable moves become elided self-moves. Under pressure a value spills to a frame slot. +[`CodeGen/RegAlloc.h`:](./codegen/reg_alloc.h) priority bin-packing over a per-slot register bitmap. It is fully backend-agnostic (register classes come from the target's `RegisterInfo`, and spill/reload/slot construction goes through `RegAllocHooks` callbacks), so the allocator never names a single target opcode. +- **slots:** instruction `i` reads at slot `2i` and writes at `2i + 1`. A copy's source dies at the read slot, so the copy's def can take its register; other uses never share a register with a def of the same instruction. +- **live ranges:** exact per-block liveness gives each vreg a segment list with lifetime holes. Fixed-register operands, call argument windows and clobbers are marked busy up front, so values live across a call land in callee-saved registers. +- **coalescing:** copy-related vregs whose ranges do not overlap merge into one bundle; the copies inside it are deleted. +- **assignment:** bundles in descending spill weight (loop-weighted uses over the square root of the length) take their copy hint or the first register free over all their segments. A bundle with none is spilled whole to a slot; spill code goes through the scratch registers, which are never allocated. # x86-64 backend Two machine passes: