From ffd00837c051f327ac7456feda9b4ff26782f99f Mon Sep 17 00:00:00 2001 From: Demetrios Chiuratto Agourakis Date: Sat, 26 Sep 2026 02:22:20 +0000 Subject: [PATCH 1/6] Cranelift: fall back to a DAG cost when scalar egraph costs saturate The scalar sum recounts a shared operand, so a chain of iadd x, x saturates to infinity and extraction can no longer prefer the original value over an identity. Keep that sum on the fast path. When any cost saturates, recompute once, charging each instruction a single time. The instruction-set cost from #12230 is not used for every function: that made compilation 1.1x-1.3x slower. This is the fallback suggested when that PR was closed. Co-Authored-By: Claude --- cranelift/codegen/src/egraph/cost.rs | 82 ++++++++++++++++- cranelift/codegen/src/egraph/elaborate.rs | 73 +++++++++++++++- .../filetests/egraph/cost-function.clif | 87 +++++++++++++++++++ 3 files changed, 237 insertions(+), 5 deletions(-) create mode 100644 cranelift/filetests/filetests/egraph/cost-function.clif diff --git a/cranelift/codegen/src/egraph/cost.rs b/cranelift/codegen/src/egraph/cost.rs index ee4a40ade376..8de53dcfd3f3 100644 --- a/cranelift/codegen/src/egraph/cost.rs +++ b/cranelift/codegen/src/egraph/cost.rs @@ -1,6 +1,86 @@ //! Cost functions for egraph representation. -use crate::ir::Opcode; +use crate::ir::{DataFlowGraph, Inst, Opcode}; +use alloc::vec::Vec; +use core::cmp::Ordering; +use cranelift_entity::EntityRef; + +/// Cost of an expression as a DAG of instructions. +/// +/// The total counts each instruction once. A value used twice by the same +/// expression, as in `iadd x, x`, does not pay for `x` twice. This is the +/// cold path: it runs only for a function whose scalar costs saturated. +#[derive(Clone, Debug)] +pub(crate) struct ExprCost { + total: Cost, + /// Sorted instruction indices. + insts: Vec, +} + +impl ExprCost { + pub(crate) fn zero() -> Self { + Self { + total: Cost::zero(), + insts: Vec::new(), + } + } + + pub(crate) fn total(&self) -> Cost { + self.total + } + + pub(crate) fn for_inst(dfg: &DataFlowGraph, inst: Inst) -> Self { + Self { + total: Cost::of_opcode(dfg.insts[inst].opcode()), + insts: vec![u32::try_from(inst.index()).unwrap()], + } + } + + /// Union `other` into `self`, adding an opcode cost only for instructions + /// that were not already required. + pub(crate) fn add(&mut self, dfg: &DataFlowGraph, other: &Self) { + if other.insts.is_empty() { + return; + } + if self.insts.is_empty() { + *self = other.clone(); + return; + } + let mut merged = Vec::with_capacity(self.insts.len() + other.insts.len()); + let mut i = 0; + let mut j = 0; + while i < self.insts.len() && j < other.insts.len() { + match self.insts[i].cmp(&other.insts[j]) { + Ordering::Less => { + merged.push(self.insts[i]); + i += 1; + } + Ordering::Greater => { + let inst = Inst::new(usize::try_from(other.insts[j]).unwrap()); + self.total = self.total + Cost::of_opcode(dfg.insts[inst].opcode()); + merged.push(other.insts[j]); + j += 1; + } + Ordering::Equal => { + merged.push(self.insts[i]); + i += 1; + j += 1; + } + } + } + while i < self.insts.len() { + merged.push(self.insts[i]); + i += 1; + } + while j < other.insts.len() { + let inst = Inst::new(usize::try_from(other.insts[j]).unwrap()); + self.total = self.total + Cost::of_opcode(dfg.insts[inst].opcode()); + merged.push(other.insts[j]); + j += 1; + } + self.insts = merged; + } +} /// A cost of computing some value in the program. /// diff --git a/cranelift/codegen/src/egraph/elaborate.rs b/cranelift/codegen/src/egraph/elaborate.rs index 813320663440..f9969a7c0e0e 100644 --- a/cranelift/codegen/src/egraph/elaborate.rs +++ b/cranelift/codegen/src/egraph/elaborate.rs @@ -2,7 +2,7 @@ //! in CFG nodes. use super::Stats; -use super::cost::Cost; +use super::cost::{Cost, ExprCost}; use crate::ctxhash::NullCtx; use crate::dominator_tree::DominatorTree; use crate::hash_map::Entry as HashEntry; @@ -14,7 +14,7 @@ use crate::trace; use crate::{FxHashMap, FxHashSet}; use alloc::vec::Vec; use cranelift_control::ControlPlane; -use cranelift_entity::{EntitySet, SecondaryMap, packed_option::ReservedValue}; +use cranelift_entity::{EntityRef, EntitySet, SecondaryMap, packed_option::ReservedValue}; use smallvec::{SmallVec, smallvec}; pub(crate) struct Elaborator<'a> { @@ -342,8 +342,9 @@ impl<'a> Elaborator<'a> { sorted } - fn compute_best_values(&mut self) { + fn compute_best_values(&mut self) -> (bool, bool) { let sorted_values = self.topo_sorted_values(); + let mut saturated = false; let best = &mut self.value_to_best_value; @@ -416,6 +417,9 @@ impl<'a> Elaborator<'a> { ); best[value] = BestEntry(cost, value); trace!(" -> cost of value {} = {:?}", value, cost); + if cost == Cost::infinity() { + saturated = true; + } } } }; @@ -448,6 +452,62 @@ impl<'a> Elaborator<'a> { // *any* e-node in the e-class. At worst we will produce suboptimal // code, but never an incorrectness. } + (saturated, use_worst) + } + + /// Recompute best values with instruction sets. + /// + /// Used only after the scalar cost of some value saturated. Paying for + /// each instruction once keeps a chain of `iadd x, x` finite, so an + /// eclass can still prefer the original value over a saturated identity. + fn compute_best_values_with_sharing(&mut self, use_worst: bool) { + let sorted_values = self.topo_sorted_values(); + let n = self.func.dfg.num_values(); + let mut exprs = vec![ExprCost::zero(); n]; + trace!("recomputing saturated eclass costs with instruction sets"); + for value in sorted_values { + let index = value.index(); + match self.func.dfg.value_def(value) { + ValueDef::Union(x, y) => { + let x_best = BestEntry(exprs[x.index()].total(), self.value_to_best_value[x].1); + let y_best = BestEntry(exprs[y.index()].total(), self.value_to_best_value[y].1); + let pick_x = if use_worst { + x_best >= y_best + } else { + x_best <= y_best + }; + let chosen = if pick_x { x.index() } else { y.index() }; + let chosen_expr = exprs[chosen].clone(); + exprs[index] = chosen_expr; + self.value_to_best_value[value] = if pick_x { x_best } else { y_best }; + } + ValueDef::Param(_, _) => { + exprs[index] = ExprCost::zero(); + self.value_to_best_value[value] = BestEntry(Cost::zero(), value); + } + ValueDef::Result(inst, _) => { + if self.func.layout.inst_block(inst).is_some() { + exprs[index] = ExprCost::zero(); + self.value_to_best_value[value] = BestEntry(Cost::zero(), value); + } else { + let operands: SmallVec<[usize; 8]> = self + .func + .dfg + .inst_values(inst) + .map(|operand| operand.index()) + .collect(); + let mut cost = ExprCost::for_inst(&self.func.dfg, inst); + for operand in operands { + cost.add(&self.func.dfg, &exprs[operand]); + } + let total = cost.total(); + exprs[index] = cost; + self.value_to_best_value[value] = BestEntry(total, value); + trace!(" -> shared cost of value {} = {:?}", value, total); + } + } + } + } } /// Elaborate use of an eclass, inserting any needed new @@ -925,7 +985,12 @@ impl<'a> Elaborator<'a> { pub(crate) fn elaborate(&mut self) { self.stats.elaborate_func += 1; self.stats.elaborate_func_pre_insts += self.func.dfg.num_insts() as u64; - self.compute_best_values(); + let (saturated, use_worst) = self.compute_best_values(); + if saturated { + // The scalar sum saturated, so it can no longer order eclasses. + // Recompute once, counting each instruction a single time. + self.compute_best_values_with_sharing(use_worst); + } self.elaborate_domtree(&self.domtree); self.stats.elaborate_func_post_insts += self.func.dfg.num_insts() as u64; } diff --git a/cranelift/filetests/filetests/egraph/cost-function.clif b/cranelift/filetests/filetests/egraph/cost-function.clif new file mode 100644 index 000000000000..10d992b576fa --- /dev/null +++ b/cranelift/filetests/filetests/egraph/cost-function.clif @@ -0,0 +1,87 @@ +;; A chain of `iadd x, x` saturates a scalar cost that recounts shared +;; operands. Extraction must still see through the identity +;; `(x * 2) - x` once that chain is costed as a DAG. + +test optimize precise-output +set opt_level=speed_and_size +target x86_64 + +function %f(i64) -> i64 { + block0(v0: i64): + v1 = iadd v0, v0 + v2 = iadd v1, v1 + v3 = iadd v2, v2 + v4 = iadd v3, v3 + v5 = iadd v4, v4 + v6 = iadd v5, v5 + v7 = iadd v6, v6 + v8 = iadd v7, v7 + v9 = iadd v8, v8 + v10 = iadd v9, v9 + v11 = iadd v10, v10 + v12 = iadd v11, v11 + v13 = iadd v12, v12 + v14 = iadd v13, v13 + v15 = iadd v14, v14 + v16 = iadd v15, v15 + v17 = iadd v16, v16 + v18 = iadd v17, v17 + v19 = iadd v18, v18 + v20 = iadd v19, v19 + v21 = iadd v20, v20 + v22 = iadd v21, v21 + v23 = iadd v22, v22 + v24 = iadd v23, v23 + v25 = iadd v24, v24 + v26 = iadd v25, v25 + v27 = iadd v26, v26 + v28 = iadd v27, v27 + v29 = iadd v28, v28 + v30 = iadd v29, v29 + v31 = iadd v30, v30 + v32 = iadd v31, v31 + v33 = iadd v32, v32 + + v34 = iconst.i64 2 + v35 = imul v33, v34 + v36 = isub v35, v33 + return v36 +} + +; function %f(i64) -> i64 fast { +; block0(v0: i64): +; v1 = iadd v0, v0 +; v2 = iadd v1, v1 +; v3 = iadd v2, v2 +; v4 = iadd v3, v3 +; v5 = iadd v4, v4 +; v6 = iadd v5, v5 +; v7 = iadd v6, v6 +; v8 = iadd v7, v7 +; v9 = iadd v8, v8 +; v10 = iadd v9, v9 +; v11 = iadd v10, v10 +; v12 = iadd v11, v11 +; v13 = iadd v12, v12 +; v14 = iadd v13, v13 +; v15 = iadd v14, v14 +; v16 = iadd v15, v15 +; v17 = iadd v16, v16 +; v18 = iadd v17, v17 +; v19 = iadd v18, v18 +; v20 = iadd v19, v19 +; v21 = iadd v20, v20 +; v22 = iadd v21, v21 +; v23 = iadd v22, v22 +; v24 = iadd v23, v23 +; v25 = iadd v24, v24 +; v26 = iadd v25, v25 +; v27 = iadd v26, v26 +; v28 = iadd v27, v27 +; v29 = iadd v28, v28 +; v30 = iadd v29, v29 +; v31 = iadd v30, v30 +; v32 = iadd v31, v31 +; v33 = iadd v32, v32 +; return v33 +; } From a54a599c3258e04303d7c31bc9d7139e7bdb34b6 Mon Sep 17 00:00:00 2001 From: Demetrios Agourakis Date: Tue, 29 Sep 2026 21:07:33 -0300 Subject: [PATCH 2/6] Cranelift: approximate shared egraph costs inline --- cranelift/codegen/src/egraph/cost.rs | 142 ++++++++++++---------- cranelift/codegen/src/egraph/elaborate.rs | 114 ++++------------- 2 files changed, 97 insertions(+), 159 deletions(-) diff --git a/cranelift/codegen/src/egraph/cost.rs b/cranelift/codegen/src/egraph/cost.rs index 8de53dcfd3f3..8ad5b33a10a0 100644 --- a/cranelift/codegen/src/egraph/cost.rs +++ b/cranelift/codegen/src/egraph/cost.rs @@ -1,84 +1,85 @@ //! Cost functions for egraph representation. -use crate::ir::{DataFlowGraph, Inst, Opcode}; -use alloc::vec::Vec; -use core::cmp::Ordering; +use crate::ir::{Inst, Opcode}; use cranelift_entity::EntityRef; -/// Cost of an expression as a DAG of instructions. +/// Approximate cost of an expression as a DAG of instructions. /// -/// The total counts each instruction once. A value used twice by the same -/// expression, as in `iadd x, x`, does not pay for `x` twice. This is the -/// cold path: it runs only for a function whose scalar costs saturated. -#[derive(Clone, Debug)] +/// In addition to the saturating total cost, this tracks an approximate +/// footprint of the instructions that contribute to the expression. When an +/// operand's whole footprint is already covered, we don't charge its total +/// again. This catches common shared-DAG shapes like `iadd x, x` without +/// allocating precise instruction sets in the egraph extraction hot path. +#[derive(Clone, Copy, Debug, PartialEq, Eq)] pub(crate) struct ExprCost { total: Cost, - /// Sorted instruction indices. - insts: Vec, + inst_buckets: u64, } impl ExprCost { pub(crate) fn zero() -> Self { Self { total: Cost::zero(), - insts: Vec::new(), + inst_buckets: 0, } } - pub(crate) fn total(&self) -> Cost { - self.total + pub(crate) fn infinity() -> Self { + Self { + total: Cost::infinity(), + inst_buckets: 0, + } } - pub(crate) fn for_inst(dfg: &DataFlowGraph, inst: Inst) -> Self { + pub(crate) fn for_inst(inst: Inst, op: Opcode) -> Self { Self { - total: Cost::of_opcode(dfg.insts[inst].opcode()), - insts: vec![u32::try_from(inst.index()).unwrap()], + total: Cost::of_opcode(op), + inst_buckets: Self::inst_bucket(inst), } } - /// Union `other` into `self`, adding an opcode cost only for instructions - /// that were not already required. - pub(crate) fn add(&mut self, dfg: &DataFlowGraph, other: &Self) { - if other.insts.is_empty() { - return; - } - if self.insts.is_empty() { - *self = other.clone(); - return; - } - let mut merged = Vec::with_capacity(self.insts.len() + other.insts.len()); - let mut i = 0; - let mut j = 0; - while i < self.insts.len() && j < other.insts.len() { - match self.insts[i].cmp(&other.insts[j]) { - Ordering::Less => { - merged.push(self.insts[i]); - i += 1; - } - Ordering::Greater => { - let inst = Inst::new(usize::try_from(other.insts[j]).unwrap()); - self.total = self.total + Cost::of_opcode(dfg.insts[inst].opcode()); - merged.push(other.insts[j]); - j += 1; - } - Ordering::Equal => { - merged.push(self.insts[i]); - i += 1; - j += 1; - } - } - } - while i < self.insts.len() { - merged.push(self.insts[i]); - i += 1; + /// Compute the cost of the operation and its given operands. + /// + /// Caller is responsible for checking that the opcode came from an instruction + /// that satisfies `inst_predicates::is_pure_for_egraph()`. + pub(crate) fn of_pure_op( + inst: Inst, + op: Opcode, + operand_costs: impl IntoIterator, + ) -> Self { + let mut cost = Self::for_inst(inst, op); + for operand_cost in operand_costs { + cost.add_operand(operand_cost); } - while j < other.insts.len() { - let inst = Inst::new(usize::try_from(other.insts[j]).unwrap()); - self.total = self.total + Cost::of_opcode(dfg.insts[inst].opcode()); - merged.push(other.insts[j]); - j += 1; + cost + } +} + +impl ExprCost { + fn inst_bucket(inst: Inst) -> u64 { + let index = u64::try_from(inst.index()).unwrap(); + let hash = index.wrapping_mul(0x9e37_79b9_7f4a_7c15); + 1u64 << (hash >> 58) + } + + fn add_operand(&mut self, other: Self) { + let new_buckets = other.inst_buckets & !self.inst_buckets; + if new_buckets != 0 { + self.total = self.total + other.total; } - self.insts = merged; + self.inst_buckets |= other.inst_buckets; + } +} + +impl PartialOrd for ExprCost { + fn partial_cmp(&self, other: &Self) -> Option { + Some(self.cmp(other)) + } +} + +impl Ord for ExprCost { + fn cmp(&self, other: &Self) -> core::cmp::Ordering { + self.total.cmp(&other.total) } } @@ -193,15 +194,6 @@ impl Cost { } } - /// Compute the cost of the operation and its given operands. - /// - /// Caller is responsible for checking that the opcode came from an instruction - /// that satisfies `inst_predicates::is_pure_for_egraph()`. - pub(crate) fn of_pure_op(op: Opcode, operand_costs: impl IntoIterator) -> Self { - let c = Self::of_opcode(op) + operand_costs.into_iter().sum(); - Cost::new(c.cost()) - } - /// Compute the cost of an operation in the side-effectful skeleton. pub(crate) fn of_skeleton_op(op: Opcode, arity: usize) -> Self { Cost::of_opcode(op) + Cost::new(u32::try_from(arity).unwrap()) @@ -255,4 +247,22 @@ mod tests { assert_eq!(a + b, Cost::infinity()); assert_eq!(b + a, Cost::infinity()); } + + #[test] + fn expr_cost_skips_fully_covered_operand() { + let x = ExprCost::for_inst(Inst::new(0), Opcode::Iconst); + let add = ExprCost::of_pure_op(Inst::new(1), Opcode::Iadd, [x, x]); + + assert_eq!(add.total, Cost::new(4)); + } + + #[test] + fn expr_cost_grows_linearly_for_repeated_self_adds() { + let mut cost = ExprCost::for_inst(Inst::new(0), Opcode::Iconst); + for index in 1..4 { + cost = ExprCost::of_pure_op(Inst::new(index), Opcode::Iadd, [cost, cost]); + } + + assert_eq!(cost.total, Cost::new(10)); + } } diff --git a/cranelift/codegen/src/egraph/elaborate.rs b/cranelift/codegen/src/egraph/elaborate.rs index f9969a7c0e0e..3fbe33df6886 100644 --- a/cranelift/codegen/src/egraph/elaborate.rs +++ b/cranelift/codegen/src/egraph/elaborate.rs @@ -2,7 +2,7 @@ //! in CFG nodes. use super::Stats; -use super::cost::{Cost, ExprCost}; +use super::cost::ExprCost; use crate::ctxhash::NullCtx; use crate::dominator_tree::DominatorTree; use crate::hash_map::Entry as HashEntry; @@ -14,7 +14,7 @@ use crate::trace; use crate::{FxHashMap, FxHashSet}; use alloc::vec::Vec; use cranelift_control::ControlPlane; -use cranelift_entity::{EntityRef, EntitySet, SecondaryMap, packed_option::ReservedValue}; +use cranelift_entity::{EntitySet, SecondaryMap, packed_option::ReservedValue}; use smallvec::{SmallVec, smallvec}; pub(crate) struct Elaborator<'a> { @@ -100,7 +100,7 @@ pub(crate) struct Elaborator<'a> { const NOT_ON_LOOP_STACK: u32 = u32::MAX; #[derive(Clone, Copy, Debug, PartialEq, Eq)] -struct BestEntry(Cost, Value); +struct BestEntry(ExprCost, Value); impl PartialOrd for BestEntry { fn partial_cmp(&self, other: &Self) -> Option { @@ -183,7 +183,7 @@ impl<'a> Elaborator<'a> { ) -> Self { let num_values = func.dfg.num_values(); let mut value_to_best_value = - SecondaryMap::with_default(BestEntry(Cost::infinity(), Value::reserved_value())); + SecondaryMap::with_default(BestEntry(ExprCost::infinity(), Value::reserved_value())); value_to_best_value.resize(num_values); Self { func, @@ -342,9 +342,8 @@ impl<'a> Elaborator<'a> { sorted } - fn compute_best_values(&mut self) -> (bool, bool) { + fn compute_best_values(&mut self) { let sorted_values = self.topo_sorted_values(); - let mut saturated = false; let best = &mut self.value_to_best_value; @@ -393,7 +392,7 @@ impl<'a> Elaborator<'a> { } ValueDef::Param(_, _) => { - best[value] = BestEntry(Cost::zero(), value); + best[value] = BestEntry(ExprCost::zero(), value); } // If the Inst is inserted into the layout (which is, @@ -402,13 +401,14 @@ impl<'a> Elaborator<'a> { // cost. ValueDef::Result(inst, _) => { if let Some(_) = self.func.layout.inst_block(inst) { - best[value] = BestEntry(Cost::zero(), value); + best[value] = BestEntry(ExprCost::zero(), value); } else { let inst_data = &self.func.dfg.insts[inst]; // N.B.: at this point we know that the opcode is // pure, so `pure_op_cost`'s precondition is // satisfied. - let cost = Cost::of_pure_op( + let cost = ExprCost::of_pure_op( + inst, inst_data.opcode(), self.func.dfg.inst_values(inst).map(|value| { debug_assert!(!best[value].1.is_reserved_value()); @@ -417,97 +417,30 @@ impl<'a> Elaborator<'a> { ); best[value] = BestEntry(cost, value); trace!(" -> cost of value {} = {:?}", value, cost); - if cost == Cost::infinity() { - saturated = true; - } } } }; // You might be expecting an assert that the best cost we just - // computed is not infinity, however infinite cost *can* happen in - // practice. First, note that our cost function doesn't know about - // any shared structure in the dataflow graph, it only sums operand - // costs. (And trying to avoid that by deduping a single operation's - // operands is a losing game because you can always just add one - // indirection and go from `add(x, x)` to `add(foo(x), bar(x))` to - // hide the shared structure.) Given that blindness to sharing, we - // can make cost grow exponentially with a linear sequence of - // operations: + // computed is not infinity, however infinite cost *can* still + // happen in practice. The expression cost tracks an approximate + // instruction footprint to avoid charging the same already-covered + // operand twice in common shared-DAG shapes. That keeps cases such + // as a chain of `iadd x, x` from growing exponentially, but it is + // still just a bounded heuristic over a 32-bit cost: // // v0 = iconst.i32 1 ;; cost = 1 - // v1 = iadd v0, v0 ;; cost = 3 + 1 + 1 - // v2 = iadd v1, v1 ;; cost = 3 + 5 + 5 - // v3 = iadd v2, v2 ;; cost = 3 + 13 + 13 - // v4 = iadd v3, v3 ;; cost = 3 + 29 + 29 - // v5 = iadd v4, v4 ;; cost = 3 + 61 + 61 - // v6 = iadd v5, v5 ;; cost = 3 + 125 + 125 + // v1 = iadd v0, v0 ;; second v0 already covered + // v2 = iadd v1, v1 ;; second v1 already covered // ;; etc... // - // Such a chain can cause cost to saturate to infinity. How do we - // choose which e-node is best when there are multiple that have - // saturated to infinity? It doesn't matter. As long as invariant - // (2) for optimization rules is upheld by our rule set (see + // If a cost does saturate to infinity, it doesn't matter which + // equally infinite e-node we pick. As long as invariant (2) for + // optimization rules is upheld by our rule set (see // `cranelift/codegen/src/opts/README.md`) it is safe to choose // *any* e-node in the e-class. At worst we will produce suboptimal // code, but never an incorrectness. } - (saturated, use_worst) - } - - /// Recompute best values with instruction sets. - /// - /// Used only after the scalar cost of some value saturated. Paying for - /// each instruction once keeps a chain of `iadd x, x` finite, so an - /// eclass can still prefer the original value over a saturated identity. - fn compute_best_values_with_sharing(&mut self, use_worst: bool) { - let sorted_values = self.topo_sorted_values(); - let n = self.func.dfg.num_values(); - let mut exprs = vec![ExprCost::zero(); n]; - trace!("recomputing saturated eclass costs with instruction sets"); - for value in sorted_values { - let index = value.index(); - match self.func.dfg.value_def(value) { - ValueDef::Union(x, y) => { - let x_best = BestEntry(exprs[x.index()].total(), self.value_to_best_value[x].1); - let y_best = BestEntry(exprs[y.index()].total(), self.value_to_best_value[y].1); - let pick_x = if use_worst { - x_best >= y_best - } else { - x_best <= y_best - }; - let chosen = if pick_x { x.index() } else { y.index() }; - let chosen_expr = exprs[chosen].clone(); - exprs[index] = chosen_expr; - self.value_to_best_value[value] = if pick_x { x_best } else { y_best }; - } - ValueDef::Param(_, _) => { - exprs[index] = ExprCost::zero(); - self.value_to_best_value[value] = BestEntry(Cost::zero(), value); - } - ValueDef::Result(inst, _) => { - if self.func.layout.inst_block(inst).is_some() { - exprs[index] = ExprCost::zero(); - self.value_to_best_value[value] = BestEntry(Cost::zero(), value); - } else { - let operands: SmallVec<[usize; 8]> = self - .func - .dfg - .inst_values(inst) - .map(|operand| operand.index()) - .collect(); - let mut cost = ExprCost::for_inst(&self.func.dfg, inst); - for operand in operands { - cost.add(&self.func.dfg, &exprs[operand]); - } - let total = cost.total(); - exprs[index] = cost; - self.value_to_best_value[value] = BestEntry(total, value); - trace!(" -> shared cost of value {} = {:?}", value, total); - } - } - } - } } /// Elaborate use of an eclass, inserting any needed new @@ -985,12 +918,7 @@ impl<'a> Elaborator<'a> { pub(crate) fn elaborate(&mut self) { self.stats.elaborate_func += 1; self.stats.elaborate_func_pre_insts += self.func.dfg.num_insts() as u64; - let (saturated, use_worst) = self.compute_best_values(); - if saturated { - // The scalar sum saturated, so it can no longer order eclasses. - // Recompute once, counting each instruction a single time. - self.compute_best_values_with_sharing(use_worst); - } + self.compute_best_values(); self.elaborate_domtree(&self.domtree); self.stats.elaborate_func_post_insts += self.func.dfg.num_insts() as u64; } From 75701e2bd61fe0904b951e702219772c68aeb8e6 Mon Sep 17 00:00:00 2001 From: Demetrios Agourakis Date: Tue, 29 Sep 2026 21:26:40 -0300 Subject: [PATCH 3/6] Cranelift: update egraph cost goldens --- .../filetests/isa/x64/iminmax-i128.clif | 37 ++++++++-------- tests/disas/array-fill-i16.wat | 44 +++++++++---------- tests/disas/gc/array-new-default-i16.wat | 9 ++-- tests/disas/x64-optimize-vector-types.wat | 6 ++- 4 files changed, 46 insertions(+), 50 deletions(-) diff --git a/cranelift/filetests/filetests/isa/x64/iminmax-i128.clif b/cranelift/filetests/filetests/isa/x64/iminmax-i128.clif index 851b38bc02a7..46df59e7e140 100644 --- a/cranelift/filetests/filetests/isa/x64/iminmax-i128.clif +++ b/cranelift/filetests/filetests/isa/x64/iminmax-i128.clif @@ -19,19 +19,18 @@ block0: ; pushq %rbp ; movq %rsp, %rbp ; block0: -; uninit %rax -; xorq %rax, %rax ; uninit %rdx ; xorq %rdx, %rdx -; movq %rax, %rdi -; subq $0x0, %rdi -; movq %rdx, %r9 -; sbbq $0x0, %r9 -; cmpq %rax, %rdi -; movq %r9, %r8 +; movq %rdx, %rsi +; subq $0x0, %rsi +; movq %rdx, %rdi +; sbbq $0x0, %rdi +; cmpq %rdx, %rsi +; movq %rdi, %r8 ; sbbq %rdx, %r8 -; cmovbq %rdi, %rax -; cmovbq %r9, %rdx +; movq %rdx, %rax +; cmovbq %rsi, %rax +; cmovbq %rdi, %rdx ; movq %rbp, %rsp ; popq %rbp ; retq @@ -41,17 +40,17 @@ block0: ; pushq %rbp ; movq %rsp, %rbp ; block1: ; offset 0x4 -; xorq %rax, %rax ; xorq %rdx, %rdx -; movq %rax, %rdi -; subq $0, %rdi -; movq %rdx, %r9 -; sbbq $0, %r9 -; cmpq %rax, %rdi -; movq %r9, %r8 +; movq %rdx, %rsi +; subq $0, %rsi +; movq %rdx, %rdi +; sbbq $0, %rdi +; cmpq %rdx, %rsi +; movq %rdi, %r8 ; sbbq %rdx, %r8 -; cmovbq %rdi, %rax -; cmovbq %r9, %rdx +; movq %rdx, %rax +; cmovbq %rsi, %rax +; cmovbq %rdi, %rdx ; movq %rbp, %rsp ; popq %rbp ; retq diff --git a/tests/disas/array-fill-i16.wat b/tests/disas/array-fill-i16.wat index 98840708b3f8..122c527da871 100644 --- a/tests/disas/array-fill-i16.wat +++ b/tests/disas/array-fill-i16.wat @@ -51,18 +51,17 @@ ;; @0031 v36 = load.i64 notrap aligned region3 v7+40 ;; @0031 v24 = iconst.i64 20 ;; @0031 v25 = iadd v9, v24 ; v24 = 20 -;; @0031 v16 = iconst.i64 1 -;; v50 = ishl v14, v16 ; v16 = 1 -;; @0031 v29 = iadd v25, v50 -;; v54 = ishl v15, v16 ; v16 = 1 -;; @0031 v38 = uadd_overflow_trap v29, v54, user2 +;; v49 = iadd v14, v14 +;; @0031 v29 = iadd v25, v49 +;; v53 = iadd v15, v15 +;; @0031 v38 = uadd_overflow_trap v29, v53, user2 ;; @0031 v37 = iadd v8, v36 ;; @0031 v39 = icmp ugt v38, v37 ;; @0031 trapnz v39, user2 ;; v47 = iconst.i64 0 ;; @0031 v42 = icmp eq v15, v47 ; v47 = 0 ;; @0031 v27 = iconst.i64 2 -;; @0031 v40 = iadd v29, v54 +;; @0031 v40 = iadd v29, v53 ;; @0031 brif v42, block3, block2(v29) ;; ;; block2(v43: i64): @@ -110,16 +109,15 @@ ;; @003f v36 = load.i64 notrap aligned region3 v7+40 ;; @003f v24 = iconst.i64 20 ;; @003f v25 = iadd v9, v24 ; v24 = 20 -;; @003f v16 = iconst.i64 1 -;; v44 = ishl v14, v16 ; v16 = 1 -;; @003f v29 = iadd v25, v44 -;; v48 = ishl v15, v16 ; v16 = 1 -;; @003f v38 = uadd_overflow_trap v29, v48, user2 +;; v43 = iadd v14, v14 +;; @003f v29 = iadd v25, v43 +;; v47 = iadd v15, v15 +;; @003f v38 = uadd_overflow_trap v29, v47, user2 ;; @003f v37 = iadd v8, v36 ;; @003f v39 = icmp ugt v38, v37 ;; @003f trapnz v39, user2 ;; @003b v5 = iconst.i32 0 -;; @003f call fn0(v0, v29, v5, v48) ; v5 = 0 +;; @003f call fn0(v0, v29, v5, v47) ; v5 = 0 ;; @0042 jump block1 ;; ;; block1: @@ -157,16 +155,15 @@ ;; @004d v36 = load.i64 notrap aligned region3 v7+40 ;; @004d v24 = iconst.i64 20 ;; @004d v25 = iadd v9, v24 ; v24 = 20 -;; @004d v16 = iconst.i64 1 -;; v44 = ishl v14, v16 ; v16 = 1 -;; @004d v29 = iadd v25, v44 -;; v48 = ishl v15, v16 ; v16 = 1 -;; @004d v38 = uadd_overflow_trap v29, v48, user2 +;; v43 = iadd v14, v14 +;; @004d v29 = iadd v25, v43 +;; v47 = iadd v15, v15 +;; @004d v38 = uadd_overflow_trap v29, v47, user2 ;; @004d v37 = iadd v8, v36 ;; @004d v39 = icmp ugt v38, v37 ;; @004d trapnz v39, user2 ;; @004d v40 = iconst.i32 255 -;; @004d call fn0(v0, v29, v40, v48) ; v40 = 255 +;; @004d call fn0(v0, v29, v40, v47) ; v40 = 255 ;; @0050 jump block1 ;; ;; block1: @@ -203,11 +200,10 @@ ;; @005d v36 = load.i64 notrap aligned region3 v7+40 ;; @005d v24 = iconst.i64 20 ;; @005d v25 = iadd v9, v24 ; v24 = 20 -;; @005d v16 = iconst.i64 1 -;; v50 = ishl v14, v16 ; v16 = 1 -;; @005d v29 = iadd v25, v50 -;; v54 = ishl v15, v16 ; v16 = 1 -;; @005d v38 = uadd_overflow_trap v29, v54, user2 +;; v49 = iadd v14, v14 +;; @005d v29 = iadd v25, v49 +;; v53 = iadd v15, v15 +;; @005d v38 = uadd_overflow_trap v29, v53, user2 ;; @005d v37 = iadd v8, v36 ;; @005d v39 = icmp ugt v38, v37 ;; @005d trapnz v39, user2 @@ -215,7 +211,7 @@ ;; @005d v42 = icmp eq v15, v47 ; v47 = 0 ;; @0057 v5 = iconst.i32 0xdead ;; @005d v27 = iconst.i64 2 -;; @005d v40 = iadd v29, v54 +;; @005d v40 = iadd v29, v53 ;; @005d brif v42, block3, block2(v29) ;; ;; block2(v43: i64): diff --git a/tests/disas/gc/array-new-default-i16.wat b/tests/disas/gc/array-new-default-i16.wat index f2caa8e7f063..9cdfe652ae6d 100644 --- a/tests/disas/gc/array-new-default-i16.wat +++ b/tests/disas/gc/array-new-default-i16.wat @@ -36,10 +36,9 @@ ;; ;; block0(v0: i64, v1: i64, v2: i32): ;; @001f v4 = uextend.i64 v2 -;; v91 = iconst.i64 1 -;; v92 = ishl v4, v91 ; v91 = 1 +;; v90 = iadd v4, v4 ;; @001f v7 = iconst.i64 32 -;; @001f v8 = ushr v92, v7 ; v7 = 32 +;; @001f v8 = ushr v90, v7 ; v7 = 32 ;; @001f trapnz v8, user18 ;; @001f v3 = iconst.i32 20 ;; v96 = iadd v2, v2 @@ -108,12 +107,12 @@ ;; @001f v76 = load.i64 notrap aligned region12 v145+40 ;; @001f v64 = iconst.i64 20 ;; @001f v65 = iadd v49, v64 ; v64 = 20 -;; @001f v78 = uadd_overflow_trap v65, v92, user2 +;; @001f v78 = uadd_overflow_trap v65, v90, user2 ;; @001f v77 = iadd v146, v76 ;; @001f v79 = icmp ugt v78, v77 ;; @001f trapnz v79, user2 ;; @001f v44 = iconst.i32 0 -;; @001f call fn1(v0, v65, v44, v92), stack_map=[i32 @ ss0+0] ; v44 = 0 +;; @001f call fn1(v0, v65, v44, v90), stack_map=[i32 @ ss0+0] ; v44 = 0 ;; @0022 jump block1 ;; ;; block1: diff --git a/tests/disas/x64-optimize-vector-types.wat b/tests/disas/x64-optimize-vector-types.wat index 28316f6823fc..d99db4ff2bda 100644 --- a/tests/disas/x64-optimize-vector-types.wat +++ b/tests/disas/x64-optimize-vector-types.wat @@ -201,10 +201,12 @@ ;; wasm[0]::function[8]: ;; pushq %rbp ;; movq %rsp, %rbp -;; por %xmm0, %xmm1 ;; pcmpeqd %xmm7, %xmm7 +;; movdqa %xmm1, %xmm2 +;; pxor %xmm7, %xmm2 +;; pandn %xmm0, %xmm1 ;; movdqa %xmm1, %xmm0 -;; pcmpeqb %xmm7, %xmm0 +;; pcmpeqb %xmm2, %xmm0 ;; movq %rbp, %rsp ;; popq %rbp ;; retq From 14a88718d4452f55f68f6411c3d3bf4381117c53 Mon Sep 17 00:00:00 2001 From: Demetrios Agourakis Date: Wed, 30 Sep 2026 22:15:17 -0300 Subject: [PATCH 4/6] Cranelift: keep egraph footprint cost compact --- cranelift/codegen/src/egraph/cost.rs | 10 +++++----- 1 file changed, 5 insertions(+), 5 deletions(-) diff --git a/cranelift/codegen/src/egraph/cost.rs b/cranelift/codegen/src/egraph/cost.rs index 8ad5b33a10a0..8c5c4ab11188 100644 --- a/cranelift/codegen/src/egraph/cost.rs +++ b/cranelift/codegen/src/egraph/cost.rs @@ -13,7 +13,7 @@ use cranelift_entity::EntityRef; #[derive(Clone, Copy, Debug, PartialEq, Eq)] pub(crate) struct ExprCost { total: Cost, - inst_buckets: u64, + inst_buckets: u32, } impl ExprCost { @@ -56,10 +56,10 @@ impl ExprCost { } impl ExprCost { - fn inst_bucket(inst: Inst) -> u64 { - let index = u64::try_from(inst.index()).unwrap(); - let hash = index.wrapping_mul(0x9e37_79b9_7f4a_7c15); - 1u64 << (hash >> 58) + fn inst_bucket(inst: Inst) -> u32 { + let index = u32::try_from(inst.index()).unwrap(); + let hash = index.wrapping_mul(0x9e37_79b9); + 1u32 << (hash >> 27) } fn add_operand(&mut self, other: Self) { From 985228bb6d8963149f21708e434091fd2752317a Mon Sep 17 00:00:00 2001 From: Demetrios Agourakis Date: Wed, 30 Sep 2026 23:15:55 -0300 Subject: [PATCH 5/6] Cranelift: update Pulley copy disassembly golden --- tests/disas/pulley-be-inline-copy.wat | 15 ++++++++------- 1 file changed, 8 insertions(+), 7 deletions(-) diff --git a/tests/disas/pulley-be-inline-copy.wat b/tests/disas/pulley-be-inline-copy.wat index 1e0d8627dc02..39e97620eb82 100644 --- a/tests/disas/pulley-be-inline-copy.wat +++ b/tests/disas/pulley-be-inline-copy.wat @@ -17,13 +17,14 @@ ) ;; wasm[0]::function[0]: ;; push_frame -;; xload64be_o32 x4, x0, 64 -;; br_if_xult64_u8 x4, 16, 0x2b // target = 0x35 -;; br_if_xult64_u8 x4, 32, 0x27 // target = 0x38 -;; 18: xload64be_o32 x6, x0, 56 -;; vload128le_o32 v6, x6, 16 -;; vstore128le_o32 x6, 0, v6 +;; xload64be_o32 x5, x0, 64 +;; br_if_xult64_u8 x5, 16, 0x2e // target = 0x38 +;; 11: xconst8 x6, 32 +;; br_if_xult64 x5, x6, 0x27 // target = 0x3b +;; 1b: xload64be_o32 x7, x0, 56 +;; vload128le_o32 v7, x7, 16 +;; vstore128le_o32 x7, 0, v7 ;; pop_frame ;; ret -;; 35: trap ;; 38: trap +;; 3b: trap From 3b53eae848eb18a8d2bfd341ff862a27cb4db126 Mon Sep 17 00:00:00 2001 From: Demetrios Agourakis Date: Thu, 1 Oct 2026 20:24:54 -0300 Subject: [PATCH 6/6] Cranelift: gate sharing-aware egraph cost on saturation --- cranelift/codegen/src/egraph/cost.rs | 9 ++ cranelift/codegen/src/egraph/elaborate.rs | 145 ++++++++++++++++++---- tests/disas/array-fill-i16.wat | 44 ++++--- tests/disas/pulley-be-inline-copy.wat | 15 ++- 4 files changed, 162 insertions(+), 51 deletions(-) diff --git a/cranelift/codegen/src/egraph/cost.rs b/cranelift/codegen/src/egraph/cost.rs index 8c5c4ab11188..8deef914222e 100644 --- a/cranelift/codegen/src/egraph/cost.rs +++ b/cranelift/codegen/src/egraph/cost.rs @@ -194,6 +194,15 @@ impl Cost { } } + /// Compute the cost of the operation and its given operands. + /// + /// Caller is responsible for checking that the opcode came from an instruction + /// that satisfies `inst_predicates::is_pure_for_egraph()`. + pub(crate) fn of_pure_op(op: Opcode, operand_costs: impl IntoIterator) -> Self { + let c = Self::of_opcode(op) + operand_costs.into_iter().sum(); + Cost::new(c.cost()) + } + /// Compute the cost of an operation in the side-effectful skeleton. pub(crate) fn of_skeleton_op(op: Opcode, arity: usize) -> Self { Cost::of_opcode(op) + Cost::new(u32::try_from(arity).unwrap()) diff --git a/cranelift/codegen/src/egraph/elaborate.rs b/cranelift/codegen/src/egraph/elaborate.rs index 3fbe33df6886..f73ccecac38e 100644 --- a/cranelift/codegen/src/egraph/elaborate.rs +++ b/cranelift/codegen/src/egraph/elaborate.rs @@ -2,7 +2,7 @@ //! in CFG nodes. use super::Stats; -use super::cost::ExprCost; +use super::cost::{Cost, ExprCost}; use crate::ctxhash::NullCtx; use crate::dominator_tree::DominatorTree; use crate::hash_map::Entry as HashEntry; @@ -100,7 +100,7 @@ pub(crate) struct Elaborator<'a> { const NOT_ON_LOOP_STACK: u32 = u32::MAX; #[derive(Clone, Copy, Debug, PartialEq, Eq)] -struct BestEntry(ExprCost, Value); +struct BestEntry(Cost, Value); impl PartialOrd for BestEntry { fn partial_cmp(&self, other: &Self) -> Option { @@ -121,6 +121,28 @@ impl Ord for BestEntry { } } +#[derive(Clone, Copy, Debug, PartialEq, Eq)] +struct ExprBestEntry(ExprCost, Value); + +impl PartialOrd for ExprBestEntry { + fn partial_cmp(&self, other: &Self) -> Option { + Some(self.cmp(other)) + } +} + +impl Ord for ExprBestEntry { + #[inline] + fn cmp(&self, other: &Self) -> core::cmp::Ordering { + self.0.cmp(&other.0).then_with(|| { + // Note that this comparison is reversed. When costs are equal, + // prefer the value with the bigger index. This is a heuristic that + // prefers results of rewrites to the original value, since we + // expect that our rewrites are generally improvements. + self.1.cmp(&other.1).reverse() + }) + } +} + #[derive(Clone, Copy, Debug)] struct ElaboratedValue { in_block: Block, @@ -183,7 +205,7 @@ impl<'a> Elaborator<'a> { ) -> Self { let num_values = func.dfg.num_values(); let mut value_to_best_value = - SecondaryMap::with_default(BestEntry(ExprCost::infinity(), Value::reserved_value())); + SecondaryMap::with_default(BestEntry(Cost::infinity(), Value::reserved_value())); value_to_best_value.resize(num_values); Self { func, @@ -345,8 +367,6 @@ impl<'a> Elaborator<'a> { fn compute_best_values(&mut self) { let sorted_values = self.topo_sorted_values(); - let best = &mut self.value_to_best_value; - // We can't make random decisions inside the fixpoint loop below because // that could cause values to change on every iteration of the loop, // which would make the loop never terminate. So in chaos testing @@ -364,6 +384,21 @@ impl<'a> Elaborator<'a> { } ); + let saw_infinity = self.compute_best_values_with_scalar_costs(&sorted_values, use_worst); + + if saw_infinity { + self.compute_best_values_with_expr_costs(&sorted_values, use_worst); + } + } + + fn compute_best_values_with_scalar_costs( + &mut self, + sorted_values: &[Value], + use_worst: bool, + ) -> bool { + let best = &mut self.value_to_best_value; + let mut saw_infinity = false; + // Because the values are topologically sorted, we know that we will see // defs before uses, so an instruction's operands' costs will already be // computed by the time we are computing the cost for the current value @@ -392,7 +427,7 @@ impl<'a> Elaborator<'a> { } ValueDef::Param(_, _) => { - best[value] = BestEntry(ExprCost::zero(), value); + best[value] = BestEntry(Cost::zero(), value); } // If the Inst is inserted into the layout (which is, @@ -401,20 +436,22 @@ impl<'a> Elaborator<'a> { // cost. ValueDef::Result(inst, _) => { if let Some(_) = self.func.layout.inst_block(inst) { - best[value] = BestEntry(ExprCost::zero(), value); + best[value] = BestEntry(Cost::zero(), value); } else { let inst_data = &self.func.dfg.insts[inst]; // N.B.: at this point we know that the opcode is // pure, so `pure_op_cost`'s precondition is // satisfied. - let cost = ExprCost::of_pure_op( - inst, + let cost = Cost::of_pure_op( inst_data.opcode(), self.func.dfg.inst_values(inst).map(|value| { debug_assert!(!best[value].1.is_reserved_value()); best[value].0 }), ); + if cost == Cost::infinity() { + saw_infinity = true; + } best[value] = BestEntry(cost, value); trace!(" -> cost of value {} = {:?}", value, cost); } @@ -422,24 +459,86 @@ impl<'a> Elaborator<'a> { }; // You might be expecting an assert that the best cost we just - // computed is not infinity, however infinite cost *can* still - // happen in practice. The expression cost tracks an approximate - // instruction footprint to avoid charging the same already-covered - // operand twice in common shared-DAG shapes. That keeps cases such - // as a chain of `iadd x, x` from growing exponentially, but it is - // still just a bounded heuristic over a 32-bit cost: + // computed is not infinity, however infinite cost *can* happen in + // practice. First, note that our cost function doesn't know about + // any shared structure in the dataflow graph, it only sums operand + // costs. (And trying to avoid that by deduping a single operation's + // operands is a losing game because you can always just add one + // indirection and go from `add(x, x)` to `add(foo(x), bar(x))` to + // hide the shared structure.) Given that blindness to sharing, we + // can make cost grow exponentially with a linear sequence of + // operations: // // v0 = iconst.i32 1 ;; cost = 1 - // v1 = iadd v0, v0 ;; second v0 already covered - // v2 = iadd v1, v1 ;; second v1 already covered + // v1 = iadd v0, v0 ;; cost = 3 + 1 + 1 + // v2 = iadd v1, v1 ;; cost = 3 + 5 + 5 + // v3 = iadd v2, v2 ;; cost = 3 + 13 + 13 + // v4 = iadd v3, v3 ;; cost = 3 + 29 + 29 + // v5 = iadd v4, v4 ;; cost = 3 + 61 + 61 + // v6 = iadd v5, v5 ;; cost = 3 + 125 + 125 // ;; etc... // - // If a cost does saturate to infinity, it doesn't matter which - // equally infinite e-node we pick. As long as invariant (2) for - // optimization rules is upheld by our rule set (see - // `cranelift/codegen/src/opts/README.md`) it is safe to choose - // *any* e-node in the e-class. At worst we will produce suboptimal - // code, but never an incorrectness. + // If this happens, the caller will recompute the same values with + // a heavier sharing-aware fallback cost. Otherwise, the common path + // remains the simple scalar cost. + } + + saw_infinity + } + + fn compute_best_values_with_expr_costs(&mut self, sorted_values: &[Value], use_worst: bool) { + let mut expr_best = SecondaryMap::with_default(ExprBestEntry( + ExprCost::infinity(), + Value::reserved_value(), + )); + expr_best.resize(self.func.dfg.num_values()); + + for value in sorted_values.iter().copied() { + let def = self.func.dfg.value_def(value); + trace!( + "recomputing sharing-aware best for value {:?} def {:?}", + value, def + ); + + match def { + ValueDef::Union(x, y) => { + debug_assert!(!expr_best[x].1.is_reserved_value()); + debug_assert!(!expr_best[y].1.is_reserved_value()); + expr_best[value] = if use_worst { + core::cmp::max(expr_best[x], expr_best[y]) + } else { + core::cmp::min(expr_best[x], expr_best[y]) + }; + trace!( + " -> sharing-aware best of union({:?}, {:?}) = {:?}", + expr_best[x], expr_best[y], expr_best[value] + ); + } + + ValueDef::Param(_, _) => { + expr_best[value] = ExprBestEntry(ExprCost::zero(), value); + } + + ValueDef::Result(inst, _) => { + if let Some(_) = self.func.layout.inst_block(inst) { + expr_best[value] = ExprBestEntry(ExprCost::zero(), value); + } else { + let inst_data = &self.func.dfg.insts[inst]; + let cost = ExprCost::of_pure_op( + inst, + inst_data.opcode(), + self.func.dfg.inst_values(inst).map(|value| { + debug_assert!(!expr_best[value].1.is_reserved_value()); + expr_best[value].0 + }), + ); + expr_best[value] = ExprBestEntry(cost, value); + trace!(" -> sharing-aware cost of value {} = {:?}", value, cost); + } + } + } + + self.value_to_best_value[value].1 = expr_best[value].1; } } diff --git a/tests/disas/array-fill-i16.wat b/tests/disas/array-fill-i16.wat index 122c527da871..98840708b3f8 100644 --- a/tests/disas/array-fill-i16.wat +++ b/tests/disas/array-fill-i16.wat @@ -51,17 +51,18 @@ ;; @0031 v36 = load.i64 notrap aligned region3 v7+40 ;; @0031 v24 = iconst.i64 20 ;; @0031 v25 = iadd v9, v24 ; v24 = 20 -;; v49 = iadd v14, v14 -;; @0031 v29 = iadd v25, v49 -;; v53 = iadd v15, v15 -;; @0031 v38 = uadd_overflow_trap v29, v53, user2 +;; @0031 v16 = iconst.i64 1 +;; v50 = ishl v14, v16 ; v16 = 1 +;; @0031 v29 = iadd v25, v50 +;; v54 = ishl v15, v16 ; v16 = 1 +;; @0031 v38 = uadd_overflow_trap v29, v54, user2 ;; @0031 v37 = iadd v8, v36 ;; @0031 v39 = icmp ugt v38, v37 ;; @0031 trapnz v39, user2 ;; v47 = iconst.i64 0 ;; @0031 v42 = icmp eq v15, v47 ; v47 = 0 ;; @0031 v27 = iconst.i64 2 -;; @0031 v40 = iadd v29, v53 +;; @0031 v40 = iadd v29, v54 ;; @0031 brif v42, block3, block2(v29) ;; ;; block2(v43: i64): @@ -109,15 +110,16 @@ ;; @003f v36 = load.i64 notrap aligned region3 v7+40 ;; @003f v24 = iconst.i64 20 ;; @003f v25 = iadd v9, v24 ; v24 = 20 -;; v43 = iadd v14, v14 -;; @003f v29 = iadd v25, v43 -;; v47 = iadd v15, v15 -;; @003f v38 = uadd_overflow_trap v29, v47, user2 +;; @003f v16 = iconst.i64 1 +;; v44 = ishl v14, v16 ; v16 = 1 +;; @003f v29 = iadd v25, v44 +;; v48 = ishl v15, v16 ; v16 = 1 +;; @003f v38 = uadd_overflow_trap v29, v48, user2 ;; @003f v37 = iadd v8, v36 ;; @003f v39 = icmp ugt v38, v37 ;; @003f trapnz v39, user2 ;; @003b v5 = iconst.i32 0 -;; @003f call fn0(v0, v29, v5, v47) ; v5 = 0 +;; @003f call fn0(v0, v29, v5, v48) ; v5 = 0 ;; @0042 jump block1 ;; ;; block1: @@ -155,15 +157,16 @@ ;; @004d v36 = load.i64 notrap aligned region3 v7+40 ;; @004d v24 = iconst.i64 20 ;; @004d v25 = iadd v9, v24 ; v24 = 20 -;; v43 = iadd v14, v14 -;; @004d v29 = iadd v25, v43 -;; v47 = iadd v15, v15 -;; @004d v38 = uadd_overflow_trap v29, v47, user2 +;; @004d v16 = iconst.i64 1 +;; v44 = ishl v14, v16 ; v16 = 1 +;; @004d v29 = iadd v25, v44 +;; v48 = ishl v15, v16 ; v16 = 1 +;; @004d v38 = uadd_overflow_trap v29, v48, user2 ;; @004d v37 = iadd v8, v36 ;; @004d v39 = icmp ugt v38, v37 ;; @004d trapnz v39, user2 ;; @004d v40 = iconst.i32 255 -;; @004d call fn0(v0, v29, v40, v47) ; v40 = 255 +;; @004d call fn0(v0, v29, v40, v48) ; v40 = 255 ;; @0050 jump block1 ;; ;; block1: @@ -200,10 +203,11 @@ ;; @005d v36 = load.i64 notrap aligned region3 v7+40 ;; @005d v24 = iconst.i64 20 ;; @005d v25 = iadd v9, v24 ; v24 = 20 -;; v49 = iadd v14, v14 -;; @005d v29 = iadd v25, v49 -;; v53 = iadd v15, v15 -;; @005d v38 = uadd_overflow_trap v29, v53, user2 +;; @005d v16 = iconst.i64 1 +;; v50 = ishl v14, v16 ; v16 = 1 +;; @005d v29 = iadd v25, v50 +;; v54 = ishl v15, v16 ; v16 = 1 +;; @005d v38 = uadd_overflow_trap v29, v54, user2 ;; @005d v37 = iadd v8, v36 ;; @005d v39 = icmp ugt v38, v37 ;; @005d trapnz v39, user2 @@ -211,7 +215,7 @@ ;; @005d v42 = icmp eq v15, v47 ; v47 = 0 ;; @0057 v5 = iconst.i32 0xdead ;; @005d v27 = iconst.i64 2 -;; @005d v40 = iadd v29, v53 +;; @005d v40 = iadd v29, v54 ;; @005d brif v42, block3, block2(v29) ;; ;; block2(v43: i64): diff --git a/tests/disas/pulley-be-inline-copy.wat b/tests/disas/pulley-be-inline-copy.wat index 39e97620eb82..1e0d8627dc02 100644 --- a/tests/disas/pulley-be-inline-copy.wat +++ b/tests/disas/pulley-be-inline-copy.wat @@ -17,14 +17,13 @@ ) ;; wasm[0]::function[0]: ;; push_frame -;; xload64be_o32 x5, x0, 64 -;; br_if_xult64_u8 x5, 16, 0x2e // target = 0x38 -;; 11: xconst8 x6, 32 -;; br_if_xult64 x5, x6, 0x27 // target = 0x3b -;; 1b: xload64be_o32 x7, x0, 56 -;; vload128le_o32 v7, x7, 16 -;; vstore128le_o32 x7, 0, v7 +;; xload64be_o32 x4, x0, 64 +;; br_if_xult64_u8 x4, 16, 0x2b // target = 0x35 +;; br_if_xult64_u8 x4, 32, 0x27 // target = 0x38 +;; 18: xload64be_o32 x6, x0, 56 +;; vload128le_o32 v6, x6, 16 +;; vstore128le_o32 x6, 0, v6 ;; pop_frame ;; ret +;; 35: trap ;; 38: trap -;; 3b: trap