From 0e4fbc3aeeea303fcd131c7d3eabb93a9ccd05ff Mon Sep 17 00:00:00 2001 From: Dustin Kirkland Date: Thu, 1 Oct 2026 01:17:33 -0500 Subject: [PATCH] Make quantization independent of the architecture The same image quantized to a different palette on x86_64 and aarch64 (#130). Two causes: - The histogram was drained from a HashMap in iteration order, and the cluster bucketing, the f64 weight sum, median cut tie-breaking and k-means chunking are all order-sensitive. The hasher is deterministic, but hashbrown probes 16-byte groups with SSE2 on x86_64 and 8-byte groups with NEON on aarch64, so the map iterates in a different order per architecture. Sort the entries by colour before the cluster pass, and re-insert the posterized entries in key order (last-wins on collapsing keys otherwise depended on the same order). - The NEON f_pixel::diff summed the channel terms as r + (g + b) while the x86_64, scalar and WASM paths sum (r + g) + b; float addition is not associative, and the refinement loop amplifies the difference in the metric. Add in the same order. With both, x86_64 and aarch64 produce identical pixels and palettes for every image tested at --quality 85-95 and 0-100, including one that goes through the posterize rehash. Output on one architecture changes for images where either effect mattered: same quality target and algorithm, a fixed tie order and a consistent metric. Fixes #130. --- src/hist.rs | 13 ++++++++++++- src/pal.rs | 10 ++++------ 2 files changed, 16 insertions(+), 7 deletions(-) diff --git a/src/hist.rs b/src/hist.rs index a95b1a10..1b572097 100644 --- a/src/hist.rs +++ b/src/hist.rs @@ -245,7 +245,13 @@ impl Histogram { let new_size = (self.hashmap.len() / 3).max(self.hashmap.capacity() / 5); let old_hashmap = mem::replace(&mut self.hashmap, HashMap::with_capacity_and_hasher(new_size, U32Hasher(0))); - self.hashmap.extend(old_hashmap.into_iter().map(move |(k, v)| { + // Re-insert in key order. Several old keys can collapse to one posterized + // key and the last one inserted wins, so the result must not depend on the + // old map's iteration order, which varies with the hash table's bucket + // layout (hashbrown's probe-group width differs between x86_64 and aarch64). + let mut entries: Vec<(u32, (u32, RGBA))> = old_hashmap.into_iter().collect(); + entries.sort_unstable_by_key(|&(k, _)| k); + self.hashmap.extend(entries.into_iter().map(move |(k, v)| { (k & new_posterize_mask, v) })); } @@ -301,6 +307,11 @@ impl Histogram { let weight = boost as f32; TempHistItem { color, weight, cluster_index } })); + // Everything after this point is order-sensitive (cluster bucketing by + // position, the f64 weight sum, median cut's tie-breaking, k-means' chunks), + // and a HashMap's iteration order depends on the hash table's bucket layout, + // which differs between architectures. Fix the order so the result does not. + temp.sort_unstable_by_key(|t| (t.color.r, t.color.g, t.color.b, t.color.a, t.cluster_index)); let mut clusters = [Cluster { begin: 0, end: 0 }; LIQ_MAXCLUSTER]; let mut next_begin = 0; diff --git a/src/pal.rs b/src/pal.rs index 903131aa..031b1a79 100644 --- a/src/pal.rs +++ b/src/pal.rs @@ -100,12 +100,10 @@ impl f_pixel { let mut max_r = [0.; 4]; vst1q_f32(max_r.as_mut_ptr(), max); - let mut max_gb = [0.; 4]; - vst1q_f32(max_gb.as_mut_ptr(), vpaddq_f32(max, max)); - - // add rgb, not a - - max_r[1] + max_gb[1] + // add rgb, not a. Same association as the other paths, (r + g) + b: + // a pairwise add gives r + (g + b), and float addition is not associative, + // so the distance metric (and with it the palette) differed by architecture. + max_r[1] + max_r[2] + max_r[3] } }