Merge branch 'maint' (bloom single-digest double-hashing)

Forward-merges the bloom SHA-256-once fix from maint. The bloom filter was
relocated to proto/bloom/ by the sans-IO refactor, so the fix applied cleanly
to proto/bloom/core.rs; the behavior-neutral test was re-homed into
proto/bloom/tests/core.rs (maint carried it in the pre-split bloom/tests.rs).
This commit is contained in:
Johnathan Corgan
2026-07-11 01:32:48 +00:00
2 changed files with 96 additions and 9 deletions
+19 -9
View File
@@ -67,16 +67,18 @@ impl BloomFilter {
/// Insert a NodeAddr into the filter. /// Insert a NodeAddr into the filter.
pub fn insert(&mut self, node_addr: &NodeAddr) { pub fn insert(&mut self, node_addr: &NodeAddr) {
let (h1, h2) = Self::base_hashes(node_addr.as_bytes());
for i in 0..self.hash_count { for i in 0..self.hash_count {
let bit_index = self.hash(node_addr.as_bytes(), i); let bit_index = self.bit_index(h1, h2, i);
self.set_bit(bit_index); self.set_bit(bit_index);
} }
} }
/// Insert raw bytes into the filter. /// Insert raw bytes into the filter.
pub fn insert_bytes(&mut self, data: &[u8]) { pub fn insert_bytes(&mut self, data: &[u8]) {
let (h1, h2) = Self::base_hashes(data);
for i in 0..self.hash_count { for i in 0..self.hash_count {
let bit_index = self.hash(data, i); let bit_index = self.bit_index(h1, h2, i);
self.set_bit(bit_index); self.set_bit(bit_index);
} }
} }
@@ -91,8 +93,9 @@ impl BloomFilter {
/// Check if the filter might contain raw bytes. /// Check if the filter might contain raw bytes.
pub fn contains_bytes(&self, data: &[u8]) -> bool { pub fn contains_bytes(&self, data: &[u8]) -> bool {
let (h1, h2) = Self::base_hashes(data);
for i in 0..self.hash_count { for i in 0..self.hash_count {
let bit_index = self.hash(data, i); let bit_index = self.bit_index(h1, h2, i);
if !self.get_bit(bit_index) { if !self.get_bit(bit_index) {
return false; return false;
} }
@@ -198,21 +201,28 @@ impl BloomFilter {
self.hash_count self.hash_count
} }
/// Compute a hash index for the given data and hash function number. /// Compute the two base hashes for `data` with a single SHA-256 digest.
/// ///
/// Uses double hashing: h(x,i) = (h1(x) + i*h2(x)) mod m /// Double hashing derives the k hash functions from two base hashes:
fn hash(&self, data: &[u8], k: u8) -> usize { /// h(x,i) = (h1(x) + i*h2(x)) mod m. Computing the digest once here and
// Use first 16 bytes of SHA-256 for h1 and h2 /// reusing `(h1, h2)` across all k functions avoids re-hashing per k.
fn base_hashes(data: &[u8]) -> (u64, u64) {
// Use first 16 bytes of SHA-256 for h1 and h2.
use sha2::{Digest, Sha256}; use sha2::{Digest, Sha256};
let mut hasher = Sha256::new(); let mut hasher = Sha256::new();
hasher.update(data); hasher.update(data);
let hash = hasher.finalize(); let hash = hasher.finalize();
// h1 from first 8 bytes // h1 from first 8 bytes, h2 from next 8 bytes (little-endian).
let h1 = u64::from_le_bytes(hash[0..8].try_into().unwrap()); let h1 = u64::from_le_bytes(hash[0..8].try_into().unwrap());
// h2 from next 8 bytes
let h2 = u64::from_le_bytes(hash[8..16].try_into().unwrap()); let h2 = u64::from_le_bytes(hash[8..16].try_into().unwrap());
(h1, h2)
}
/// Derive the bit index for hash function `k` from the base hashes.
///
/// Uses double hashing: h(x,k) = (h1(x) + k*h2(x)) mod m.
fn bit_index(&self, h1: u64, h2: u64, k: u8) -> usize {
let combined = h1.wrapping_add((k as u64).wrapping_mul(h2)); let combined = h1.wrapping_add((k as u64).wrapping_mul(h2));
(combined as usize) % self.num_bits (combined as usize) % self.num_bits
} }
+77
View File
@@ -288,3 +288,80 @@ fn test_bloom_filter_debug_format() {
assert!(debug.contains("fill_ratio")); assert!(debug.contains("fill_ratio"));
assert!(debug.contains("est_count")); assert!(debug.contains("est_count"));
} }
#[test]
fn test_bloom_filter_bit_indices_match_double_hashing_formula() {
use sha2::{Digest, Sha256};
// Independently recompute the documented double-hashing bit indices:
// one SHA-256 digest of the input, h1 = bytes[0..8] LE, h2 = bytes[8..16]
// LE, then for k in 0..hash_count: (h1 + k*h2) mod num_bits. This pins
// bit-identical behavior regardless of the internal implementation.
fn expected_indices(data: &[u8], num_bits: usize, hash_count: u8) -> Vec<usize> {
let digest = Sha256::digest(data);
let h1 = u64::from_le_bytes(digest[0..8].try_into().unwrap());
let h2 = u64::from_le_bytes(digest[8..16].try_into().unwrap());
(0..hash_count)
.map(|k| {
let combined = h1.wrapping_add((k as u64).wrapping_mul(h2));
(combined as usize) % num_bits
})
.collect()
}
fn bit_is_set(filter: &BloomFilter, index: usize) -> bool {
let byte = filter.as_bytes()[index / 8];
(byte >> (index % 8)) & 1 == 1
}
let configs = [(1024usize, 5u8), (8192usize, 7u8)];
let inputs: [&[u8]; 4] = [b"", b"alpha", b"the quick brown fox", &[0u8, 1, 2, 3, 255]];
for (num_bits, hash_count) in configs {
for data in inputs {
let mut filter = BloomFilter::with_params(num_bits, hash_count).unwrap();
let expected = expected_indices(data, num_bits, hash_count);
filter.insert_bytes(data);
// Every expected bit is set.
for &idx in &expected {
assert!(
bit_is_set(&filter, idx),
"expected bit {} set for input {:?} (num_bits={}, k={})",
idx,
data,
num_bits,
hash_count
);
}
// No unexpected bits are set: the set-bit count never exceeds the
// number of distinct expected indices.
let distinct: alloc::collections::BTreeSet<usize> = expected.iter().copied().collect();
assert_eq!(
filter.count_ones(),
distinct.len(),
"unexpected bits set for input {:?}",
data
);
// contains reports the inserted item as present.
assert!(filter.contains_bytes(data));
}
}
// NodeAddr path uses the same formula over its byte view.
let node = make_node_addr(7);
let mut filter = BloomFilter::with_params(1024, 5).unwrap();
let expected = expected_indices(node.as_bytes(), 1024, 5);
filter.insert(&node);
for &idx in &expected {
assert!(bit_is_set(&filter, idx));
}
assert!(filter.contains(&node));
// Spot-check a definitely-absent item is reported absent.
let absent = make_node_addr(200);
assert!(!filter.contains(&absent));
}