//! Dependency-free, `no_std`, deterministic SHA-256 (FIPS 180-4). //! //! WHY A HAND-ROLLED SHA-256. This crate is the offline-testable REFERENCE model //! of the per-validator inner logic the aggregation zkVM guest runs. Keeping it at //! zero external dependencies means `cargo test` verifies the RFC 8391 XMSS core //! against the official known-answer vector with no network and no toolchain //! surprises, on any machine. //! //! WHAT THE REAL SP1 GUEST USES INSTEAD. Inside the SP1 zkVM the production guest //! would replace this module with the SP1-patched `sha2` crate (the accelerated //! SHA-256 precompile), exactly as `../zk-light-client/guest/Cargo.toml` patches //! `sha2` and `k256`. The BYTES are identical either way, so this module is a //! faithful stand-in for the circuit's hash gate; only the in-circuit cost differs. //! This is single-block-friendly, allocation-free, and float-free by construction. const H0: [u32; 8] = [ 0x6a09e667, 0xbb67ae85, 0x3c6ef372, 0xa54ff53a, 0x510e527f, 0x9b05688c, 0x1f83d9ab, 0x5be0cd19, ]; const K: [u32; 64] = [ 0x428a2f98, 0x71374491, 0xb5c0fbcf, 0xe9b5dba5, 0x3956c25b, 0x59f111f1, 0x923f82a4, 0xab1c5ed5, 0xd807aa98, 0x12835b01, 0x243185be, 0x550c7dc3, 0x72be5d74, 0x80deb1fe, 0x9bdc06a7, 0xc19bf174, 0xe49b69c1, 0xefbe4786, 0x0fc19dc6, 0x240ca1cc, 0x2de92c6f, 0x4a7484aa, 0x5cb0a9dc, 0x76f988da, 0x983e5152, 0xa831c66d, 0xb00327c8, 0xbf597fc7, 0xc6e00bf3, 0xd5a79147, 0x06ca6351, 0x14292967, 0x27b70a85, 0x2e1b2138, 0x4d2c6dfc, 0x53380d13, 0x650a7354, 0x766a0abb, 0x81c2c92e, 0x92722c85, 0xa2bfe8a1, 0xa81a664b, 0xc24b8b70, 0xc76c51a3, 0xd192e819, 0xd6990624, 0xf40e3585, 0x106aa070, 0x19a4c116, 0x1e376c08, 0x2748774c, 0x34b0bcb5, 0x391c0cb3, 0x4ed8aa4a, 0x5b9cca4f, 0x682e6ff3, 0x748f82ee, 0x78a5636f, 0x84c87814, 0x8cc70208, 0x90befffa, 0xa4506ceb, 0xbef9a3f7, 0xc67178f2, ]; /// Incremental SHA-256 state. `no_std`, no allocation, deterministic. pub struct Sha256 { h: [u32; 8], buf: [u8; 64], buf_len: usize, len_bits: u64, } impl Sha256 { #[inline] pub fn new() -> Self { Sha256 { h: H0, buf: [0u8; 64], buf_len: 0, len_bits: 0 } } pub fn update(&mut self, mut data: &[u8]) { self.len_bits = self.len_bits.wrapping_add((data.len() as u64) * 8); // Fill any partial buffer first. if self.buf_len > 0 { let take = core::cmp::min(64 - self.buf_len, data.len()); self.buf[self.buf_len..self.buf_len + take].copy_from_slice(&data[..take]); self.buf_len += take; data = &data[take..]; if self.buf_len == 64 { let block = self.buf; self.compress(&block); self.buf_len = 0; } } // Compress full 64-byte blocks straight from the input. while data.len() >= 64 { let mut block = [0u8; 64]; block.copy_from_slice(&data[..64]); self.compress(&block); data = &data[64..]; } // Stash the remainder. if !data.is_empty() { self.buf[..data.len()].copy_from_slice(data); self.buf_len = data.len(); } } pub fn finalize(mut self) -> [u8; 32] { let len_bits = self.len_bits; // Padding: 0x80, then zeros, then the 64-bit big-endian bit length. let mut pad = [0u8; 72]; pad[0] = 0x80; // total padded region so that (buf_len + 1 + zeros + 8) % 64 == 0 let pad_zeros = if self.buf_len < 56 { 56 - self.buf_len } else { 120 - self.buf_len }; let total = pad_zeros + 8; // includes the 0x80 byte within pad_zeros count below // Note: pad_zeros here already counts the 0x80 byte position; write length at the end. pad[pad_zeros..pad_zeros + 8].copy_from_slice(&len_bits.to_be_bytes()); self.update_no_len(&pad[..total]); let mut out = [0u8; 32]; for (i, word) in self.h.iter().enumerate() { out[i * 4..i * 4 + 4].copy_from_slice(&word.to_be_bytes()); } out } /// Same as `update` but does NOT advance the tracked message length (used only /// to feed the precomputed padding block in `finalize`). fn update_no_len(&mut self, mut data: &[u8]) { if self.buf_len > 0 { let take = core::cmp::min(64 - self.buf_len, data.len()); self.buf[self.buf_len..self.buf_len + take].copy_from_slice(&data[..take]); self.buf_len += take; data = &data[take..]; if self.buf_len == 64 { let block = self.buf; self.compress(&block); self.buf_len = 0; } } while data.len() >= 64 { let mut block = [0u8; 64]; block.copy_from_slice(&data[..64]); self.compress(&block); data = &data[64..]; } if !data.is_empty() { self.buf[..data.len()].copy_from_slice(data); self.buf_len = data.len(); } } fn compress(&mut self, block: &[u8; 64]) { let mut w = [0u32; 64]; for i in 0..16 { w[i] = u32::from_be_bytes([ block[i * 4], block[i * 4 + 1], block[i * 4 + 2], block[i * 4 + 3], ]); } for i in 16..64 { let s0 = w[i - 15].rotate_right(7) ^ w[i - 15].rotate_right(18) ^ (w[i - 15] >> 3); let s1 = w[i - 2].rotate_right(17) ^ w[i - 2].rotate_right(19) ^ (w[i - 2] >> 10); w[i] = w[i - 16] .wrapping_add(s0) .wrapping_add(w[i - 7]) .wrapping_add(s1); } let mut a = self.h[0]; let mut b = self.h[1]; let mut c = self.h[2]; let mut d = self.h[3]; let mut e = self.h[4]; let mut f = self.h[5]; let mut g = self.h[6]; let mut hh = self.h[7]; for i in 0..64 { let s1 = e.rotate_right(6) ^ e.rotate_right(11) ^ e.rotate_right(25); let ch = (e & f) ^ ((!e) & g); let t1 = hh .wrapping_add(s1) .wrapping_add(ch) .wrapping_add(K[i]) .wrapping_add(w[i]); let s0 = a.rotate_right(2) ^ a.rotate_right(13) ^ a.rotate_right(22); let maj = (a & b) ^ (a & c) ^ (b & c); let t2 = s0.wrapping_add(maj); hh = g; g = f; f = e; e = d.wrapping_add(t1); d = c; c = b; b = a; a = t1.wrapping_add(t2); } self.h[0] = self.h[0].wrapping_add(a); self.h[1] = self.h[1].wrapping_add(b); self.h[2] = self.h[2].wrapping_add(c); self.h[3] = self.h[3].wrapping_add(d); self.h[4] = self.h[4].wrapping_add(e); self.h[5] = self.h[5].wrapping_add(f); self.h[6] = self.h[6].wrapping_add(g); self.h[7] = self.h[7].wrapping_add(hh); } } impl Default for Sha256 { fn default() -> Self { Self::new() } } /// One-shot SHA-256 over `data`. #[inline] pub fn sha256(data: &[u8]) -> [u8; 32] { let mut h = Sha256::new(); h.update(data); h.finalize() }