Project homepage Mailing List  Warmcat.com  API Docs  Github Mirror 
    npro  
 Modern all-safe Rust Network Protocol library supporting h1, h2, h3, ws, wt sans-IO and with socket IO + tls
git clone https://npro.rs/repo/npro
 
root / crates / npro-test / h1 / requests / uri-escape-nul.http
Author[]Andy Green <andy@warmcat.com> 2026-10-03 09:24 UTC
Committer[]Andy Green <andy@warmcat.com> 2026-10-05 06:32 UTC
Tree0e2b5f26b68cf9d015ad304a4be0c451014bd40e   Raw Patch
 
npro-core: SHA-1 and base64 for the ws handshake
npro-core: SHA-1 and base64 for the ws handshake

The ws handshake needs both, to send Sec-WebSocket-Key and to make or
check Sec-WebSocket-Accept, base64(SHA-1(key + GUID)) (RFC 6455 4.2.2).
C has its own of each in lib/misc.  So does npro, rather than taking two
dependencies for one handshake: together they are about 200 lines,
checked against the RFCs' vectors.

sha1:
- Sha1::new / update / finish, and Sha1::digest in one call;
- the schedule is kept as a rolling window of 16 words, so nothing indexes
  past a fixed array;
- it is documented as being here for the handshake only: SHA-1 is broken
  for anything that needs collision resistance.

Tests: the RFC 3174 vectors (including a million 'a'), padding either
side of every block boundary, and the same digest for every split of a
300-byte message.

base64 is encoding only, standard alphabet with padding (RFC 4648 4).
Nothing in the handshake decodes either value.  It writes into the
caller's buffer, refusing a short one untouched, as every composer does.

Tests:
- the RFC 4648 vectors and every sextet value;
- RFC 6455 1.3's accept value, through SHA-1;
- with the replay feature, the key C's ws-client transcript sends, drawn
  from lws' random seeded with 1.

The crate builds with std only under test, so tests can use Vec and
format!; the no_std gate still builds the real crate for a target
without std.

Co-Authored-By: Claude Opus 5.5 (1M context) <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_019kg5Eemy68ZaqDBcUJQG6J
diff --git a/crates/npro-core/src/base64.rs b/crates/npro-core/src/base64.rs new file mode 100644 index 0000000..e111aa9 --- /dev/null +++ b/crates/npro-core/src/base64.rs @@ -0,0 +1,136 @@ +//! Base64 encoding (RFC 4648 4, the standard alphabet with padding), for +//! the ws handshake's `Sec-WebSocket-Key` and `Sec-WebSocket-Accept`. +//! +//! Only encoding: the handshake never needs to decode either value. The +//! output goes into the caller's buffer, as everything the protocols +//! compose does. + +/// The standard alphabet. +const ALPHABET: &[u8; 64] = b"ABCDEFGHIJKLMNOPQRSTUVWXYZabcdefghijklmnopqrstuvwxyz0123456789+/"; + +/// The output buffer cannot hold the encoding. +#[derive(Clone, Copy, Debug, PartialEq, Eq)] +pub struct BufferTooSmall; + +impl core::fmt::Display for BufferTooSmall { + fn fmt(&self, f: &mut core::fmt::Formatter<'_>) -> core::fmt::Result { + f.write_str("buffer too small for the base64 encoding") + } +} + +impl core::error::Error for BufferTooSmall {} + +/// The length of the encoding of `n` bytes, or `None` if it would not fit +/// a `usize`. +#[must_use] +pub const fn encoded_len(n: usize) -> Option<usize> { + n.div_ceil(3).checked_mul(4) +} + +/// Encodes `src` into the start of `dst`, returning how many bytes of `dst` +/// it used. +/// +/// ``` +/// use npro_core::base64; +/// +/// let mut out = [0; 8]; +/// let n = base64::encode(b"foob", &mut out)?; +/// assert_eq!(&out[..n], b"Zm9vYg=="); +/// # Ok::<(), base64::BufferTooSmall>(()) +/// ``` +/// +/// # Errors +/// +/// [`BufferTooSmall`] if `dst` is shorter than [`encoded_len`] of `src`; +/// nothing is written then. +pub fn encode(src: &[u8], dst: &mut [u8]) -> Result<usize, BufferTooSmall> { + let len = encoded_len(src.len()).ok_or(BufferTooSmall)?; + let out = dst.get_mut(..len).ok_or(BufferTooSmall)?; + + for (o, i) in out.chunks_exact_mut(4).zip(src.chunks(3)) { + let (b0, b1, b2, have) = match *i { + [a, b, c] => (a, b, c, 4), + [a, b] => (a, b, 0, 3), + [a] => (a, 0, 0, 2), + _ => (0, 0, 0, 0), + }; + let sextets = [b0 >> 2, (b0 << 4 | b1 >> 4), (b1 << 2 | b2 >> 6), b2]; + for (n, (d, s)) in o.iter_mut().zip(sextets).enumerate() { + *d = if n < have { symbol(s) } else { b'=' }; + } + } + + Ok(len) +} + +/// The alphabet's symbol for the low six bits of `v`. +fn symbol(v: u8) -> u8 { + // masked to six bits, the index is always inside the 64-symbol alphabet + ALPHABET.get(usize::from(v & 0x3f)).copied().unwrap_or(b'=') +} + +#[cfg(test)] +mod tests { + use super::*; + + fn enc(src: &[u8]) -> String { + let mut out = vec![0; encoded_len(src.len()).unwrap()]; + let n = encode(src, &mut out).unwrap(); + assert_eq!(n, out.len()); + String::from_utf8(out).unwrap() + } + + #[test] + fn rfc_4648_vectors() { + for (src, want) in [ + ("", ""), + ("f", "Zg=="), + ("fo", "Zm8="), + ("foo", "Zm9v"), + ("foob", "Zm9vYg=="), + ("fooba", "Zm9vYmE="), + ("foobar", "Zm9vYmFy"), + ] { + assert_eq!(enc(src.as_bytes()), want); + } + } + + #[test] + fn every_symbol_and_the_high_bits() { + // 0x00..=0xff covers every sextet value in every position + let all: Vec<u8> = (0..=255).collect(); + let s = enc(&all); + for c in ALPHABET { + assert!(s.as_bytes().contains(c)); + } + assert!(s.starts_with("AAECAwQF")); + assert!(s.ends_with("+/w==")); + } + + #[test] + fn a_short_buffer_is_refused_untouched() { + let mut out = [b'x'; 7]; + assert_eq!(encode(b"foob", &mut out), Err(BufferTooSmall)); + assert_eq!(out, [b'x'; 7]); + } + + #[test] + fn the_ws_accept_of_rfc_6455() { + // RFC 6455 1.3: base64(SHA-1(key + GUID)) + let mut h = crate::sha1::Sha1::new(); + h.update(b"dGhlIHNhbXBsZSBub25jZQ=="); + h.update(b"258EAFA5-E914-47DA-95CA-C5AB0DC85B11"); + assert_eq!(enc(&h.finish()), "s3pPLMBiTxaQ9kYGzzhZRbK+xOo="); + } + + #[cfg(feature = "replay")] + #[test] + fn the_ws_client_transcripts_key() { + // C's ws-client transcript: the key its client sends, drawn as the + // first 16 bytes of lws' random seeded with 1 + use crate::random::{Random, SeededRandom}; + let mut key = [0; 16]; + SeededRandom::new(1).fill(&mut key).unwrap(); + assert_eq!(enc(&key), "OvomtQpKCWUnZW7tMR6Gqw=="); + } +} diff --git a/crates/npro-core/src/lib.rs b/crates/npro-core/src/lib.rs index d21ea3c..f00bff3 100644 --- a/crates/npro-core/src/lib.rs +++ b/crates/npro-core/src/lib.rs @@ -5,7 +5,9 @@ //! substrate the protocols are built from. Nothing here owns a socket, a //! thread or a clock; the IO side, or a test, supplies all of them. -#![no_std] +#![cfg_attr(not(test), no_std)] #![forbid(unsafe_code)] +pub mod base64; pub mod random; +pub mod sha1; diff --git a/crates/npro-core/src/sha1.rs b/crates/npro-core/src/sha1.rs new file mode 100644 index 0000000..76913f0 --- /dev/null +++ b/crates/npro-core/src/sha1.rs @@ -0,0 +1,229 @@ +//! SHA-1, for the one thing the protocols need it for: the ws handshake's +//! `Sec-WebSocket-Accept` (RFC 6455 4.2.2), which is fixed by the RFC. +//! +//! SHA-1 is broken for collision resistance and must not be used for +//! anything that needs it; the ws handshake does not. C lws has its own in +//! `lib/misc/sha-1.c`, and so does npro rather than taking a dependency for +//! one handshake hash. + +/// The size of a SHA-1 digest, in bytes. +pub const DIGEST_LEN: usize = 20; + +/// The size of a SHA-1 block, in bytes. +const BLOCK: usize = 64; + +/// Where the 64-bit message length goes in the last block. +const LEN_AT: usize = BLOCK - 8; + +/// The initial state (RFC 3174 6.1). +const H0: [u32; 5] = [ + 0x6745_2301, + 0xefcd_ab89, + 0x98ba_dcfe, + 0x1032_5476, + 0xc3d2_e1f0, +]; + +/// A SHA-1 hash in progress. +/// +/// ``` +/// use npro_core::sha1::Sha1; +/// +/// let mut h = Sha1::new(); +/// h.update(b"ab"); +/// h.update(b"c"); +/// assert_eq!(h.finish(), Sha1::digest(b"abc")); +/// ``` +#[derive(Clone, Debug)] +pub struct Sha1 { + state: [u32; 5], + /// Bytes of the current block taken so far. + block: [u8; BLOCK], + /// How much of `block` is filled. + fill: usize, + /// Length of the message so far, in bytes, modulo 2^64. + len: u64, +} + +impl Default for Sha1 { + fn default() -> Self { + Self::new() + } +} + +impl Sha1 { + /// A hash of nothing yet. + #[must_use] + pub const fn new() -> Self { + Self { + state: H0, + block: [0; BLOCK], + fill: 0, + len: 0, + } + } + + /// The digest of `data`, in one call. + /// + /// ``` + /// use npro_core::sha1::Sha1; + /// + /// assert_eq!( + /// Sha1::digest(b"abc")[..4], + /// [0xa9, 0x99, 0x3e, 0x36] + /// ); + /// ``` + #[must_use] + pub fn digest(data: &[u8]) -> [u8; DIGEST_LEN] { + let mut h = Self::new(); + h.update(data); + h.finish() + } + + /// Takes more of the message. + pub fn update(&mut self, data: &[u8]) { + for &b in data { + self.byte(b); + } + // the length is the message's modulo 2^64 bits (RFC 3174 4); a + // usize that does not fit 64 bits is no platform npro builds for + let n = u64::try_from(data.len()).unwrap_or(u64::MAX); + self.len = self.len.wrapping_add(n); + } + + /// Pads the message and gives its digest. + #[must_use] + pub fn finish(mut self) -> [u8; DIGEST_LEN] { + let bits = self.len.wrapping_mul(8); + + self.byte(0x80); + while self.fill != LEN_AT { + self.byte(0); + } + for b in bits.to_be_bytes() { + self.byte(b); + } + + let mut out = [0; DIGEST_LEN]; + for (o, w) in out.chunks_exact_mut(4).zip(self.state) { + for (d, s) in o.iter_mut().zip(w.to_be_bytes()) { + *d = s; + } + } + out + } + + /// Takes one byte into the block, compressing it when it is full. + fn byte(&mut self, b: u8) { + if let Some(slot) = self.block.get_mut(self.fill) { + *slot = b; + } + self.fill = self.fill.saturating_add(1); + if self.fill == BLOCK { + self.compress(); + self.fill = 0; + } + } + + /// The compression function over the full block (RFC 3174 6.1). + fn compress(&mut self) { + // the message schedule, kept as the last 16 words: each round's + // word is w[0], and the next is made from w[13], w[8], w[2], w[0] + let mut w = [0u32; 16]; + for (d, c) in w.iter_mut().zip(self.block.chunks_exact(4)) { + if let &[a, b, c, e] = c { + *d = u32::from_be_bytes([a, b, c, e]); + } + } + + let [mut a, mut b, mut c, mut d, mut e] = self.state; + + for t in 0..80 { + let (f, k) = match t { + 0..20 => ((b & c) | (!b & d), 0x5a82_7999), + 20..40 => (b ^ c ^ d, 0x6ed9_eba1), + 40..60 => ((b & c) | (b & d) | (c & d), 0x8f1b_bcdc), + _ => (b ^ c ^ d, 0xca62_c1d6), + }; + let temp = a + .rotate_left(5) + .wrapping_add(f) + .wrapping_add(e) + .wrapping_add(w[0]) + .wrapping_add(k); + e = d; + d = c; + c = b.rotate_left(30); + b = a; + a = temp; + + let next = (w[13] ^ w[8] ^ w[2] ^ w[0]).rotate_left(1); + w.rotate_left(1); + w[15] = next; + } + + for (s, v) in self.state.iter_mut().zip([a, b, c, d, e]) { + *s = s.wrapping_add(v); + } + } +} + +#[cfg(test)] +mod tests { + use super::*; + + fn hex(d: [u8; DIGEST_LEN]) -> String { + d.iter().map(|b| format!("{b:02x}")).collect() + } + + #[test] + fn rfc_3174_vectors() { + assert_eq!( + hex(Sha1::digest(b"abc")), + "a9993e364706816aba3e25717850c26c9cd0d89d" + ); + assert_eq!( + hex(Sha1::digest( + b"abcdbcdecdefdefgefghfghighijhijkijkljklmklmnlmnomnopnopq" + )), + "84983e441c3bd26ebaae4aa1f95129e5e54670f1" + ); + assert_eq!( + hex(Sha1::digest(&vec![b'a'; 1_000_000])), + "34aa973cd4c4daa4f61eeb2bdbad27316534016f" + ); + assert_eq!( + hex(Sha1::digest(b"")), + "da39a3ee5e6b4b0d3255bfef95601890afd80709" + ); + } + + #[test] + fn padding_at_every_block_boundary() { + // lengths around 55 / 56 / 64 bytes put the 0x80 and the length in + // the last block or the next one: digests of 'a' repeated n times, + // as Python's hashlib.sha1 gives them + for (n, want) in [ + (55, "c1c8bbdc22796e28c0e15163d20899b65621d65a"), + (56, "c2db330f6083854c99d4b5bfb6e8f29f201be699"), + (63, "03f09f5b158a7a8cdad920bddc29b81c18a551f5"), + (64, "0098ba824b5c16427bd7a1122a5a442a25ec644d"), + (65, "11655326c708d70319be2610e8a57d9a5b959d3b"), + ] { + assert_eq!(hex(Sha1::digest(&vec![b'a'; n])), want, "{n} bytes"); + } + } + + #[test] + fn the_same_digest_however_the_message_is_split() { + let msg: Vec<u8> = (0..=255u8).cycle().take(300).collect(); + let whole = Sha1::digest(&msg); + for cut in 0..=msg.len() { + let (x, y) = msg.split_at(cut); + let mut h = Sha1::new(); + h.update(x); + h.update(y); + assert_eq!(h.finish(), whole, "split at {cut}"); + } + } +}
Page fetched 0s ago, creation time: 2ms (vhost etag hits: 0%, cache hits: 0%)