Project homepage Mailing List  Warmcat.com  API Docs  Github Mirror 
    npro  
 Modern all-safe Rust Network Protocol library supporting h1, h2, h3, ws, wt sans-IO and with socket IO + tls
git clone https://npro.rs/repo/npro
 
root / crates / npro-test / transcripts / h1-reqline-leading-empty.json
Author[]Andy Green <andy@warmcat.com> 2026-10-03 09:24 UTC
Committer[]Andy Green <andy@warmcat.com> 2026-10-05 06:32 UTC
Tree2241f0dba054000c5b60fb4adbfec602650a7386   Raw Patch
 
npro-core: incremental UTF-8 validation for ws text
npro-core: incremental UTF-8 validation for ws text

ws text messages and close reasons must be well-formed UTF-8 (RFC 6455
8.1).  They arrive in pieces of any size, and a bad byte fails the
connection with 1007 as soon as it is seen.  This is C's
lws_check_utf8().  C keeps where it is inside a character in one byte read
through a table; here it is an enum:
- between characters;
- inside one, with how many continuation bytes are left and the range
  the next must fall in, from RFC 3629's table;
- failed.

So no overlongs, no surrogates, and nothing past U+10FFFF.
at_boundary() says whether the text so far ends between characters, which
a whole message must; C calls a message that does not "partial utf8".
Once a piece is refused every later one is too, so a validator cannot be
fed on past a failure by mistake.

Tests check it against the standard library's from_utf8, which follows
RFC 3629 too:
- exhaustively over every 1- and 2-byte sequence, and every 3-byte one
  after a multi-byte lead;
- the RFC's boundary code points;
- every two-cut split of mixed texts, which gives the same verdict as the
  whole.

As a one-off check that C behaves the same, C's lws_check_utf8() and its
table were compiled into a scratch harness, outside this tree, and
compared with a strict RFC 3629 reference.  Every 1- to 3-byte sequence
was fed whole and at every cut, plus every 4-byte one after an F0..FF
lead: 134,414,848 cases, none differing.

Co-Authored-By: Claude Opus 5.5 (1M context) <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_019kg5Eemy68ZaqDBcUJQG6J
diff --git a/crates/npro-core/src/lib.rs b/crates/npro-core/src/lib.rs index f00bff3..b6f33f5 100644 --- a/crates/npro-core/src/lib.rs +++ b/crates/npro-core/src/lib.rs @@ -11,3 +11,4 @@ pub mod base64; pub mod random; pub mod sha1; +pub mod utf8; diff --git a/crates/npro-core/src/utf8.rs b/crates/npro-core/src/utf8.rs new file mode 100644 index 0000000..0a7f14e --- /dev/null +++ b/crates/npro-core/src/utf8.rs @@ -0,0 +1,217 @@ +//! Incremental UTF-8 validation, for ws text messages and close reasons +//! (RFC 6455 8.1), which arrive in pieces of any size. +//! +//! This is C's `lws_check_utf8()`: well-formed UTF-8 as RFC 3629 4 defines +//! it, so no overlong forms, no surrogates and nothing past U+10FFFF, with +//! a bad byte refused as soon as it is seen. C keeps where it is inside a +//! character in one byte read through a table; here it is an enum, and the +//! ranges are those of RFC 3629's table. + +/// The bytes are not well-formed UTF-8. +#[derive(Clone, Copy, Debug, PartialEq, Eq)] +pub struct Invalid; + +impl core::fmt::Display for Invalid { + fn fmt(&self, f: &mut core::fmt::Formatter<'_>) -> core::fmt::Result { + f.write_str("invalid UTF-8") + } +} + +impl core::error::Error for Invalid {} + +/// How many continuation bytes the current character still needs. +#[derive(Clone, Copy, Debug, PartialEq, Eq)] +enum Left { + One, + Two, + Three, +} + +/// Where the validator is. +#[derive(Clone, Copy, Debug, PartialEq, Eq)] +enum State { + /// Between characters. + Boundary, + /// Inside a character: the next byte must be in `lo..=hi`. + Inside { left: Left, lo: u8, hi: u8 }, + /// An invalid byte was seen; nothing more is valid. + Failed, +} + +/// Validates UTF-8 fed to it in pieces. +/// +/// ``` +/// use npro_core::utf8::Utf8Validator; +/// +/// let mut v = Utf8Validator::new(); +/// let euro = "€".as_bytes(); // e2 82 ac +/// v.feed(&euro[..1])?; +/// assert!(!v.at_boundary()); // a message ending here is "partial utf8" +/// v.feed(&euro[1..])?; +/// assert!(v.at_boundary()); +/// assert!(v.feed(&[0xc0, 0xaf]).is_err()); // an overlong '/' +/// # Ok::<(), npro_core::utf8::Invalid>(()) +/// ``` +#[derive(Clone, Copy, Debug, PartialEq, Eq)] +pub struct Utf8Validator { + state: State, +} + +impl Default for Utf8Validator { + fn default() -> Self { + Self::new() + } +} + +impl Utf8Validator { + /// A validator between characters, at the start of a message. + #[must_use] + pub const fn new() -> Self { + Self { + state: State::Boundary, + } + } + + /// Takes the next piece of the text. + /// + /// # Errors + /// + /// [`Invalid`] at the first byte that cannot be part of well-formed + /// UTF-8, here or in an earlier piece. After that, every piece is + /// refused: a ws connection given invalid text closes with 1007. + pub fn feed(&mut self, bytes: &[u8]) -> Result<(), Invalid> { + for &b in bytes { + self.state = next(self.state, b); + if self.state == State::Failed { + return Err(Invalid); + } + } + if self.state == State::Failed { + return Err(Invalid); + } + Ok(()) + } + + /// Whether the text so far ends between characters, as a whole message + /// must. + #[must_use] + pub fn at_boundary(&self) -> bool { + self.state == State::Boundary + } +} + +/// The state after byte `b` (RFC 3629 4). +const fn next(s: State, b: u8) -> State { + match s { + State::Failed => State::Failed, + State::Boundary => match b { + 0x00..=0x7f => State::Boundary, + 0xc2..=0xdf => inside(Left::One, 0x80, 0xbf), + 0xe0 => inside(Left::Two, 0xa0, 0xbf), + 0xe1..=0xec | 0xee..=0xef => inside(Left::Two, 0x80, 0xbf), + 0xed => inside(Left::Two, 0x80, 0x9f), + 0xf0 => inside(Left::Three, 0x90, 0xbf), + 0xf1..=0xf3 => inside(Left::Three, 0x80, 0xbf), + 0xf4 => inside(Left::Three, 0x80, 0x8f), + // continuation bytes, the overlong leads c0 and c1, and leads + // past U+10FFFF + _ => State::Failed, + }, + State::Inside { left, lo, hi } => { + if b < lo || b > hi { + return State::Failed; + } + match left { + Left::One => State::Boundary, + Left::Two => inside(Left::One, 0x80, 0xbf), + Left::Three => inside(Left::Two, 0x80, 0xbf), + } + } + } +} + +const fn inside(left: Left, lo: u8, hi: u8) -> State { + State::Inside { left, lo, hi } +} + +#[cfg(test)] +mod tests { + use super::*; + + /// What RFC 3629 says of a whole text, as the standard library says it. + fn oracle(bytes: &[u8]) -> bool { + core::str::from_utf8(bytes).is_ok() + } + + fn whole(bytes: &[u8]) -> bool { + let mut v = Utf8Validator::new(); + v.feed(bytes).is_ok() && v.at_boundary() + } + + #[test] + fn agrees_with_the_standard_library_on_every_short_sequence() { + for a in 0..=255u8 { + assert_eq!(whole(&[a]), oracle(&[a]), "{a:02x}"); + for b in 0..=255u8 { + assert_eq!(whole(&[a, b]), oracle(&[a, b]), "{a:02x} {b:02x}"); + } + } + // every three bytes after a multi-byte lead + for a in 0xc0..=0xffu8 { + for b in 0..=255u8 { + for c in 0..=255u8 { + assert_eq!(whole(&[a, b, c]), oracle(&[a, b, c])); + } + } + } + } + + #[test] + fn the_rfc_3629_boundaries() { + for (bytes, ok, what) in [ + (&[0xf4, 0x8f, 0xbf, 0xbf][..], true, "U+10FFFF"), + (&[0xf4, 0x90, 0x80, 0x80][..], false, "U+110000"), + (&[0xed, 0x9f, 0xbf][..], true, "U+D7FF"), + (&[0xed, 0xa0, 0x80][..], false, "a surrogate"), + (&[0xee, 0x80, 0x80][..], true, "U+E000"), + (&[0xe0, 0x9f, 0xbf][..], false, "an overlong U+07FF"), + (&[0xf0, 0x8f, 0xbf, 0xbf][..], false, "an overlong U+FFFF"), + (&[0xf0, 0x90, 0x80, 0x80][..], true, "U+10000"), + ] { + assert_eq!(whole(bytes), ok, "{what}"); + assert_eq!(oracle(bytes), ok, "{what}"); + } + } + + #[test] + fn the_same_verdict_however_the_text_is_split() { + let texts: [&[u8]; 4] = [ + "a€𝄞ç\u{10ffff}z".as_bytes(), + &[0x61, 0xe2, 0x82, 0xac, 0xed, 0xa0, 0x80], + &[0xf0, 0x9d, 0x84], + &[0xe2, 0x28, 0xa1], + ]; + for t in texts { + let want = whole(t); + for i in 0..=t.len() { + for j in i..=t.len() { + let mut v = Utf8Validator::new(); + let ok = v.feed(&t[..i]).is_ok() + && v.feed(&t[i..j]).is_ok() + && v.feed(&t[j..]).is_ok() + && v.at_boundary(); + assert_eq!(ok, want, "{t:02x?} cut at {i} and {j}"); + } + } + } + } + + #[test] + fn a_failure_stays_failed() { + let mut v = Utf8Validator::new(); + assert!(v.feed(&[0xff]).is_err()); + assert!(v.feed(b"fine").is_err()); + assert!(v.feed(&[]).is_err()); + assert!(!v.at_boundary()); + } +}
Page fetched 0s ago, creation time: 2ms (vhost etag hits: 0%, cache hits: 0%)