From 310469a2f15f2bab0ad40eb1c92b3d3b6e407f1a Mon Sep 17 00:00:00 2001 From: kurok <22548029+kurok@users.noreply.github.com> Date: Fri, 28 Aug 2026 08:17:53 +0100 Subject: [PATCH 1/3] Vendored mailparse: word-at-a-time scans in plain std, memchr dropped Upstream declined the memchr version of the two mailparse fixes as an added dependency. This replaces the two vendored functions with a dependency-free implementation in one new module, vendor/mailparse/src/bytescan.rs, and drops memchr from the lockfile. The same code is what the upstream proposal now carries, so if a release includes it the vendor directory can go; PATCH.md has the removal steps back alongside the sync procedure. The scans read a usize at a time in plain std with no unsafe: chunks_exact and from_le_bytes for the loads, the exact haszero / hasless word tests to decide whether a word needs a closer look, and trailing_zeros to go straight to a hit rather than rescan the word. The boundary search tests four words per branch in the no-hit case; the whitespace strip, whose hits come every line, one word, walking the mask's set bits and re-checking each with is_ascii_whitespace so 0x0B and the other control bytes are kept as before. Interleaved A/B against master's memchr version on the M4, both rustc 1.98.0: metadata paths within 0.4%, full parse 0.256 -> 0.240 ms, parse_many 2.13 -> 1.97 ms. The strip is faster because it is one pass where the memchr version made two searches per run. Three earlier variants were measured and rejected on the way: single-word steps cost the metadata path 28%, two-word steps that rescanned on a hit cost the full parse 7%, and splitting the no-hit test into two branches cost the metadata path 18%. 803 tests pass on the built wheel; the vendored crate's suite passes, including differential tests over every alignment and byte value. Whether the new loops are themselves placement-sensitive on x86 is the one question a local A/B cannot answer; one toolchain A/B on the merged master will. Signed-off-by: kurok <22548029+kurok@users.noreply.github.com> --- CHANGELOG.md | 9 ++ CONTRIBUTING.md | 19 ++- Cargo.lock | 7 - vendor/mailparse/Cargo.toml | 3 - vendor/mailparse/PATCH.md | 120 +++++++-------- vendor/mailparse/src/body.rs | 72 +-------- vendor/mailparse/src/bytescan.rs | 257 +++++++++++++++++++++++++++++++ vendor/mailparse/src/lib.rs | 3 +- 8 files changed, 338 insertions(+), 152 deletions(-) create mode 100644 vendor/mailparse/src/bytescan.rs diff --git a/CHANGELOG.md b/CHANGELOG.md index 5849902..f194543 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -9,6 +9,15 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0 ### Changed +- **The vendored mailparse fast paths are now dependency-free.** Upstream declined the + `memchr` version as an added dependency, so the two functions now use word-at-a-time + scans in plain `std` (`vendor/mailparse/src/bytescan.rs`: `chunks_exact` + + `from_le_bytes`, the exact `haszero` / `hasless` word tests, `trailing_zeros` to jump to + a hit; no `unsafe`), and `memchr` leaves the lockfile. Same code is now proposed + upstream (staktrace/mailparse#142), so the vendored copy can be dropped if a release + carries it. Interleaved A/B against the `memchr` version, Apple M4: metadata paths + within 0.4%, full parse 0.256 -> 0.240 ms, `parse_many` 2.13 -> 1.97 ms -- the strip is + one pass where the `memchr` version made two searches per run. - **Rust toolchain pin moved from 1.97.1 to 1.98.0** (#120). The pin was a holding action against a measured 15-30% slowdown under 1.98.0; the cause turned out to be the two mailparse loops above -- the compiler had not changed their instructions, diff --git a/CONTRIBUTING.md b/CONTRIBUTING.md index 37e5634..69c605c 100644 --- a/CONTRIBUTING.md +++ b/CONTRIBUTING.md @@ -314,19 +314,22 @@ runners' Zen CPUs a scalar loop straddling the wrong 64-byte boundary falls out of the micro-op cache and runs at half speed. An Apple M4 measured the same two builds at +/-0.2%. -The fix replaces that scan with `memchr::memmem` -- vectorised, and laid out -independently of this crate -- via the patched copy in `vendor/mailparse` -(a permanent carry: upstream declined the change as an added dependency; -`vendor/mailparse/PATCH.md` has the sync procedure). Metadata mode went -0.365 -> 0.030 ms and the full parse 1.10 -> 0.76 ms. +The fix replaces that scan with a word-at-a-time search -- 32 bytes per branch, +so even a placement-induced halving would cost a fraction of what the byte loop +did -- via the patched copy in `vendor/mailparse` (`src/bytescan.rs`; upstream +declined a `memchr` version as an added dependency, and this dependency-free one +is proposed in its place -- `vendor/mailparse/PATCH.md` has the sync and removal +procedures). Metadata mode went 0.365 -> 0.030 ms and the full parse +1.10 -> 0.76 ms. With that gone, sampling the *full* parse put **77.7%** of what remained in `decode_base64`'s whitespace filter -- `iter().filter().cloned().collect()`, a test and a push per byte -- and the toolchain A/B confirmed it carried the rest of the placement sensitivity: decoding paths still moved +22% on one runner and +5% -on another while the metadata paths had gone flat. Same fix, same place: the -whitespace is found with `memchr` and the runs between are copied whole. The -full parse went 0.83 -> 0.28 ms on top. +on another while the metadata paths had gone flat. Same fix, same place: words +with no byte below `0x21` are skipped whole and the runs between whitespace are +copied in one piece. The full parse went 0.83 -> 0.28 ms on top, and 0.26 -> 0.24 +again when the `memchr` version gave way to the dependency-free one. **What remains true.** The gate has a false-positive mode its noise floor cannot see: the controls are pure Python and do not care how the extension was laid out, diff --git a/Cargo.lock b/Cargo.lock index 2df0a24..00becaa 100644 --- a/Cargo.lock +++ b/Cargo.lock @@ -66,16 +66,9 @@ version = "0.16.1" dependencies = [ "charset", "data-encoding", - "memchr", "quoted_printable", ] -[[package]] -name = "memchr" -version = "2.8.3" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "cf8baf1c55e62ffcace7a9f06f4bd9cd3f0c4beb022d3b367256b91b87513d98" - [[package]] name = "once_cell" version = "1.21.4" diff --git a/vendor/mailparse/Cargo.toml b/vendor/mailparse/Cargo.toml index 16c5663..60f4e0c 100644 --- a/vendor/mailparse/Cargo.toml +++ b/vendor/mailparse/Cargo.toml @@ -37,9 +37,6 @@ categories = [ license = "0BSD" repository = "https://github.com/staktrace/mailparse" -[dependencies.memchr] -version = "2.7.0" - [dependencies.charset] version = "0.1.3" diff --git a/vendor/mailparse/PATCH.md b/vendor/mailparse/PATCH.md index 4c39ec0..9dd6384 100644 --- a/vendor/mailparse/PATCH.md +++ b/vendor/mailparse/PATCH.md @@ -1,39 +1,33 @@ # Patched copy of `mailparse` 0.16.1 This directory is [mailparse 0.16.1](https://crates.io/crates/mailparse/0.16.1) as published, -with **two functions changed** and one dependency added. It is applied through -`[patch.crates-io]` in the root `Cargo.toml` (and `fuzz/Cargo.toml`), so `cargo` sees the -same crate name and version and every other dependency resolves exactly as before. -`memchr = "2.7.0"` was added to this crate's `[dependencies]` (MIT OR Unlicense, no -dependencies of its own). - -## The changes - -Both replace a byte-at-a-time loop over the whole message body with a vectorised search. -Both return exactly what the code they replace returned. - -**1. `find_from_u8` in `src/lib.rs`** -- the search `parse_mail` runs for every MIME -boundary -- scanned byte by byte: - -```rust -for i in ix_start..=ix_end { - if line[i] == key[0] { /* compare the rest */ } -} -``` - -It now calls `memchr::memmem::find`. Same result: first occurrence of `key` at or after -`ix_start`, `None` when there is none. - -**2. `decode_base64` in `src/body.rs`** stripped whitespace before decoding with -`body.iter().filter(|c| !c.is_ascii_whitespace()).cloned().collect()`: a test and a -bounds-checked push per byte. It now calls `strip_ascii_whitespace`, which finds -whitespace with `memchr` and copies the runs between whole -- one search and one memcpy -per 76-byte line in the common case. The set of bytes removed is unchanged (exactly -`u8::is_ascii_whitespace`: space, tab, LF, form feed, CR) and a unit test in `body.rs` -checks it against the original filter over every byte value. - -`diff -r` against the registry copy shows exactly `src/lib.rs`, `src/body.rs`, -`Cargo.toml` and this file. +with **one module added and two functions changed to use it**. No dependency is added. It +is applied through `[patch.crates-io]` in the root `Cargo.toml` (and `fuzz/Cargo.toml`), so +`cargo` sees the same crate name and version and every other dependency resolves exactly +as before. + +## The change + +`src/bytescan.rs` (new) scans a machine word at a time instead of a byte at a time, in +plain `std` with no `unsafe`: `usize` chunks via `chunks_exact` + `from_le_bytes`, the +exact `haszero` / `hasless` word tests (Anderson, *Bit Twiddling Hacks*) to decide whether +a word needs a closer look, and `trailing_zeros` to jump to a hit rather than rescan. + +1. **`find_from_u8` in `src/lib.rs`** -- the search `parse_mail` runs for every MIME + boundary -- scanned byte by byte. It now calls `bytescan::find`, which tests four words + per branch for the key's first byte and compares the rest only at candidates. Same + result: first occurrence of `key` at or after `ix_start`, `None` when there is none. +2. **`decode_base64` in `src/body.rs`** stripped whitespace with + `iter().filter(|c| !c.is_ascii_whitespace()).cloned().collect()`: a test and a + bounds-checked push per byte. It now calls `bytescan::strip_ascii_whitespace`, which + skips any word with no byte below `0x21`, walks the mask bits of a word that has one, + re-checks each candidate with `is_ascii_whitespace` (so `0x0B` and the other control + bytes are kept, as before), and copies the runs between in one piece. + +`bytescan::tests` compares both against the loops they replace over a generated corpus at +every alignment and over every byte value. `diff -r` against the registry copy shows +exactly `src/bytescan.rs`, `src/lib.rs` (a `mod` line and one function), `src/body.rs` +(one function) and this file. ## Why @@ -45,35 +39,37 @@ when a loop straddled a 64-byte boundary. A rustc minor version (#120) and a version-string bump (#204) each moved the loops and each read as a regression of up to 96% -- with zero change to the instructions executed. -Measured on this machine (interleaved A/B, Apple M4), original master to both patches: +Interleaved A/B on an Apple M4, original master to this copy: | benchmark | before | after | |---|---|---| -| `parse_email(mode="metadata")` | 0.365 ms | 0.034 ms | -| `parse_email` (full) | 1.094 ms | 0.281 ms | -| `parse_many` (8 x 767 KiB) | 9.082 ms | 2.177 ms | - -## Keeping this in sync - -This copy is permanent. The change was proposed upstream -([staktrace/mailparse#142](https://github.com/staktrace/mailparse/pull/142)) and declined on -2026-08-28: the project does not accept pull requests that add external dependencies, -particularly ones relying heavily on unsafe code -- which describes `memchr`'s SIMD paths -exactly. So no future mailparse release will carry these functions, and every upstream -release has to be merged into this directory by hand: - -1. `diff -r` the new release against the previous one (both in - `~/.cargo/registry/src/*/mailparse-/`) and apply that diff to this copy -- - *not* the other way round, or the two functions revert. -2. Keep `find_from_u8`, `strip_ascii_whitespace` and the `strip_ascii_whitespace_tests` - module as they are here, and `memchr` in `Cargo.toml`. -3. Bump the version in this copy's `Cargo.toml` and the `mailparse = "..."` requirement in - the root `Cargo.toml` and `fuzz/Cargo.toml` together, since `[patch]` only applies when - the patched version satisfies the requirement. -4. Run this copy's own suite (the lint job does: `cargo test --manifest-path - vendor/mailparse/Cargo.toml`), then the benchmark gate. The numbers should not move. - -If upstreaming is ever wanted, the shape that could be accepted is a dependency-free one: -a word-at-a-time (SWAR) scan in plain `std`, which is what `core`'s own `memchr` does -internally. It would be slower than `memchr`'s SIMD paths and would need measuring against -this copy before replacing it. +| `parse_email(mode="metadata")` | 0.365 ms | 0.030 ms | +| `parse_email` (full) | 1.094 ms | 0.240 ms | +| `parse_many` (8 x 767 KiB) | 9.082 ms | 1.973 ms | + +A first version used `memchr` (PRs #213, #214). Upstream declined it for adding a +dependency, so this dependency-free version replaced it (it measured within noise on the +metadata paths and 3-11% faster on the decoding paths, because the strip is one pass). + +## Upstream, and when this goes away + +The same change is proposed upstream as +[staktrace/mailparse#142](https://github.com/staktrace/mailparse/pull/142) (revised +2026-08-28 without the dependency). If a mailparse release includes it: + +1. bump `mailparse` in the root `Cargo.toml` and `fuzz/Cargo.toml` to that release, +2. delete the two `[patch.crates-io]` sections and this directory, +3. drop the `vendor/mailparse/**/*` entry from `[tool.maturin] include` in + `pyproject.toml` and the `vendored mailparse tests` step from the lint job, +4. run the benchmark gate: the numbers should not move. + +Until then, each upstream mailparse release is a hand-merge into this copy: + +1. `diff -r` the new release against the previous one (both under + `~/.cargo/registry/src/*/mailparse-/`) and apply that diff here -- not the + other way round, or the two functions revert; +2. keep `src/bytescan.rs`, the `mod bytescan;` line and the two call sites; +3. bump the version in this copy's `Cargo.toml` and the `mailparse = "..."` requirement in + both root manifests together, since `[patch]` only applies when the patched version + satisfies the requirement; +4. run this copy's own suite (the lint job does), then the benchmark gate. diff --git a/vendor/mailparse/src/body.rs b/vendor/mailparse/src/body.rs index 2c51f29..60a7d8f 100644 --- a/vendor/mailparse/src/body.rs +++ b/vendor/mailparse/src/body.rs @@ -137,36 +137,10 @@ impl<'a> BinaryBody<'a> { } fn decode_base64(body: &[u8]) -> Result, MailParseError> { - let cleaned = strip_ascii_whitespace(body); + let cleaned = crate::bytescan::strip_ascii_whitespace(body); Ok(data_encoding::BASE64_MIME_PERMISSIVE.decode(&cleaned)?) } -/// Copy `body` without its ASCII whitespace -- the same bytes `u8::is_ascii_whitespace` -/// names: space, tab, newline, form feed, carriage return. -/// -/// Whitespace is located with a vectorised search and the runs between are copied -/// whole, rather than testing and pushing one byte at a time. A base64 body is -/// almost entirely 76-byte lines ending in CRLF, so the common case is one search -/// and one copy per line. Tabs and form feeds are rare enough that they get a second -/// search over each run instead of a place in the first. -fn strip_ascii_whitespace(body: &[u8]) -> Vec { - let mut cleaned = Vec::with_capacity(body.len()); - let mut rest = body; - loop { - let end = memchr::memchr3(b'\r', b'\n', b' ', rest).unwrap_or(rest.len()); - let mut run = &rest[..end]; - while let Some(j) = memchr::memchr2(b'\t', 0x0c, run) { - cleaned.extend_from_slice(&run[..j]); - run = &run[j + 1..]; - } - cleaned.extend_from_slice(run); - if end == rest.len() { - return cleaned; - } - rest = &rest[end + 1..]; - } -} - fn decode_quoted_printable(body: &[u8]) -> Result, MailParseError> { Ok(quoted_printable::decode( body, @@ -183,47 +157,3 @@ fn get_body_as_string(body: &[u8], ctype: &ParsedContentType) -> Result Vec { - body.iter() - .filter(|c| !c.is_ascii_whitespace()) - .cloned() - .collect() - } - - #[test] - fn matches_the_filter_it_replaces() { - let cases: &[&[u8]] = &[ - b"", - b" ", - b" \t\r\n\x0c", - b"abc", - b" abc ", - b"ab cd\tef\ngh\rij\x0ckl", - b"\x0b", // vertical tab is NOT ascii whitespace; must be kept - b"a\x0bb", - b"\t\tab\x0c\x0ccd", - b"QUJD\r\nREVG\r\n", - b"QUJD REVG\tR0hJ\x0cSktM", - ]; - for case in cases { - assert_eq!(strip_ascii_whitespace(case), reference(case), "{:?}", case); - } - // every byte value, alone and next to whitespace - for b in 0u8..=255 { - let single = [b]; - assert_eq!( - strip_ascii_whitespace(&single), - reference(&single), - "{:?}", - b - ); - let mixed = [b' ', b, b'\t', b, b'\r', b'\n', b]; - assert_eq!(strip_ascii_whitespace(&mixed), reference(&mixed), "{:?}", b); - } - } -} diff --git a/vendor/mailparse/src/bytescan.rs b/vendor/mailparse/src/bytescan.rs new file mode 100644 index 0000000..4e30c3e --- /dev/null +++ b/vendor/mailparse/src/bytescan.rs @@ -0,0 +1,257 @@ +//! Byte scans that look at a machine word at a time instead of a byte at a time. +//! +//! Two loops in the parse path used to test every byte of the message body on its +//! own: the search for the next MIME boundary, and the whitespace strip before +//! base64 decoding. On a message with large attachments they are most of the parse, +//! and a byte-at-a-time loop's speed also depends on where the linker happens to +//! place it -- the same instructions ran at half speed on x86-64 when the loop +//! straddled a 64-byte boundary. +//! +//! These versions read `usize`-sized chunks with `chunks_exact` and +//! `from_ne_bytes` (plain loads; no `unsafe`) and use two classic word tricks to +//! decide whether a chunk needs a closer look: +//! +//! - `zero_byte_mask(x)`: non-zero iff some byte of `x` is zero. +//! - `below_mask(x, n)`: non-zero iff some byte of `x` is below `n`, exact for `n <= 128`. +//! +//! Where a mask is non-zero, its set bits say which bytes to look at, so a hit costs a +//! `trailing_zeros` rather than a rescan of the word. The byte search tests four words +//! per branch; the whitespace strip, whose hits are frequent (every line), one. +//! +//! Both are exact, so a chunk is only examined byte by byte when it really contains +//! a candidate. See Anderson, "Bit Twiddling Hacks", `haszero` and `hasless`. + +// `chunks_exact` with a constant size is what clippy would rewrite to `as_chunks`, which +// is only stable from Rust 1.88 -- above this crate's minimum supported version. +#![allow(clippy::chunks_exact_to_as_chunks)] + +const WORD: usize = core::mem::size_of::(); +const LO: usize = usize::from_ne_bytes([0x01; WORD]); +const HI: usize = usize::from_ne_bytes([0x80; WORD]); + +/// Each byte of the result has its high bit set iff that byte of `x` is zero -- exactly +/// so for the lowest such byte; bytes above it may be marked spuriously by the borrow, +/// which is why callers re-check when they walk more than the first hit. +#[inline] +fn zero_byte_mask(x: usize) -> usize { + x.wrapping_sub(LO) & !x & HI +} + +/// Each byte of the result has its high bit set iff that byte of `x` is below `n` +/// (`n <= 128`), with the same caveat as `zero_byte_mask` above the lowest hit: never a +/// false negative, possibly a false positive, so callers re-check. +#[inline] +fn below_mask(x: usize, n: u8) -> usize { + x.wrapping_sub(LO.wrapping_mul(n as usize)) & !x & HI +} + +/// The word at `chunk`, byte 0 in the low-order bits, so that a mask bit at position +/// `p` refers to byte `p / 8` regardless of the machine's endianness. +#[inline] +fn word(chunk: &[u8]) -> usize { + // Callers pass exactly `WORD` bytes. Spelled with `copy_from_slice` rather than + // `try_into` so it reads the same in every edition; it compiles to one load. + let mut bytes = [0u8; WORD]; + bytes.copy_from_slice(chunk); + usize::from_le_bytes(bytes) +} + +/// Byte index of the lowest set bit of a non-zero mask. +#[inline] +fn first_hit(mask: usize) -> usize { + (mask.trailing_zeros() / 8) as usize +} + +/// Words per step of the byte search: the four masks are OR-ed, so the no-hit case is +/// one branch per 32 bytes, and the four loads and tests are independent work the CPU +/// can overlap. +const STEP_WORDS: usize = 4; +const STEP: usize = STEP_WORDS * WORD; + +/// Index of the first `needle` in `haystack`. +pub(crate) fn find_byte(haystack: &[u8], needle: u8) -> Option { + let repeated = LO.wrapping_mul(needle as usize); + let mut steps = haystack.chunks_exact(STEP); + let mut offset = 0; + for step in &mut steps { + let m0 = zero_byte_mask(word(&step[..WORD]) ^ repeated); + let m1 = zero_byte_mask(word(&step[WORD..2 * WORD]) ^ repeated); + let m2 = zero_byte_mask(word(&step[2 * WORD..3 * WORD]) ^ repeated); + let m3 = zero_byte_mask(word(&step[3 * WORD..]) ^ repeated); + if (m0 | m1 | m2 | m3) != 0 { + let (k, m) = [m0, m1, m2, m3] + .iter() + .copied() + .enumerate() + .find(|&(_, m)| m != 0) + .unwrap(); + return Some(offset + k * WORD + first_hit(m)); + } + offset += STEP; + } + steps + .remainder() + .iter() + .position(|&b| b == needle) + .map(|i| offset + i) +} + +/// Index of the first occurrence of `key` in `haystack`, or `None`. +/// +/// Scans for `key[0]` a word at a time and compares the rest only at candidates. +pub(crate) fn find(haystack: &[u8], key: &[u8]) -> Option { + let first = *key.first()?; + let mut at = 0; + while let Some(i) = find_byte(&haystack[at..], first) { + let candidate = at + i; + if haystack[candidate..].starts_with(key) { + return Some(candidate); + } + at = candidate + 1; + } + None +} + +/// `body` without its ASCII whitespace -- exactly the bytes `u8::is_ascii_whitespace` +/// names: space, tab, line feed, form feed, carriage return. +/// +/// Every whitespace byte is below `0x21`, so a word with no byte below `0x21` has no +/// whitespace and is skipped whole. For a word that has candidates, the mask says +/// where they are, and each is re-checked with `is_ascii_whitespace`, so `0x0B` and +/// the other control characters below `0x21` are kept, as before. Runs of kept bytes +/// are copied in one piece rather than pushed one at a time. A base64 body is mostly +/// 76-byte lines ending in CRLF, so the common case is two mask bits and one copy per +/// line. +pub(crate) fn strip_ascii_whitespace(body: &[u8]) -> Vec { + let mut cleaned = Vec::with_capacity(body.len()); + let mut run_start = 0; + let mut offset = 0; + let mut words = body.chunks_exact(WORD); + for chunk in &mut words { + let mut mask = below_mask(word(chunk), 0x21); + while mask != 0 { + let i = offset + first_hit(mask); + if body[i].is_ascii_whitespace() { + cleaned.extend_from_slice(&body[run_start..i]); + run_start = i + 1; + } + mask &= mask - 1; + } + offset += WORD; + } + for (i, &b) in words.remainder().iter().enumerate() { + if b.is_ascii_whitespace() { + cleaned.extend_from_slice(&body[run_start..offset + i]); + run_start = offset + i + 1; + } + } + cleaned.extend_from_slice(&body[run_start..]); + cleaned +} + +#[cfg(test)] +mod tests { + use super::{find, find_byte, strip_ascii_whitespace}; + + fn naive_find(haystack: &[u8], key: &[u8]) -> Option { + if key.is_empty() || haystack.len() < key.len() { + return None; + } + (0..=haystack.len() - key.len()).find(|&i| &haystack[i..i + key.len()] == key) + } + + fn naive_strip(body: &[u8]) -> Vec { + body.iter() + .filter(|c| !c.is_ascii_whitespace()) + .cloned() + .collect() + } + + /// Deterministic pseudo-random bytes drawn from a small alphabet, so that needles + /// actually occur, at every alignment relative to the word size. + fn corpus() -> Vec> { + let mut out = Vec::new(); + let mut state: u32 = 0x9E37_79B9; + for len in 0..80 { + for _ in 0..4 { + let mut v = Vec::with_capacity(len); + for _ in 0..len { + state ^= state << 13; + state ^= state >> 17; + state ^= state << 5; + const ALPHABET: &[u8; 12] = b"-\r\n \tab=\x0c\x0b\x00\xff"; + v.push(ALPHABET[(state % 12) as usize]); + } + out.push(v); + } + } + out + } + + #[test] + fn find_byte_matches_position() { + for h in corpus() { + for needle in [b'-', b'\n', b'a', 0x00, 0xff, b'z'] { + assert_eq!( + find_byte(&h, needle), + h.iter().position(|&b| b == needle), + "{:?} / {:?}", + h, + needle + ); + } + } + } + + #[test] + fn find_matches_naive_search() { + let keys: &[&[u8]] = &[ + b"-", + b"--", + b"--ab", + b"\n", + b"\r\n", + b"ab=", + b"zz", + b"\x00\xff", + ]; + for h in corpus() { + for key in keys { + assert_eq!(find(&h, key), naive_find(&h, key), "{:?} / {:?}", h, key); + } + } + assert_eq!(find(b"abc", b""), None); + assert_eq!(find(b"", b"a"), None); + assert_eq!(find(b"ab", b"abc"), None); + } + + #[test] + fn strip_matches_filter() { + for h in corpus() { + assert_eq!(strip_ascii_whitespace(&h), naive_strip(&h), "{:?}", h); + } + for b in 0u8..=255 { + let single = [b]; + assert_eq!( + strip_ascii_whitespace(&single), + naive_strip(&single), + "{:?}", + b + ); + let mixed = [ + b' ', b, b'\t', b, b'\r', b'\n', b, b, b, b, b, b, b, b, b, b, b'\x0c', b, + ]; + assert_eq!( + strip_ascii_whitespace(&mixed), + naive_strip(&mixed), + "{:?}", + b + ); + } + // vertical tab is below 0x21 but is not ASCII whitespace: kept + assert_eq!( + strip_ascii_whitespace(b"a\x0bb\x0b\x0b\x0b\x0b\x0b\x0bc"), + b"a\x0bb\x0b\x0b\x0b\x0b\x0b\x0bc" + ); + } +} diff --git a/vendor/mailparse/src/lib.rs b/vendor/mailparse/src/lib.rs index 4203e21..9554119 100644 --- a/vendor/mailparse/src/lib.rs +++ b/vendor/mailparse/src/lib.rs @@ -13,6 +13,7 @@ use charset::{decode_latin1, Charset}; mod addrparse; pub mod body; +mod bytescan; mod dateparse; mod header; pub mod headers; @@ -121,7 +122,7 @@ pub(crate) fn find_from(line: &str, ix_start: usize, key: &str) -> Option fn find_from_u8(line: &[u8], ix_start: usize, key: &[u8]) -> Option { assert!(!key.is_empty()); assert!(ix_start <= line.len()); - memchr::memmem::find(&line[ix_start..], key).map(|v| ix_start + v) + bytescan::find(&line[ix_start..], key).map(|v| ix_start + v) } #[test] From e5f48ac43cab4807bd0820d32c50cf319220bf79 Mon Sep 17 00:00:00 2001 From: kurok <22548029+kurok@users.noreply.github.com> Date: Fri, 28 Aug 2026 08:25:29 +0100 Subject: [PATCH 2/3] Test eight words per branch in the byte search The gate on the four-word version measured the metadata paths 13-14% behind the memchr build on an EPYC 7763, where memchr's AVX2 path compares 32 bytes per instruction. Doubling the words per branch to eight -- one cache line -- halves the branch count again and gives the CPU eight independent loads and tests to overlap. On the M4 this variant is 4-5% ahead of memchr on the metadata paths and 12% ahead on the decoding paths; whether the same holds on x86 is what this push asks the gate. Signed-off-by: kurok <22548029+kurok@users.noreply.github.com> --- CONTRIBUTING.md | 2 +- vendor/mailparse/PATCH.md | 5 +++-- vendor/mailparse/src/bytescan.rs | 24 +++++++++++++----------- 3 files changed, 17 insertions(+), 14 deletions(-) diff --git a/CONTRIBUTING.md b/CONTRIBUTING.md index 69c605c..bcff589 100644 --- a/CONTRIBUTING.md +++ b/CONTRIBUTING.md @@ -314,7 +314,7 @@ runners' Zen CPUs a scalar loop straddling the wrong 64-byte boundary falls out of the micro-op cache and runs at half speed. An Apple M4 measured the same two builds at +/-0.2%. -The fix replaces that scan with a word-at-a-time search -- 32 bytes per branch, +The fix replaces that scan with a word-at-a-time search -- 64 bytes per branch, so even a placement-induced halving would cost a fraction of what the byte loop did -- via the patched copy in `vendor/mailparse` (`src/bytescan.rs`; upstream declined a `memchr` version as an added dependency, and this dependency-free one diff --git a/vendor/mailparse/PATCH.md b/vendor/mailparse/PATCH.md index 9dd6384..e17581f 100644 --- a/vendor/mailparse/PATCH.md +++ b/vendor/mailparse/PATCH.md @@ -14,8 +14,9 @@ exact `haszero` / `hasless` word tests (Anderson, *Bit Twiddling Hacks*) to deci a word needs a closer look, and `trailing_zeros` to jump to a hit rather than rescan. 1. **`find_from_u8` in `src/lib.rs`** -- the search `parse_mail` runs for every MIME - boundary -- scanned byte by byte. It now calls `bytescan::find`, which tests four words - per branch for the key's first byte and compares the rest only at candidates. Same + boundary -- scanned byte by byte. It now calls `bytescan::find`, which tests eight words + -- a cache line -- per branch for the key's first byte and compares the rest only at + candidates. Same result: first occurrence of `key` at or after `ix_start`, `None` when there is none. 2. **`decode_base64` in `src/body.rs`** stripped whitespace with `iter().filter(|c| !c.is_ascii_whitespace()).cloned().collect()`: a test and a diff --git a/vendor/mailparse/src/bytescan.rs b/vendor/mailparse/src/bytescan.rs index 4e30c3e..1c76285 100644 --- a/vendor/mailparse/src/bytescan.rs +++ b/vendor/mailparse/src/bytescan.rs @@ -15,7 +15,7 @@ //! - `below_mask(x, n)`: non-zero iff some byte of `x` is below `n`, exact for `n <= 128`. //! //! Where a mask is non-zero, its set bits say which bytes to look at, so a hit costs a -//! `trailing_zeros` rather than a rescan of the word. The byte search tests four words +//! `trailing_zeros` rather than a rescan of the word. The byte search tests eight words //! per branch; the whitespace strip, whose hits are frequent (every line), one. //! //! Both are exact, so a chunk is only examined byte by byte when it really contains @@ -62,10 +62,10 @@ fn first_hit(mask: usize) -> usize { (mask.trailing_zeros() / 8) as usize } -/// Words per step of the byte search: the four masks are OR-ed, so the no-hit case is -/// one branch per 32 bytes, and the four loads and tests are independent work the CPU -/// can overlap. -const STEP_WORDS: usize = 4; +/// Words per step of the byte search: the masks are OR-ed, so the no-hit case is one +/// branch per 64 bytes -- a cache line -- and the loads and tests are independent work +/// the CPU can overlap. +const STEP_WORDS: usize = 8; const STEP: usize = STEP_WORDS * WORD; /// Index of the first `needle` in `haystack`. @@ -74,12 +74,14 @@ pub(crate) fn find_byte(haystack: &[u8], needle: u8) -> Option { let mut steps = haystack.chunks_exact(STEP); let mut offset = 0; for step in &mut steps { - let m0 = zero_byte_mask(word(&step[..WORD]) ^ repeated); - let m1 = zero_byte_mask(word(&step[WORD..2 * WORD]) ^ repeated); - let m2 = zero_byte_mask(word(&step[2 * WORD..3 * WORD]) ^ repeated); - let m3 = zero_byte_mask(word(&step[3 * WORD..]) ^ repeated); - if (m0 | m1 | m2 | m3) != 0 { - let (k, m) = [m0, m1, m2, m3] + let mut masks = [0usize; STEP_WORDS]; + let mut any = 0; + for (k, m) in masks.iter_mut().enumerate() { + *m = zero_byte_mask(word(&step[k * WORD..(k + 1) * WORD]) ^ repeated); + any |= *m; + } + if any != 0 { + let (k, m) = masks .iter() .copied() .enumerate() From 162ab0c4b97b1b93fe5314c460a6081f45b0987e Mon Sep 17 00:00:00 2001 From: kurok <22548029+kurok@users.noreply.github.com> Date: Fri, 28 Aug 2026 08:34:32 +0100 Subject: [PATCH 3/3] Keep memchr for the byte search; dependency-free strip only The gate measured the dependency-free byte search behind memchr on x86 twice: +13-14% on the metadata paths with four words per branch, +9-11% with eight. A word-at-a-time scan tops out below a 32-byte AVX2 compare, and the rule for this copy is no degradation, so find_from_u8 goes back to memchr::memmem. The strip stays dependency-free. It is one pass where the memchr version made two searches per run, and it measured faster on both CPUs tried: -9 to -12% on the decoding paths on the M4, -2.5 to -2.9% on the gate's EPYC 7763. So the vendored copy now carries the half of the upstream proposal that is strictly no slower here, and PATCH.md records what switching the other half would cost if a mailparse release ever includes it. The unused search functions and their tests leave bytescan.rs in this copy; the upstream branch keeps them. Signed-off-by: kurok <22548029+kurok@users.noreply.github.com> --- CHANGELOG.md | 20 +++--- CONTRIBUTING.md | 23 +++--- Cargo.lock | 7 ++ vendor/mailparse/Cargo.toml | 3 + vendor/mailparse/PATCH.md | 73 ++++++++++--------- vendor/mailparse/src/bytescan.rs | 116 ++----------------------------- vendor/mailparse/src/lib.rs | 6 +- 7 files changed, 85 insertions(+), 163 deletions(-) diff --git a/CHANGELOG.md b/CHANGELOG.md index f194543..24f9ac4 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -9,15 +9,17 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0 ### Changed -- **The vendored mailparse fast paths are now dependency-free.** Upstream declined the - `memchr` version as an added dependency, so the two functions now use word-at-a-time - scans in plain `std` (`vendor/mailparse/src/bytescan.rs`: `chunks_exact` + - `from_le_bytes`, the exact `haszero` / `hasless` word tests, `trailing_zeros` to jump to - a hit; no `unsafe`), and `memchr` leaves the lockfile. Same code is now proposed - upstream (staktrace/mailparse#142), so the vendored copy can be dropped if a release - carries it. Interleaved A/B against the `memchr` version, Apple M4: metadata paths - within 0.4%, full parse 0.256 -> 0.240 ms, `parse_many` 2.13 -> 1.97 ms -- the strip is - one pass where the `memchr` version made two searches per run. +- **The base64 whitespace strip in the vendored mailparse is now dependency-free**, and + faster: `vendor/mailparse/src/bytescan.rs` skips any word with no byte below `0x21`, walks + the exact `hasless` mask of one that might, and copies runs in one piece -- plain `std`, + no `unsafe`. One pass where the `memchr` version made two searches per run: full parse + 0.254 -> 0.228 ms and `parse_many` 2.09 -> 1.83 ms on an Apple M4, -2.5 to -2.9% on the + CI gate's EPYC 7763. The MIME boundary search keeps `memchr`: dependency-free versions of + it measured +13-14% (four words per branch) and +9-11% (eight) on the metadata paths on + x86, so it stays. Context: upstream declined the `memchr` change as an added dependency + (staktrace/mailparse#142); a dependency-free version of both loops is offered instead + (staktrace/mailparse#143), and `vendor/mailparse/PATCH.md` records what switching to it + would cost if a release carries it. - **Rust toolchain pin moved from 1.97.1 to 1.98.0** (#120). The pin was a holding action against a measured 15-30% slowdown under 1.98.0; the cause turned out to be the two mailparse loops above -- the compiler had not changed their instructions, diff --git a/CONTRIBUTING.md b/CONTRIBUTING.md index bcff589..429ecb4 100644 --- a/CONTRIBUTING.md +++ b/CONTRIBUTING.md @@ -314,22 +314,23 @@ runners' Zen CPUs a scalar loop straddling the wrong 64-byte boundary falls out of the micro-op cache and runs at half speed. An Apple M4 measured the same two builds at +/-0.2%. -The fix replaces that scan with a word-at-a-time search -- 64 bytes per branch, -so even a placement-induced halving would cost a fraction of what the byte loop -did -- via the patched copy in `vendor/mailparse` (`src/bytescan.rs`; upstream -declined a `memchr` version as an added dependency, and this dependency-free one -is proposed in its place -- `vendor/mailparse/PATCH.md` has the sync and removal -procedures). Metadata mode went 0.365 -> 0.030 ms and the full parse -1.10 -> 0.76 ms. +The fix replaces that scan with `memchr::memmem` -- vectorised, and laid out +independently of this crate -- via the patched copy in `vendor/mailparse` +(upstream declined the change as an added dependency and is offered a +dependency-free version instead; `vendor/mailparse/PATCH.md` has the reasoning, +the sync procedure and what switching would cost). Metadata mode went +0.365 -> 0.030 ms and the full parse 1.10 -> 0.76 ms. With that gone, sampling the *full* parse put **77.7%** of what remained in `decode_base64`'s whitespace filter -- `iter().filter().cloned().collect()`, a test and a push per byte -- and the toolchain A/B confirmed it carried the rest of the placement sensitivity: decoding paths still moved +22% on one runner and +5% -on another while the metadata paths had gone flat. Same fix, same place: words -with no byte below `0x21` are skipped whole and the runs between whitespace are -copied in one piece. The full parse went 0.83 -> 0.28 ms on top, and 0.26 -> 0.24 -again when the `memchr` version gave way to the dependency-free one. +on another while the metadata paths had gone flat. Same place, and here the fix +is dependency-free (`vendor/mailparse/src/bytescan.rs`): a word with no byte below +`0x21` is skipped whole, the mask of one that might says which bytes to check, +and the runs between whitespace are copied in one piece. The full parse went +0.83 -> 0.28 ms on top, and 0.25 -> 0.23 again when this one-pass version replaced +the `memchr` two-search one. **What remains true.** The gate has a false-positive mode its noise floor cannot see: the controls are pure Python and do not care how the extension was laid out, diff --git a/Cargo.lock b/Cargo.lock index 00becaa..2df0a24 100644 --- a/Cargo.lock +++ b/Cargo.lock @@ -66,9 +66,16 @@ version = "0.16.1" dependencies = [ "charset", "data-encoding", + "memchr", "quoted_printable", ] +[[package]] +name = "memchr" +version = "2.8.3" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "cf8baf1c55e62ffcace7a9f06f4bd9cd3f0c4beb022d3b367256b91b87513d98" + [[package]] name = "once_cell" version = "1.21.4" diff --git a/vendor/mailparse/Cargo.toml b/vendor/mailparse/Cargo.toml index 60f4e0c..16c5663 100644 --- a/vendor/mailparse/Cargo.toml +++ b/vendor/mailparse/Cargo.toml @@ -37,6 +37,9 @@ categories = [ license = "0BSD" repository = "https://github.com/staktrace/mailparse" +[dependencies.memchr] +version = "2.7.0" + [dependencies.charset] version = "0.1.3" diff --git a/vendor/mailparse/PATCH.md b/vendor/mailparse/PATCH.md index e17581f..ccee9d3 100644 --- a/vendor/mailparse/PATCH.md +++ b/vendor/mailparse/PATCH.md @@ -1,34 +1,32 @@ # Patched copy of `mailparse` 0.16.1 This directory is [mailparse 0.16.1](https://crates.io/crates/mailparse/0.16.1) as published, -with **one module added and two functions changed to use it**. No dependency is added. It -is applied through `[patch.crates-io]` in the root `Cargo.toml` (and `fuzz/Cargo.toml`), so +with **two functions changed** (one via a new module) and one dependency added. It is +applied through `[patch.crates-io]` in the root `Cargo.toml` (and `fuzz/Cargo.toml`), so `cargo` sees the same crate name and version and every other dependency resolves exactly -as before. +as before. `memchr = "2.7.0"` is added to this crate's `[dependencies]` (MIT OR Unlicense, +no dependencies of its own). -## The change +## The changes -`src/bytescan.rs` (new) scans a machine word at a time instead of a byte at a time, in -plain `std` with no `unsafe`: `usize` chunks via `chunks_exact` + `from_le_bytes`, the -exact `haszero` / `hasless` word tests (Anderson, *Bit Twiddling Hacks*) to decide whether -a word needs a closer look, and `trailing_zeros` to jump to a hit rather than rescan. +Both replace a byte-at-a-time loop over the whole message body. Both return exactly what +the code they replace returned. 1. **`find_from_u8` in `src/lib.rs`** -- the search `parse_mail` runs for every MIME - boundary -- scanned byte by byte. It now calls `bytescan::find`, which tests eight words - -- a cache line -- per branch for the key's first byte and compares the rest only at - candidates. Same - result: first occurrence of `key` at or after `ix_start`, `None` when there is none. + boundary -- scanned byte by byte. It now calls `memchr::memmem::find`. Same result: + first occurrence of `key` at or after `ix_start`, `None` when there is none. 2. **`decode_base64` in `src/body.rs`** stripped whitespace with `iter().filter(|c| !c.is_ascii_whitespace()).cloned().collect()`: a test and a - bounds-checked push per byte. It now calls `bytescan::strip_ascii_whitespace`, which - skips any word with no byte below `0x21`, walks the mask bits of a word that has one, - re-checks each candidate with `is_ascii_whitespace` (so `0x0B` and the other control - bytes are kept, as before), and copies the runs between in one piece. + bounds-checked push per byte. It now calls `bytescan::strip_ascii_whitespace` + (`src/bytescan.rs`, new, plain `std`, no `unsafe`): a word with no byte below `0x21` + cannot contain whitespace and is skipped whole; for one that might, the exact `hasless` + mask (Anderson, *Bit Twiddling Hacks*) says which bytes to look at, each re-checked with + `is_ascii_whitespace` so `0x0B` and the other control bytes are kept, as before; runs + are copied in one piece. Its tests compare it against the filter it replaces over a + generated corpus at every alignment and over every byte value. -`bytescan::tests` compares both against the loops they replace over a generated corpus at -every alignment and over every byte value. `diff -r` against the registry copy shows -exactly `src/bytescan.rs`, `src/lib.rs` (a `mod` line and one function), `src/body.rs` -(one function) and this file. +`diff -r` against the registry copy shows exactly `src/bytescan.rs`, `src/lib.rs` (a `mod` +line and one function), `src/body.rs` (one function), `Cargo.toml` and this file. ## Why @@ -45,31 +43,44 @@ Interleaved A/B on an Apple M4, original master to this copy: | benchmark | before | after | |---|---|---| | `parse_email(mode="metadata")` | 0.365 ms | 0.030 ms | -| `parse_email` (full) | 1.094 ms | 0.240 ms | -| `parse_many` (8 x 767 KiB) | 9.082 ms | 1.973 ms | +| `parse_email` (full) | 1.094 ms | 0.228 ms | +| `parse_many` (8 x 767 KiB) | 9.082 ms | 1.834 ms | -A first version used `memchr` (PRs #213, #214). Upstream declined it for adding a -dependency, so this dependency-free version replaced it (it measured within noise on the -metadata paths and 3-11% faster on the decoding paths, because the strip is one pass). +## Why the two functions use different tools -## Upstream, and when this goes away +Upstream declined a version of this change that used `memchr` for both (staktrace/mailparse#142: +no new external dependencies), so a dependency-free word-at-a-time version of both was +written and is what upstream is now offered +([staktrace/mailparse#143](https://github.com/staktrace/mailparse/pull/143)). This copy +takes the half of it that is strictly no slower here: -The same change is proposed upstream as -[staktrace/mailparse#142](https://github.com/staktrace/mailparse/pull/142) (revised -2026-08-28 without the dependency). If a mailparse release includes it: +- The **strip** is the dependency-free version. It is one pass where the `memchr` version + made two searches per run, and measured faster on both CPUs tried (Apple M4: -9 to -12% + on the decoding paths; EPYC 7763 on the CI gate: -2.5 to -2.9%). +- The **byte search stays on `memchr`**. Three dependency-free variants went through the + gate: four words per branch measured +13-14% on the metadata paths on an EPYC 7763, eight + words (a cache line) +9-11% -- 6 us per 767 KiB. A word-at-a-time scan tops out below a + 32-byte AVX2 compare, and the rule for this copy is no degradation. + +So if a mailparse release ever includes #143, switching this copy's search to it (and +dropping the directory) is a decision that costs about 10% on `mode="metadata"` on x86 and +nothing on the decoding paths. The removal steps for that case: 1. bump `mailparse` in the root `Cargo.toml` and `fuzz/Cargo.toml` to that release, 2. delete the two `[patch.crates-io]` sections and this directory, 3. drop the `vendor/mailparse/**/*` entry from `[tool.maturin] include` in `pyproject.toml` and the `vendored mailparse tests` step from the lint job, -4. run the benchmark gate: the numbers should not move. +4. run the benchmark gate and read the metadata rows with the number above in mind. + +## Keeping this in sync Until then, each upstream mailparse release is a hand-merge into this copy: 1. `diff -r` the new release against the previous one (both under `~/.cargo/registry/src/*/mailparse-/`) and apply that diff here -- not the other way round, or the two functions revert; -2. keep `src/bytescan.rs`, the `mod bytescan;` line and the two call sites; +2. keep `src/bytescan.rs`, the `mod bytescan;` line, the two call sites and `memchr` in + `Cargo.toml`; 3. bump the version in this copy's `Cargo.toml` and the `mailparse = "..."` requirement in both root manifests together, since `[patch]` only applies when the patched version satisfies the requirement; diff --git a/vendor/mailparse/src/bytescan.rs b/vendor/mailparse/src/bytescan.rs index 1c76285..87f2fc6 100644 --- a/vendor/mailparse/src/bytescan.rs +++ b/vendor/mailparse/src/bytescan.rs @@ -1,8 +1,8 @@ //! Byte scans that look at a machine word at a time instead of a byte at a time. //! -//! Two loops in the parse path used to test every byte of the message body on its -//! own: the search for the next MIME boundary, and the whitespace strip before -//! base64 decoding. On a message with large attachments they are most of the parse, +//! The whitespace strip before base64 decoding used to test every byte of the body on +//! its own. (In this copy the MIME boundary search uses `memchr` instead -- see +//! `find_from_u8` and PATCH.md; upstream is offered a word-at-a-time version of both.) On a message with large attachments they are most of the parse, //! and a byte-at-a-time loop's speed also depends on where the linker happens to //! place it -- the same instructions ran at half speed on x86-64 when the loop //! straddled a 64-byte boundary. @@ -11,12 +11,10 @@ //! `from_ne_bytes` (plain loads; no `unsafe`) and use two classic word tricks to //! decide whether a chunk needs a closer look: //! -//! - `zero_byte_mask(x)`: non-zero iff some byte of `x` is zero. //! - `below_mask(x, n)`: non-zero iff some byte of `x` is below `n`, exact for `n <= 128`. //! //! Where a mask is non-zero, its set bits say which bytes to look at, so a hit costs a -//! `trailing_zeros` rather than a rescan of the word. The byte search tests eight words -//! per branch; the whitespace strip, whose hits are frequent (every line), one. +//! `trailing_zeros` rather than a rescan of the word. //! //! Both are exact, so a chunk is only examined byte by byte when it really contains //! a candidate. See Anderson, "Bit Twiddling Hacks", `haszero` and `hasless`. @@ -29,14 +27,6 @@ const WORD: usize = core::mem::size_of::(); const LO: usize = usize::from_ne_bytes([0x01; WORD]); const HI: usize = usize::from_ne_bytes([0x80; WORD]); -/// Each byte of the result has its high bit set iff that byte of `x` is zero -- exactly -/// so for the lowest such byte; bytes above it may be marked spuriously by the borrow, -/// which is why callers re-check when they walk more than the first hit. -#[inline] -fn zero_byte_mask(x: usize) -> usize { - x.wrapping_sub(LO) & !x & HI -} - /// Each byte of the result has its high bit set iff that byte of `x` is below `n` /// (`n <= 128`), with the same caveat as `zero_byte_mask` above the lowest hit: never a /// false negative, possibly a false positive, so callers re-check. @@ -62,58 +52,6 @@ fn first_hit(mask: usize) -> usize { (mask.trailing_zeros() / 8) as usize } -/// Words per step of the byte search: the masks are OR-ed, so the no-hit case is one -/// branch per 64 bytes -- a cache line -- and the loads and tests are independent work -/// the CPU can overlap. -const STEP_WORDS: usize = 8; -const STEP: usize = STEP_WORDS * WORD; - -/// Index of the first `needle` in `haystack`. -pub(crate) fn find_byte(haystack: &[u8], needle: u8) -> Option { - let repeated = LO.wrapping_mul(needle as usize); - let mut steps = haystack.chunks_exact(STEP); - let mut offset = 0; - for step in &mut steps { - let mut masks = [0usize; STEP_WORDS]; - let mut any = 0; - for (k, m) in masks.iter_mut().enumerate() { - *m = zero_byte_mask(word(&step[k * WORD..(k + 1) * WORD]) ^ repeated); - any |= *m; - } - if any != 0 { - let (k, m) = masks - .iter() - .copied() - .enumerate() - .find(|&(_, m)| m != 0) - .unwrap(); - return Some(offset + k * WORD + first_hit(m)); - } - offset += STEP; - } - steps - .remainder() - .iter() - .position(|&b| b == needle) - .map(|i| offset + i) -} - -/// Index of the first occurrence of `key` in `haystack`, or `None`. -/// -/// Scans for `key[0]` a word at a time and compares the rest only at candidates. -pub(crate) fn find(haystack: &[u8], key: &[u8]) -> Option { - let first = *key.first()?; - let mut at = 0; - while let Some(i) = find_byte(&haystack[at..], first) { - let candidate = at + i; - if haystack[candidate..].starts_with(key) { - return Some(candidate); - } - at = candidate + 1; - } - None -} - /// `body` without its ASCII whitespace -- exactly the bytes `u8::is_ascii_whitespace` /// names: space, tab, line feed, form feed, carriage return. /// @@ -153,14 +91,7 @@ pub(crate) fn strip_ascii_whitespace(body: &[u8]) -> Vec { #[cfg(test)] mod tests { - use super::{find, find_byte, strip_ascii_whitespace}; - - fn naive_find(haystack: &[u8], key: &[u8]) -> Option { - if key.is_empty() || haystack.len() < key.len() { - return None; - } - (0..=haystack.len() - key.len()).find(|&i| &haystack[i..i + key.len()] == key) - } + use super::strip_ascii_whitespace; fn naive_strip(body: &[u8]) -> Vec { body.iter() @@ -190,43 +121,6 @@ mod tests { out } - #[test] - fn find_byte_matches_position() { - for h in corpus() { - for needle in [b'-', b'\n', b'a', 0x00, 0xff, b'z'] { - assert_eq!( - find_byte(&h, needle), - h.iter().position(|&b| b == needle), - "{:?} / {:?}", - h, - needle - ); - } - } - } - - #[test] - fn find_matches_naive_search() { - let keys: &[&[u8]] = &[ - b"-", - b"--", - b"--ab", - b"\n", - b"\r\n", - b"ab=", - b"zz", - b"\x00\xff", - ]; - for h in corpus() { - for key in keys { - assert_eq!(find(&h, key), naive_find(&h, key), "{:?} / {:?}", h, key); - } - } - assert_eq!(find(b"abc", b""), None); - assert_eq!(find(b"", b"a"), None); - assert_eq!(find(b"ab", b"abc"), None); - } - #[test] fn strip_matches_filter() { for h in corpus() { diff --git a/vendor/mailparse/src/lib.rs b/vendor/mailparse/src/lib.rs index 9554119..3d02258 100644 --- a/vendor/mailparse/src/lib.rs +++ b/vendor/mailparse/src/lib.rs @@ -122,7 +122,11 @@ pub(crate) fn find_from(line: &str, ix_start: usize, key: &str) -> Option fn find_from_u8(line: &[u8], ix_start: usize, key: &[u8]) -> Option { assert!(!key.is_empty()); assert!(ix_start <= line.len()); - bytescan::find(&line[ix_start..], key).map(|v| ix_start + v) + // `memchr` rather than `bytescan::find`: on x86-64 its AVX2 path is ~10% ahead of a + // word-at-a-time scan on the structure-only parse (measured on the CI gate), and this + // copy's rule is no degradation. The dependency-free `bytescan` version of this search + // is what upstream is offered; see PATCH.md. + memchr::memmem::find(&line[ix_start..], key).map(|v| ix_start + v) } #[test]