Skip to content
Open
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
7 changes: 7 additions & 0 deletions fuzz/Cargo.toml
Original file line number Diff line number Diff line change
Expand Up @@ -45,6 +45,13 @@ test = false
doc = false
bench = false

[[bin]]
name = "fuzz_batch_records"
path = "fuzz_targets/fuzz_batch_records.rs"
test = false
doc = false
bench = false

[[bin]]
name = "fuzz_varint"
path = "fuzz_targets/fuzz_varint.rs"
Expand Down
85 changes: 85 additions & 0 deletions fuzz/fuzz_targets/fuzz_batch_records.rs
Original file line number Diff line number Diff line change
@@ -0,0 +1,85 @@
// Copyright ⓒ 2024-2026 Peter Morgan <peter.james.morgan@gmail.com>
//
// Licensed under the Apache License, Version 2.0 (the "License");
// you may not use this file except in compliance with the License.
// You may obtain a copy of the License at
//
// http://www.apache.org/licenses/LICENSE-2.0
//
// Unless required by applicable law or agreed to in writing, software
// distributed under the License is distributed on an "AS IS" BASIS,
// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
// See the License for the specific language governing permissions and
// limitations under the License.

//! Decodes the records of a batch for every compression codec.
//!
//! Input layout, so the fuzzer spends its time in the decompressors and the
//! record decoder rather than rediscovering batch header fields:
//!
//! | bytes | meaning |
//! |------------|--------------------------------------------------|
//! | `[0] % 5` | codec: none, gzip, snappy, lz4, zstd |
//! | `[1..3]` | `record_count`, big-endian `u16` |
//! | `[3..]` | `record_data`, the (possibly compressed) records |

#![no_main]
use bytes::Bytes;
use libfuzzer_sys::fuzz_target;
use nisshi_sans_io::{
Compression,
record::{Record, deflated, inflated},
};

const CODECS: [Compression; 5] = [
Compression::None,
Compression::Gzip,
Compression::Snappy,
Compression::Lz4,
Compression::Zstd,
];

fuzz_target!(|data: &[u8]| {
let [selector, count_hi, count_lo, record_data @ ..] = data else {
return;
};

let compression = CODECS[usize::from(*selector) % CODECS.len()].clone();
let uncompressed = matches!(compression, Compression::None);

let batch = deflated::Batch {
attributes: i16::from(compression),
record_count: u32::from(u16::from_be_bytes([*count_hi, *count_lo])),
record_data: Bytes::copy_from_slice(record_data),
..Default::default()
};

// The by-reference and by-value conversions are separate
// implementations (they differ for uncompressed batches), and the
// storage backends use both, so exercise each.
let by_reference = Vec::<Record>::try_from(&batch);
let by_value = Vec::<Record>::try_from(batch.clone());

// Differential oracle: the two must agree on whether the batch decodes
// and, when it does, on the records. Error variants are not compared.
//
// One known difference is allowed. For an uncompressed batch, the
// by-value conversion rejects bytes left after the last record, and the
// by-reference conversion accepts them. Without this exception the
// fuzzer reports it within seconds and finds nothing else. Compressed
// batches share one implementation, so they must agree. Remove the
// exception once the by-reference conversion rejects trailing bytes.
match (&by_reference, &by_value) {
(Ok(by_reference), Ok(by_value)) => assert_eq!(by_reference, by_value),
(Err(_), Err(_)) => {}
(Ok(_), Err(_)) if uncompressed => {}
_ => panic!(
"by-reference and by-value decoding disagree: {:?} vs {:?}",
by_reference.as_ref().map(Vec::len),
by_value.as_ref().map(Vec::len),
),
}

let _ = inflated::Batch::try_from(&batch);
let _ = inflated::Batch::try_from(batch);
});
8 changes: 6 additions & 2 deletions fuzz/fuzz_targets/fuzz_deflated_batch.rs
Original file line number Diff line number Diff line change
Expand Up @@ -15,8 +15,12 @@
#![no_main]
use bytes::Bytes;
use libfuzzer_sys::fuzz_target;
use nisshi_sans_io::record::deflated;
use nisshi_sans_io::record::{Record, deflated};

fuzz_target!(|data: &[u8]| {
let _ = deflated::Batch::try_from(Bytes::copy_from_slice(data));
// A parsed header carries the codec in its attribute bits, so this also
// drives the record decoder end to end from the wire bytes.
if let Ok(batch) = deflated::Batch::try_from(Bytes::copy_from_slice(data)) {
let _ = Vec::<Record>::try_from(&batch);
}
});
96 changes: 95 additions & 1 deletion fuzz/fuzz_targets/generate_seeds.rs
Original file line number Diff line number Diff line change
Expand Up @@ -14,13 +14,19 @@

//! Seed corpus generator for fuzz targets.
//!
//! Run with: cargo +nightly run --manifest-path fuzz/Cargo.toml --bin generate_seeds
//! Run with: just fuzz-generate-seed (from the repository root)
//!
//! Writes binary seed files into fuzz/corpus/{target}/ directories.

use std::fs;
use std::path::Path;

use bytes::Bytes;
use nisshi_sans_io::{
Compression,
record::{Header, Record, deflated, inflated},
};

fn write_seed(dir: &str, name: &str, data: &[u8]) {
let path = Path::new(dir).join(name);
fs::create_dir_all(dir).unwrap();
Expand Down Expand Up @@ -801,5 +807,93 @@ fn main() {
// Random multi-byte
write_seed(dir, "random_multi", &[0xD2, 0x85, 0xD8, 0xCC, 0x04]);

// ── fuzz_batch_records seeds ──────────────────────────────────────
//
// Layout: codec selector, big-endian u16 record count, record data
// (see fuzz_batch_records.rs).

let dir = "fuzz/corpus/fuzz_batch_records";

for (selector, name, compression) in [
(0u8, "none", Compression::None),
(1, "gzip", Compression::Gzip),
(2, "snappy", Compression::Snappy),
(3, "lz4", Compression::Lz4),
(4, "zstd", Compression::Zstd),
] {
let batch = encoded_batch(compression);
let record_count = u16::try_from(batch.record_count).unwrap();
write_seed(
dir,
name,
&batch_records_seed(selector, record_count, &batch.record_data),
);
}

// Snappy with xerial (snappy-java) framing: the 8-byte magic, then
// version, compatible version and the first chunk's length (4 bytes
// each), then that raw snappy chunk.
let xerial_magic = b"\x82SNAPPY\0";
{
let batch = encoded_batch(Compression::Snappy);
let mut framed = xerial_magic.to_vec();
framed.extend(1i32.to_be_bytes());
framed.extend(1i32.to_be_bytes());
framed.extend(
i32::try_from(batch.record_data.len())
.unwrap()
.to_be_bytes(),
);
framed.extend_from_slice(&batch.record_data);

let record_count = u16::try_from(batch.record_count).unwrap();
write_seed(
dir,
"snappy_xerial",
&batch_records_seed(2, record_count, &framed),
);
}

// The xerial magic followed by fewer than the 12 header bytes.
for short in 0..12 {
let mut truncated = xerial_magic.to_vec();
truncated.extend(std::iter::repeat_n(0u8, short));
write_seed(
dir,
&format!("snappy_xerial_truncated_{short}"),
&batch_records_seed(2, 1, &truncated),
);
}

println!("\nDone! Seed corpus generated.");
}

/// A small batch of records, encoded with `compression`.
fn encoded_batch(compression: Compression) -> deflated::Batch {
let mut builder = inflated::Batch::builder().attributes(i16::from(compression));

for offset_delta in 0..3 {
builder = builder.record(
Record::builder()
.offset_delta(offset_delta)
.key(Some(Bytes::from(format!("key-{offset_delta}"))))
.value(Some(Bytes::from(
"the quick brown fox jumps over the lazy dog",
)))
.header(
Header::builder()
.key(Bytes::from("header"))
.value(Bytes::from("value")),
),
);
}

builder.build().and_then(deflated::Batch::try_from).unwrap()
}

fn batch_records_seed(selector: u8, record_count: u16, record_data: &[u8]) -> Vec<u8> {
let mut seed = vec![selector];
seed.extend(record_count.to_be_bytes());
seed.extend_from_slice(record_data);
seed
}
5 changes: 4 additions & 1 deletion justfile
Original file line number Diff line number Diff line change
Expand Up @@ -92,7 +92,10 @@ fuzz-request-decode: (cargo-fuzz "run" "fuzz_request_decode" "--" "-max_total_ti

fuzz-member-metadata: (cargo-fuzz "run" "fuzz_member_metadata" "--" "-max_total_time=60")

fuzz-generate-seed: (cargo-fuzz "run" "--package" "fuzz" "--bin" "generate_seeds")
fuzz-batch-records: (cargo-fuzz "run" "fuzz_batch_records" "--" "-max_total_time=60")

fuzz-generate-seed:
cargo run --package fuzz --bin generate_seeds

check:
cargo check --workspace --all-features --all-targets
Expand Down
56 changes: 48 additions & 8 deletions nisshi-sans-io/src/primitive/varint.rs
Original file line number Diff line number Diff line change
Expand Up @@ -154,12 +154,21 @@ impl VarInt {
.next_element::<u8>()?
.ok_or_else(|| de::Error::custom("u8"))?;

let overflow = || de::Error::custom("overflow");

if byte & CONTINUATION == CONTINUATION {
let intermediate = u32::from(byte & MASK);
accumulator += intermediate << shift;
shift += 7;
accumulator = u32::from(byte & MASK)
.checked_shl(u32::from(shift))
.and_then(|intermediate| accumulator.checked_add(intermediate))
.ok_or_else(overflow)?;

shift = shift.checked_add(7).ok_or_else(overflow)?;
} else {
accumulator += u32::from(byte) << shift;
accumulator = u32::from(byte)
.checked_shl(u32::from(shift))
.and_then(|intermediate| accumulator.checked_add(intermediate))
.ok_or_else(overflow)?;

done = true;
}
}
Expand Down Expand Up @@ -349,12 +358,21 @@ impl LongVarInt {
.next_element::<u8>()?
.ok_or_else(|| de::Error::custom("u8"))?;

let overflow = || de::Error::custom("overflow");

if byte & CONTINUATION == CONTINUATION {
let intermediate = u64::from(byte & MASK);
accumulator += intermediate << shift;
shift += 7;
accumulator = u64::from(byte & MASK)
.checked_shl(u32::from(shift))
.and_then(|intermediate| accumulator.checked_add(intermediate))
.ok_or_else(overflow)?;

shift = shift.checked_add(7).ok_or_else(overflow)?;
} else {
accumulator += u64::from(byte) << shift;
accumulator = u64::from(byte)
.checked_shl(u32::from(shift))
.and_then(|intermediate| accumulator.checked_add(intermediate))
.ok_or_else(overflow)?;

done = true;
}
}
Expand Down Expand Up @@ -636,6 +654,28 @@ mod tests {
// Ok(())
// }

// Untrusted record data: a varint with more continuation bytes than its
// type can hold must be an error, not an arithmetic overflow panic.
fn overlong(continuation_bytes: usize) -> std::io::Cursor<Vec<u8>> {
let mut encoded = vec![0xffu8; continuation_bytes];
encoded.push(0x01);
std::io::Cursor::new(encoded)
}

#[test]
fn deserialize_overlong_varint_is_an_error() {
let mut reader = overlong(6);
let mut decoder = crate::de::Decoder::new(&mut reader);
assert!(VarInt::deserialize(&mut decoder).is_err());
}

#[test]
fn deserialize_overlong_long_varint_is_an_error() {
let mut reader = overlong(11);
let mut decoder = crate::de::Decoder::new(&mut reader);
assert!(LongVarInt::deserialize(&mut decoder).is_err());
}

#[test]
fn encode_decode() -> Result<()> {
let expected = 1;
Expand Down
Loading