diff --git a/.bazelrc b/.bazelrc index c6e32411..9c235b52 100644 --- a/.bazelrc +++ b/.bazelrc @@ -26,6 +26,12 @@ build:ubsan --linkopt=-fsanitize=undefined build:ubsan --linkopt=-fno-sanitize-recover=all build:ubsan --linkopt=-fsanitize-link-c++-runtime +# libFuzzer: CC=clang bazel build --config=fuzz //fuzz:json_decode_fuzz +build:fuzz --copt=-fsanitize=fuzzer,address +build:fuzz --linkopt=-fsanitize=fuzzer,address +build:fuzz --strip=never +build:fuzz --copt=-g + # CI settings. build:ci --announce_rc --verbose_failures diff --git a/.github/workflows/ci.yml b/.github/workflows/ci.yml index 6cd5201f..67c5726c 100644 --- a/.github/workflows/ci.yml +++ b/.github/workflows/ci.yml @@ -47,6 +47,31 @@ jobs: CXX: clang++ run: bazelisk test //... --config=ci --config=asan --config=ubsan + fuzz: + name: fuzz (libFuzzer smoke) + runs-on: ubuntu-24.04 + steps: + - uses: actions/checkout@v4 + # Each harness runs a short bounded libFuzzer session (a "smoke" run, not + # a soak) so regressions in the parsers surface on every PR. The + # deterministic-driver variants also run as ordinary tests in the bazel + # matrix above; this job additionally exercises the real fuzzer runtime. + # Build only the libFuzzer binaries — not //fuzz:all, which would also + # pull in the *_fuzz_smoke cc_tests. Those link their own main() (the + # deterministic driver) and would clash with the libFuzzer main() that + # --config=fuzz injects. The smoke tests run in the bazel matrix above. + - name: build and run each harness (30s) + env: + CC: clang + CXX: clang++ + run: | + set -euo pipefail + for target in json_decode cbor_decode uri server_dispatch; do + echo "== fuzzing $target" + bazelisk build --config=fuzz "//fuzz:${target}_fuzz" + ./bazel-bin/fuzz/${target}_fuzz -max_total_time=30 -print_final_stats=1 + done + consumer: name: bazel consumer (${{ matrix.os }}) strategy: diff --git a/docs/fuzzing.md b/docs/fuzzing.md new file mode 100644 index 00000000..3332737f --- /dev/null +++ b/docs/fuzzing.md @@ -0,0 +1,41 @@ +# Fuzzing + +The parsers that face untrusted bytes — JSON decode, CBOR decode, URI +percent-decoding/encoding, and full generated-server request dispatch — have +libFuzzer harnesses under [`fuzz/`](../fuzz). Each is written once and built +two ways: + +- **Deterministic driver** (`*_fuzz_smoke` cc_tests): `fuzz_driver_main.cc` + feeds ~20k pseudo-random inputs (a fixed xorshift seed, half biased toward + printable text so parsers get past the first byte) through the same + `LLVMFuzzerTestOneInput`. These run in every CI job, including the + ASan/UBSan matrix — no fuzzer runtime required, fully reproducible. +- **Real libFuzzer** (`*_fuzz` cc_binaries, tagged `manual`): built with + `--config=fuzz` (clang `-fsanitize=fuzzer,address`). The `fuzz` CI job runs + each for 30 seconds on every PR; run longer locally to soak. + +```sh +# Reproducible smoke run (any toolchain): +bazel test //fuzz:all + +# Real fuzzing (clang required): +CC=clang CXX=clang++ bazel build --config=fuzz //fuzz:json_decode_fuzz +./bazel-bin/fuzz/json_decode_fuzz -max_total_time=300 # soak 5 min +./bazel-bin/fuzz/json_decode_fuzz path/to/crash-input # reproduce +``` + +## Invariants the harnesses enforce + +- Decoders never crash and never throw on any input; whatever they accept + must re-encode cleanly (`json`/`cbor` round-trip). +- `PercentDecode` accepts or rejects without crashing; accepted values + survive an encode → decode round trip; the encoders accept arbitrary bytes. +- Server dispatch always returns a response with a valid HTTP status for any + method/target/headers/body — the malformed-request suite, unbounded. + +New parsers should land with a harness here. + +## Corpus + +Harnesses run corpus-free today (libFuzzer generates from scratch). A seed +corpus and OSS-Fuzz integration are post-0.1.0 (PLAN Phase 7). diff --git a/fuzz/BUILD.bazel b/fuzz/BUILD.bazel new file mode 100644 index 00000000..f3bc5f19 --- /dev/null +++ b/fuzz/BUILD.bazel @@ -0,0 +1,42 @@ +load("@rules_cc//cc:defs.bzl", "cc_binary", "cc_test") + +# Each harness builds two ways: as an ordinary test through the deterministic +# driver (runs in every CI matrix job, including ASan/UBSan), and as a real +# libFuzzer binary under --config=fuzz (the dedicated CI fuzz job; `manual` +# keeps the binaries out of plain //... builds, which have no fuzzer runtime). + +FUZZ_TARGETS = { + "json_decode": ["//runtime:json"], + "cbor_decode": [ + "//runtime:cbor", + "//runtime:core", + ], + "uri": ["//runtime:http"], + "server_dispatch": [ + "//examples/weather/generated:server", + "//runtime:http", + ], +} + +[ + cc_test( + name = name + "_fuzz_smoke", + size = "small", + srcs = [ + "fuzz_driver_main.cc", + name + "_fuzz.cc", + ], + deps = deps, + ) + for name, deps in FUZZ_TARGETS.items() +] + +[ + cc_binary( + name = name + "_fuzz", + srcs = [name + "_fuzz.cc"], + tags = ["manual"], + deps = deps, + ) + for name, deps in FUZZ_TARGETS.items() +] diff --git a/fuzz/cbor_decode_fuzz.cc b/fuzz/cbor_decode_fuzz.cc new file mode 100644 index 00000000..24c14949 --- /dev/null +++ b/fuzz/cbor_decode_fuzz.cc @@ -0,0 +1,18 @@ +// Fuzz target: CBOR bytes -> Document -> CBOR bytes. Decode must never crash; +// anything it accepts must re-encode without throwing. +#include +#include +#include + +#include "smithy/cbor/cbor.h" +#include "smithy/core/blob.h" + +extern "C" int LLVMFuzzerTestOneInput(const std::uint8_t* data, std::size_t size) { + const auto blob = + smithy::Blob::FromString(std::string(reinterpret_cast(data), size)); + auto doc = smithy::cbor::Decode(blob); + if (doc.ok()) { + (void)smithy::cbor::Encode(*doc); + } + return 0; +} diff --git a/fuzz/fuzz_driver_main.cc b/fuzz/fuzz_driver_main.cc new file mode 100644 index 00000000..fbad88a2 --- /dev/null +++ b/fuzz/fuzz_driver_main.cc @@ -0,0 +1,30 @@ +// Standalone driver so every fuzz target also runs as an ordinary test: +// deterministic pseudo-random inputs (xorshift) of varied sizes, plus edge +// sizes. Real fuzzing links libFuzzer instead (bazel --config=fuzz). +#include +#include +#include + +extern "C" int LLVMFuzzerTestOneInput(const std::uint8_t* data, std::size_t size); + +int main() { + LLVMFuzzerTestOneInput(nullptr, 0); + std::uint64_t state = 0x9E3779B97F4A7C15ULL; + auto next = [&state] { + state ^= state << 13; + state ^= state >> 7; + state ^= state << 17; + return state; + }; + std::vector buffer; + for (int iteration = 0; iteration < 20000; ++iteration) { + buffer.resize(next() % 300); + for (auto& byte : buffer) byte = static_cast(next()); + // Bias toward text-ish inputs half the time so parsers get past byte one. + if (iteration % 2 == 0) { + for (auto& byte : buffer) byte = static_cast(' ' + byte % 95); + } + LLVMFuzzerTestOneInput(buffer.data(), buffer.size()); + } + return 0; +} diff --git a/fuzz/json_decode_fuzz.cc b/fuzz/json_decode_fuzz.cc new file mode 100644 index 00000000..15024895 --- /dev/null +++ b/fuzz/json_decode_fuzz.cc @@ -0,0 +1,16 @@ +// Fuzz target: JSON text -> Document -> JSON text. Decode must never crash; +// anything it accepts must re-encode without throwing. +#include +#include +#include + +#include "smithy/json/json.h" + +extern "C" int LLVMFuzzerTestOneInput(const std::uint8_t* data, std::size_t size) { + const std::string_view text(reinterpret_cast(data), size); + auto doc = smithy::json::Decode(text); + if (doc.ok()) { + (void)smithy::json::Encode(*doc); + } + return 0; +} diff --git a/fuzz/server_dispatch_fuzz.cc b/fuzz/server_dispatch_fuzz.cc new file mode 100644 index 00000000..5dcee2c4 --- /dev/null +++ b/fuzz/server_dispatch_fuzz.cc @@ -0,0 +1,71 @@ +// Fuzz target: full generated-server dispatch (router + bindings + serde + +// validation) fed raw method/target/headers/body. Must never crash and must +// always produce a response; the malformed-request corpus's job, unbounded. +#include +#include +#include +#include +#include + +#include "example/weather/server.h" +#include "smithy/http/message.h" + +namespace { + +class NullHandler final : public example::weather::WeatherHandler { + public: + smithy::Outcome GetCity( + const example::weather::GetCityInput&) override { + return example::weather::GetCityOutput{.name = "x"}; + } + smithy::Outcome DeleteCity( + const example::weather::DeleteCityInput&) override { + return example::weather::DeleteCityOutput{}; + } + smithy::Outcome ListCities( + const example::weather::ListCitiesInput&) override { + return example::weather::ListCitiesOutput{}; + } + smithy::Outcome GetForecast( + const example::weather::GetForecastInput&) override { + return example::weather::GetForecastOutput{}; + } + smithy::Outcome GetCurrentTime( + const example::weather::GetCurrentTimeInput&) override { + return example::weather::GetCurrentTimeOutput{}; + } + smithy::Outcome GetReport( + const example::weather::GetReportInput& input) override { + return example::weather::GetReportOutput{.path = input.reportPath, .sizeBytes = 0}; + } +}; + +smithy::http::RequestHandler& Handler() { + static auto* server = new example::weather::WeatherServer(std::make_shared()); + static auto* handler = new smithy::http::RequestHandler(server->Handler()); + return *handler; +} + +} // namespace + +extern "C" int LLVMFuzzerTestOneInput(const std::uint8_t* data, std::size_t size) { + // Layout: method '\n' target '\n' header-value '\n' body. + const std::string all(reinterpret_cast(data), size); + smithy::http::HttpRequest request; + std::size_t start = 0; + std::string* fields[] = {&request.method, &request.target}; + for (std::string* field : fields) { + const auto newline = all.find('\n', start); + if (newline == std::string::npos) break; + *field = all.substr(start, newline - start); + start = newline + 1; + } + const auto newline = all.find('\n', start); + if (newline != std::string::npos) { + request.headers.Set("content-type", all.substr(start, newline - start)); + request.body = all.substr(newline + 1); + } + const auto response = Handler()(request); + if (response.status < 100 || response.status > 599) std::abort(); + return 0; +} diff --git a/fuzz/uri_fuzz.cc b/fuzz/uri_fuzz.cc new file mode 100644 index 00000000..5c07a126 --- /dev/null +++ b/fuzz/uri_fuzz.cc @@ -0,0 +1,22 @@ +// Fuzz target: URI machinery. PercentDecode must never crash, and anything it +// accepts must survive an encode -> decode round trip; the encoders must +// accept arbitrary bytes. +#include +#include +#include +#include + +#include "smithy/http/uri.h" + +extern "C" int LLVMFuzzerTestOneInput(const std::uint8_t* data, std::size_t size) { + const std::string_view text(reinterpret_cast(data), size); + auto decoded = smithy::http::PercentDecode(text); + if (decoded.ok()) { + auto reencoded = smithy::http::EncodeQueryComponent(*decoded); + auto redecoded = smithy::http::PercentDecode(reencoded); + if (!redecoded.ok() || *redecoded != *decoded) std::abort(); + } + (void)smithy::http::EncodePathSegment(text); + (void)smithy::http::EncodeGreedyPathSegment(text); + return 0; +} diff --git a/runtime/src/cbor/cbor.cc b/runtime/src/cbor/cbor.cc index 3b98ecf7..20b0ed9c 100644 --- a/runtime/src/cbor/cbor.cc +++ b/runtime/src/cbor/cbor.cc @@ -180,7 +180,9 @@ class Decoder { Outcome DecodeChunkedString(Major major, bool indefinite, std::uint64_t length) { std::string out; if (!indefinite) { - if (pos_ + length > size_) return Fail("truncated string"); + // length is attacker-controlled and may be near 2^64: compare against + // the remaining bytes so the check cannot overflow and wrap. + if (length > size_ - pos_) return Fail("truncated string"); out.assign(reinterpret_cast(data_ + pos_), length); pos_ += length; return out; @@ -195,7 +197,7 @@ class Decoder { auto chunk_length = ReadArgument(initial & 0x1F, &chunk_indefinite); if (!chunk_length) return std::move(chunk_length).error(); if (chunk_indefinite) return Fail("nested indefinite string chunk"); - if (pos_ + *chunk_length > size_) return Fail("truncated string chunk"); + if (*chunk_length > size_ - pos_) return Fail("truncated string chunk"); out.append(reinterpret_cast(data_ + pos_), *chunk_length); pos_ += *chunk_length; } diff --git a/runtime/src/json/json.cc b/runtime/src/json/json.cc index cd5507ad..c35b03cb 100644 --- a/runtime/src/json/json.cc +++ b/runtime/src/json/json.cc @@ -88,7 +88,15 @@ Outcome FromBackend(const nlohmann::json& value) { } // namespace -std::string Encode(const Document& doc) { return ToBackend(doc).dump(); } +std::string Encode(const Document& doc) { + // error_handler_t::replace: substitute U+FFFD for invalid UTF-8 instead of + // throwing (nlohmann's default). Strings can carry raw bytes from @httpLabel + // segments, headers, or blobs echoed into a response; a strict dump would + // throw type_error.316 uncaught and terminate the server. Replacement keeps + // the output valid JSON and the process alive. + return ToBackend(doc).dump(/*indent=*/-1, /*indent_char=*/' ', + /*ensure_ascii=*/false, nlohmann::json::error_handler_t::replace); +} Outcome Decode(std::string_view text) { const nlohmann::json parsed = nlohmann::json::parse(text, /*cb=*/nullptr, diff --git a/runtime/tests/cbor/cbor_test.cc b/runtime/tests/cbor/cbor_test.cc index cb8a69b5..b256a489 100644 --- a/runtime/tests/cbor/cbor_test.cc +++ b/runtime/tests/cbor/cbor_test.cc @@ -175,6 +175,18 @@ TEST(CborTest, RejectsMalformedInput) { } } +// Regression (fuzzer-found): a text/byte string whose declared length is near +// 2^64 must be rejected as truncated, not fed to string::assign. The old +// bounds check `pos + length > size` overflowed and wrapped past the guard. +TEST(CborTest, RejectsHugeStringLengthWithoutOverflow) { + // 0x7b = text string, 8-byte length; 0xff...ff = 2^64 - 1 bytes claimed. + EXPECT_FALSE(Decode(FromHex("7bffffffffffffffff")).ok()); + // 0x5b = byte string, same trick. + EXPECT_FALSE(Decode(FromHex("5bffffffffffffffff")).ok()); + // Indefinite text string with a giant chunk length. + EXPECT_FALSE(Decode(FromHex("7f7bffffffffffffffff")).ok()); +} + TEST(CborTest, RejectsExcessiveNesting) { std::string hex; for (int i = 0; i < 100; ++i) hex += "81"; // 100 nested single-element arrays diff --git a/runtime/tests/json/json_test.cc b/runtime/tests/json/json_test.cc index d79a82b0..df4408ed 100644 --- a/runtime/tests/json/json_test.cc +++ b/runtime/tests/json/json_test.cc @@ -66,6 +66,27 @@ TEST(JsonTest, RejectsMalformedText) { } } +// Regression (fuzzer-found): a string carrying invalid UTF-8 — raw bytes can +// reach Encode via @httpLabel segments, headers, or blobs echoed into a +// response — must not throw (nlohmann's strict dump would terminate the +// server). Invalid sequences become U+FFFD and the output stays valid JSON +// that round-trips. +TEST(JsonTest, EncodeReplacesInvalidUtf8InsteadOfThrowing) { + for (const std::string raw : { + std::string("\xe1"), // lone 3-byte lead + std::string("\xff\xfe"), // never-valid bytes + std::string("a\x80\x80z"), // stray continuation bytes + std::string("ok\xc3"), // truncated 2-byte sequence + }) { + const std::string encoded = Encode(Document(raw)); + const auto reparsed = Decode(encoded); + ASSERT_TRUE(reparsed.ok()) << "did not round-trip: " << encoded; + EXPECT_TRUE(reparsed->is_string()); + } + // Valid UTF-8 (including multibyte) passes through untouched. + EXPECT_EQ(Encode(Document(std::string("caf\xc3\xa9"))), "\"caf\xc3\xa9\""); +} + TEST(JsonTest, DecodesNestedLists) { const auto doc = Decode(R"([1,[2,"three"],null])"); ASSERT_TRUE(doc.ok());