mirror of
https://github.com/troglobit/finit.git
synced 2026-10-01 05:22:48 +07:00
test: fuzz target for the message parser
__msg_parse() turns bytes off a socket into pointers, before anything has vouched for the peer, and it is the only place in libink that does. It had no test of its own beyond whatever the other tests happened to send it, all of it well-formed. The target checks the parser's contract, not merely that it survived. A header field must point into the header field array, and terminate inside it, and the parse must never claim more bytes than it was handed. Crash-only would pass a parser that walked into the body and returned fields from there, since those bytes were handed over too. The expected bounds are derived from the raw header rather than from the parser, so the two have to agree independently. Every input is copied into an allocation sized to it first. Reading past the end of a roomy buffer stays inside the allocation and the sanitizer never sees it; against an exact one the same read is a fault, which is where the sharpest findings come from. Under libFuzzer it is an ordinary fuzz target and named files replay, which is how a find gets reproduced. With no arguments it runs a fixed sweep -- every truncation, every single-byte corruption, every value of the length that decides where the header ends, and seeded garbage -- so the suite covers the same contract on every build, without clang or a corpus in the tree. It takes 40 ms. CI fuzzes it properly on every pull request, keeps the crashers, and carries the corpus between runs so it reaches deeper over time than any single run can. Note that clang links the fuzzer runtime against the newest GCC tree it finds, so the libstdc++ headers have to match that one and not the default compiler, which is worth saying since installing the obvious package leaves you exactly where you started. Signed-off-by: Joachim Wiberg <troglobit@gmail.com>
This commit is contained in:
@@ -18,6 +18,74 @@ concurrency:
|
||||
cancel-in-progress: true
|
||||
|
||||
jobs:
|
||||
fuzz:
|
||||
# The message parser is the only place bytes off a socket become
|
||||
# pointers, so give it a real fuzzer on every PR, not just the
|
||||
# fixed sweep 'make check' runs.
|
||||
name: fuzz msg-parse
|
||||
runs-on: ubuntu-latest
|
||||
if: github.event_name != 'push' || github.ref == 'refs/heads/master'
|
||||
steps:
|
||||
- name: Install dependencies
|
||||
run: |
|
||||
sudo apt-get -y update
|
||||
sudo apt-get -y install pkg-config libconfuse-dev clang
|
||||
# clang picks the newest gcc tree it finds and needs the
|
||||
# matching libstdc++ headers to link the fuzzer runtime
|
||||
sudo apt-get -y install libstdc++-14-dev || true
|
||||
wget https://github.com/troglobit/libuev/releases/download/v2.4.1/libuev-2.4.1.tar.xz
|
||||
wget https://github.com/troglobit/libite/releases/download/v2.6.2/libite-2.6.2.tar.gz
|
||||
tar xf libuev-2.4.1.tar.xz
|
||||
tar xf libite-2.6.2.tar.gz
|
||||
(cd libuev-2.4.1 && ./configure && make -j9 && sudo make install-strip)
|
||||
(cd libite-2.6.2 && ./configure && make -j9 && sudo make install-strip)
|
||||
sudo ldconfig
|
||||
- uses: actions/checkout@v4
|
||||
- name: Configure
|
||||
run: |
|
||||
./autogen.sh
|
||||
./configure --prefix=/usr --exec-prefix= --sysconfdir=/etc --localstatedir=/var
|
||||
- name: Build fuzz target
|
||||
run: |
|
||||
clang -fsanitize=fuzzer,address -DLINK_FUZZ_LIBFUZZER -D_GNU_SOURCE \
|
||||
-I libink -I . -o fuzz-msg-parse \
|
||||
test/src/fuzz-msg-parse.c libink/*.c
|
||||
# Restores the newest corpus and saves a fresh one, since a cache
|
||||
# entry is immutable once written. Caches made on a branch are
|
||||
# private to it, so the corpus that accumulates on master is what
|
||||
# pull requests start from, rather than nothing.
|
||||
- name: Restore corpus
|
||||
uses: actions/cache@v4
|
||||
with:
|
||||
path: .fuzz-corpus
|
||||
key: fuzz-corpus-${{ github.run_id }}
|
||||
restore-keys: fuzz-corpus-
|
||||
- name: Fuzz
|
||||
run: |
|
||||
mkdir -p .fuzz-corpus
|
||||
./fuzz-msg-parse .fuzz-corpus -max_total_time=120 -max_len=4096 \
|
||||
-print_final_stats=1
|
||||
# Without this the corpus only ever grows, and most of what it
|
||||
# accumulates reaches code some earlier input already reached.
|
||||
- name: Minimise corpus
|
||||
if: always()
|
||||
run: |
|
||||
mkdir -p .fuzz-corpus-min
|
||||
./fuzz-msg-parse -merge=1 .fuzz-corpus-min .fuzz-corpus
|
||||
rm -rf .fuzz-corpus
|
||||
mv .fuzz-corpus-min .fuzz-corpus
|
||||
echo "corpus: $(ls .fuzz-corpus | wc -l) inputs"
|
||||
- name: Upload crashers
|
||||
if: failure()
|
||||
uses: actions/upload-artifact@v4
|
||||
with:
|
||||
name: fuzz-crashers
|
||||
path: |
|
||||
crash-*
|
||||
leak-*
|
||||
timeout-*
|
||||
if-no-files-found: ignore
|
||||
|
||||
build:
|
||||
# Verify we can build on latest Ubuntu with both gcc and clang
|
||||
name: ${{ matrix.compiler }}
|
||||
|
||||
@@ -52,3 +52,8 @@ GTAGS
|
||||
|
||||
# VS Code user settings
|
||||
.vscode
|
||||
|
||||
# Fuzzing: the target is built by hand with clang, see doc/build.md,
|
||||
# and the corpus it grows is a local artefact, not something to ship
|
||||
/fuzz-msg-parse
|
||||
/.fuzz-corpus/
|
||||
|
||||
@@ -35,6 +35,13 @@ endif
|
||||
install-dev:
|
||||
@make -C src install-pkgincludeHEADERS
|
||||
|
||||
# The fuzz target is built by hand with clang (doc/build.md), so it is
|
||||
# not in any _PROGRAMS and nothing else would clear it or the corpus
|
||||
# it grows.
|
||||
distclean-local:
|
||||
-rm -f $(top_builddir)/fuzz-msg-parse
|
||||
-rm -rf $(top_builddir)/.fuzz-corpus
|
||||
|
||||
# Target to run when building a release
|
||||
release: distcheck
|
||||
@for file in $(DIST_ARCHIVES); do \
|
||||
|
||||
@@ -126,6 +126,60 @@ Linux config to:
|
||||
> `/etc/fstab`.
|
||||
|
||||
|
||||
Testing
|
||||
-------
|
||||
|
||||
`make check` runs the test suite in a private namespace, so it is safe
|
||||
on a running system. It needs `unshare` and, on Ubuntu, unprivileged
|
||||
user namespaces enabled:
|
||||
|
||||
```shell
|
||||
sudo sysctl kernel.apparmor_restrict_unprivileged_userns=0
|
||||
make check
|
||||
```
|
||||
|
||||
The D-Bus message parser is the only place in Finit where bytes off a
|
||||
socket become pointers, so it also has a fuzz target. `make check`
|
||||
runs it as a fixed sweep, which takes milliseconds and needs nothing
|
||||
beyond the normal build. To fuzz it properly, build it with clang and
|
||||
libFuzzer:
|
||||
|
||||
```shell
|
||||
clang -fsanitize=fuzzer,address -DLINK_FUZZ_LIBFUZZER -D_GNU_SOURCE \
|
||||
-I libink -I . -o fuzz-msg-parse \
|
||||
test/src/fuzz-msg-parse.c libink/*.c
|
||||
mkdir -p .fuzz-corpus
|
||||
./fuzz-msg-parse .fuzz-corpus -max_total_time=300
|
||||
```
|
||||
|
||||
Give it a corpus directory as above and it saves what it learns there,
|
||||
so the next run picks up where this one left off instead of starting
|
||||
cold. Nothing writes to it unless you name it: without the argument
|
||||
libFuzzer keeps everything in memory and the run leaves only crashes
|
||||
behind. `make distclean` clears the corpus and the target.
|
||||
|
||||
Once a corpus has grown, most of it reaches code some earlier input
|
||||
already reached. Minimise it:
|
||||
|
||||
```shell
|
||||
./fuzz-msg-parse -merge=1 .fuzz-corpus-min .fuzz-corpus
|
||||
rm -rf .fuzz-corpus && mv .fuzz-corpus-min .fuzz-corpus
|
||||
```
|
||||
|
||||
Run `./configure` first, the target needs the generated `config.h`.
|
||||
If the link fails with `cannot find -lstdc++`, install the `libstdc++`
|
||||
headers matching the newest GCC on the system, not the default one:
|
||||
clang picks the newest tree it finds, and that is the one that needs
|
||||
them. CI runs this target on every pull request, carrying its
|
||||
corpus between runs so it reaches deeper over time.
|
||||
|
||||
Feed a file back to the target to reproduce a find:
|
||||
|
||||
```shell
|
||||
./fuzz-msg-parse crash-3f2a...
|
||||
```
|
||||
|
||||
|
||||
Running
|
||||
-------
|
||||
|
||||
|
||||
@@ -75,6 +75,7 @@ EXTRA_DIST += unexpected-restart.sh
|
||||
EXTRA_DIST += dbus-auth.sh
|
||||
EXTRA_DIST += dbus-authz.sh
|
||||
EXTRA_DIST += dbus-broker.sh
|
||||
EXTRA_DIST += fuzz-msg-parse.sh
|
||||
EXTRA_DIST += dbus-bus.sh
|
||||
EXTRA_DIST += dbus-manager.sh
|
||||
EXTRA_DIST += dbus-service.sh
|
||||
@@ -145,6 +146,7 @@ TESTS += dbus-service.sh
|
||||
TESTS += dbus-cond.sh
|
||||
TESTS += dbus-initctl.sh
|
||||
TESTS += dbus-introspect.sh
|
||||
TESTS += fuzz-msg-parse.sh
|
||||
# Needs the plugin to bring up the bus, not just the built-in one
|
||||
if BUILD_DBUS_PLUGIN
|
||||
TESTS += dbus-broker.sh
|
||||
|
||||
Executable
+20
@@ -0,0 +1,20 @@
|
||||
#!/bin/sh
|
||||
# libink: __msg_parse() against truncation, corruption, and garbage.
|
||||
#
|
||||
# Runs the fuzz target's fixed sweep, which is the same contract check
|
||||
# libFuzzer drives, so the suite covers it on every build without
|
||||
# needing clang. Anything it finds aborts, and the sanitizers CI
|
||||
# builds with turn a stray read into a failure here rather than a
|
||||
# puzzle on a target.
|
||||
|
||||
set -eu
|
||||
|
||||
TEST_DIR=$(dirname "$0")
|
||||
DRIVER="$TEST_DIR/src/fuzz-msg-parse"
|
||||
|
||||
[ -x "$DRIVER" ] || {
|
||||
echo "fuzz-msg-parse not built, D-Bus support is off"
|
||||
exit 77
|
||||
}
|
||||
|
||||
exec "$DRIVER"
|
||||
@@ -14,4 +14,9 @@ noinst_PROGRAMS += dbus-auth-client
|
||||
dbus_auth_client_SOURCES = dbus-auth-client.c
|
||||
dbus_auth_client_CPPFLAGS = -D_GNU_SOURCE -I$(top_srcdir)/libink
|
||||
dbus_auth_client_LDADD = $(top_builddir)/libink/libink.la
|
||||
|
||||
noinst_PROGRAMS += fuzz-msg-parse
|
||||
fuzz_msg_parse_SOURCES = fuzz-msg-parse.c
|
||||
fuzz_msg_parse_CPPFLAGS = -D_GNU_SOURCE -I$(top_srcdir)/libink -I$(top_builddir)
|
||||
fuzz_msg_parse_LDADD = $(top_builddir)/libink/libink.la
|
||||
endif
|
||||
|
||||
@@ -0,0 +1,232 @@
|
||||
/* libink — fuzz target for __msg_parse().
|
||||
*
|
||||
* __msg_parse() is the one place where bytes off a socket become
|
||||
* pointers, before any authentication has vouched for the peer. It
|
||||
* hands back borrowed pointers into the caller's buffer, so "did not
|
||||
* crash" is too weak a pass: a field that points outside the header it
|
||||
* was supposed to come from, or a string with no terminator inside it,
|
||||
* is a bug the caller hits later and somewhere else. Every input is
|
||||
* checked against that contract here.
|
||||
*
|
||||
* Built two ways. With libFuzzer (clang -fsanitize=fuzzer,address
|
||||
* -DLINK_FUZZ_LIBFUZZER) it is a normal fuzz target. Otherwise it
|
||||
* gets the driver below: named files are replayed, which is how a
|
||||
* crash found by the fuzzer is reproduced, and with no arguments it
|
||||
* runs a fixed sweep so the suite exercises the same contract on
|
||||
* every build without needing clang or a corpus in the tree.
|
||||
*
|
||||
* Copyright (c) 2026 Joachim Wiberg <troglobit@gmail.com>
|
||||
* SPDX-License-Identifier: MIT
|
||||
*/
|
||||
|
||||
#include <stdint.h>
|
||||
#include <stdio.h>
|
||||
#include <stdlib.h>
|
||||
#include <string.h>
|
||||
|
||||
#include "internal.h"
|
||||
|
||||
#define FRAME_MAX 512
|
||||
#define HDR_FIXED_SIZE 16 /* must agree with proto.c */
|
||||
|
||||
static void fail(const char *what, size_t size)
|
||||
{
|
||||
fprintf(stderr, "fuzz-msg-parse: %s, on a %zu byte input\n", what, size);
|
||||
abort();
|
||||
}
|
||||
|
||||
/* A header field is read out of the header field array and nowhere
|
||||
* else. Checking it against the whole buffer would be too generous:
|
||||
* a parser that walked off the end of the fields and into the body
|
||||
* would still be pointing at bytes it was handed, and pass. The
|
||||
* bound is derived from the raw header here rather than taken from
|
||||
* the parser, so the two have to agree independently.
|
||||
*
|
||||
* The string must also terminate inside that region, or whoever
|
||||
* borrows it reads past what it was given. */
|
||||
static void check_str(const char *s, const uint8_t *base, size_t len,
|
||||
size_t hdr_end, const char *what)
|
||||
{
|
||||
const char *first = (const char *)base + HDR_FIXED_SIZE;
|
||||
const char *last = (const char *)base + hdr_end;
|
||||
size_t room;
|
||||
|
||||
if (!s)
|
||||
return;
|
||||
if (s < first || s >= last)
|
||||
fail(what, len);
|
||||
|
||||
room = (size_t)(last - s);
|
||||
if (strnlen(s, room) == room)
|
||||
fail(what, len);
|
||||
}
|
||||
|
||||
int LLVMFuzzerTestOneInput(const uint8_t *data, size_t size);
|
||||
|
||||
int LLVMFuzzerTestOneInput(const uint8_t *data, size_t size)
|
||||
{
|
||||
struct link_msg m;
|
||||
size_t hdr_end, body_off;
|
||||
uint32_t fields_len;
|
||||
ssize_t rc;
|
||||
|
||||
rc = __msg_parse(data, size, &m);
|
||||
if (rc <= 0)
|
||||
return 0; /* need more, or refused: both fine */
|
||||
|
||||
/* Consuming more than it was handed would desynchronise the
|
||||
* read loop and make it skip into the next message. */
|
||||
if ((size_t)rc > size)
|
||||
fail("claimed more bytes than it was given", size);
|
||||
|
||||
/* Only 'l' messages parse, so reading the length this way is
|
||||
* safe, and rc > 0 means the fixed header was all there. */
|
||||
memcpy(&fields_len, data + 12, sizeof(fields_len));
|
||||
hdr_end = HDR_FIXED_SIZE + fields_len;
|
||||
body_off = (hdr_end + 7) & ~(size_t)7;
|
||||
if (hdr_end > size || body_off > size)
|
||||
fail("accepted a header longer than the message", size);
|
||||
|
||||
check_str(m.path, data, size, hdr_end, "path outside the header");
|
||||
check_str(m.interface, data, size, hdr_end, "interface outside the header");
|
||||
check_str(m.member, data, size, hdr_end, "member outside the header");
|
||||
check_str(m.error_name, data, size, hdr_end, "error name outside the header");
|
||||
check_str(m.destination, data, size, hdr_end, "destination outside the header");
|
||||
check_str(m.sender, data, size, hdr_end, "sender outside the header");
|
||||
check_str(m.signature, data, size, hdr_end, "signature outside the header");
|
||||
|
||||
if (m.body_avail) {
|
||||
if (m.body != data + body_off)
|
||||
fail("body does not start where the header ends", size);
|
||||
if (m.body_avail > size - body_off)
|
||||
fail("body runs past the buffer", size);
|
||||
}
|
||||
|
||||
return 0;
|
||||
}
|
||||
|
||||
#ifndef LINK_FUZZ_LIBFUZZER
|
||||
/* Hand the parser a buffer sized to the input and nothing more.
|
||||
* Reading past the end of a roomy array stays inside the allocation
|
||||
* and the sanitizer never sees it; against an exact allocation the
|
||||
* same read is a fault. This is what libFuzzer does, and the reason
|
||||
* it finds things a fixed buffer cannot. */
|
||||
static void run(const uint8_t *data, size_t size)
|
||||
{
|
||||
uint8_t *exact = malloc(size ? size : 1);
|
||||
|
||||
if (!exact) {
|
||||
fprintf(stderr, "fuzz-msg-parse: out of memory\n");
|
||||
abort();
|
||||
}
|
||||
memcpy(exact, data, size);
|
||||
LLVMFuzzerTestOneInput(exact, size);
|
||||
free(exact);
|
||||
}
|
||||
|
||||
/* Deterministic, so a failure reproduces from the same build. */
|
||||
static uint32_t prng(uint32_t *state)
|
||||
{
|
||||
uint32_t x = *state;
|
||||
|
||||
x ^= x << 13;
|
||||
x ^= x >> 17;
|
||||
x ^= x << 5;
|
||||
|
||||
return *state = x;
|
||||
}
|
||||
|
||||
static int sweep(void)
|
||||
{
|
||||
uint8_t frame[FRAME_MAX], copy[FRAME_MAX];
|
||||
static const uint8_t poke[] = { 0x00, 0x01, 0x7f, 0x80, 0xff };
|
||||
uint32_t state = 0x1234abcd;
|
||||
ssize_t len;
|
||||
size_t i, j, n;
|
||||
|
||||
len = __msg_build_method_call(frame, sizeof(frame), 1,
|
||||
"/org/finit/manager", "org.finit.Manager1",
|
||||
"ListServices", NULL, NULL, 0);
|
||||
if (len <= 0) {
|
||||
fprintf(stderr, "fuzz-msg-parse: cannot build a reference frame\n");
|
||||
return 1;
|
||||
}
|
||||
|
||||
/* Every prefix: the read loop hands over whatever arrived, and
|
||||
* a short read must come back "need more", never a parse. */
|
||||
for (i = 0; i <= (size_t)len; i++)
|
||||
run(frame, i);
|
||||
|
||||
/* One byte wrong, everywhere, with the values that flip a
|
||||
* length or an offset furthest. */
|
||||
for (i = 0; i < (size_t)len; i++) {
|
||||
for (j = 0; j < sizeof(poke); j++) {
|
||||
memcpy(copy, frame, (size_t)len);
|
||||
copy[i] = poke[j];
|
||||
run(copy, (size_t)len);
|
||||
}
|
||||
}
|
||||
|
||||
/* fields_len decides where the header walk stops, so it is the
|
||||
* byte that decides whether a field is read from the header or
|
||||
* from somewhere else. Walk it across the whole frame, and
|
||||
* past it, at every truncation the read loop could hand over. */
|
||||
for (i = 0; i <= (size_t)len + 8; i++) {
|
||||
uint32_t fl = (uint32_t)i;
|
||||
|
||||
memcpy(copy, frame, (size_t)len);
|
||||
memcpy(copy + 12, &fl, sizeof(fl));
|
||||
for (j = 0; j <= (size_t)len; j++)
|
||||
run(copy, j);
|
||||
}
|
||||
|
||||
/* And input that was never a message to begin with. */
|
||||
for (n = 0; n < 20000; n++) {
|
||||
size_t sz = prng(&state) % (FRAME_MAX + 1);
|
||||
|
||||
for (i = 0; i < sz; i++)
|
||||
copy[i] = (uint8_t)prng(&state);
|
||||
|
||||
/* Half of them keep a plausible header, so the parser
|
||||
* gets past its first checks and into field walking. */
|
||||
if (sz >= 16 && (n & 1)) {
|
||||
copy[0] = 'l';
|
||||
copy[3] = LINK_PROTOCOL_VERSION;
|
||||
}
|
||||
run(copy, sz);
|
||||
}
|
||||
|
||||
return 0;
|
||||
}
|
||||
|
||||
static int replay(const char *path)
|
||||
{
|
||||
uint8_t buf[64 * 1024];
|
||||
size_t len;
|
||||
FILE *fp;
|
||||
|
||||
fp = fopen(path, "rb");
|
||||
if (!fp) {
|
||||
perror(path);
|
||||
return 1;
|
||||
}
|
||||
len = fread(buf, 1, sizeof(buf), fp);
|
||||
fclose(fp);
|
||||
|
||||
run(buf, len);
|
||||
return 0;
|
||||
}
|
||||
|
||||
int main(int argc, char *argv[])
|
||||
{
|
||||
int i, rc = 0;
|
||||
|
||||
if (argc < 2)
|
||||
return sweep();
|
||||
|
||||
for (i = 1; i < argc; i++)
|
||||
rc |= replay(argv[i]);
|
||||
|
||||
return rc;
|
||||
}
|
||||
#endif /* !LINK_FUZZ_LIBFUZZER */
|
||||
Reference in New Issue
Block a user