Files
nxdns/tests/fuzz/http_util_fuzz.zig
T
mokhtar 6f67940995
CI / test (push) Successful in 1m22s
CI / test-aarch64 (push) Successful in 5m6s
CI / frontend (push) Successful in 45s
CI / cross (push) Successful in 7m53s
CI / docker (push) Failing after 1h10m57s
milestone 18: collapse duplicated infrastructure into shared listener core, crud list helper, resource shells, transport race, name and line helpers, ui modules
2026-08-07 18:20:30 +02:00

187 lines
7.8 KiB
Zig

//! Fuzz targets for the HTTP request parsers (`src/web/http_util.zig`).
//!
//! These read the third untrusted-byte family nxdns accepts: the request line
//! and the query string a browser — or anything else on the LAN — sends. Every
//! target holds the same contract as the DNS and blocklist targets: rejecting
//! bytes with an error is correct, panicking or reading out of bounds is not.
//!
//! Two invariants the file itself states are what a success has to satisfy:
//!
//! - split before decode. `parsePath` cuts segments at `/` and only then
//! percent-decodes each one, so a `%2F` inside a segment stays inside it.
//! The decoded segment does hold a literal `/` — `http_util.zig`'s own test
//! asserts that — but it is one segment, not two. So the property is a
//! count: the segmentation is the one the *raw* bytes describe, and a
//! decoder that ran before the split would hand back more segments than
//! the raw bytes have.
//! - decode only shrinks. `decodeInPlace` writes behind its own read cursor,
//! so the result is never longer than its input and always aliases the front
//! of the same buffer. If either ever stopped holding, the write cursor
//! would have passed the read cursor and the decoder would be reading bytes
//! it had already overwritten.
//!
//! `http_util.zig` imports only `std`, so the module here roots straight at it;
//! no staged aggregator is involved.
//!
//! Runner semantics: under a plain `zig build test` each target runs once per
//! corpus entry plus once on empty input, which makes the corpus a regression
//! suite. `zig build test --fuzz=<n>` gives each target `n` generated inputs.
const std = @import("std");
const http_util = @import("http_util");
const smith_encode = @import("smith_encode.zig");
const sliceInput = smith_encode.sliceInput;
const pairInput = smith_encode.pairInput;
const Smith = std.testing.Smith;
/// A target longer than this is a 414 before it reaches any parser
/// (`http_util.max_target_len`), so a longer input buys no new code paths.
const max_input = 4096;
/// `Smith` entity ids. The query target needs stable, distinct ids for its query
/// string and its key; the single-slice targets take the first.
const primary_hash: u32 = 1;
const secondary_hash: u32 = 2;
const fuzz_options: std.testing.FuzzInputOptions = .{ .corpus = &corpus };
test "fuzz http_util.parsePath" {
try std.testing.fuzz({}, pathTarget, fuzz_options);
}
test "fuzz http_util.decodeInPlace" {
try std.testing.fuzz({}, decodeTarget, fuzz_options);
}
test "fuzz http_util.queryValue" {
try std.testing.fuzz({}, queryTarget, fuzz_options);
}
fn pathTarget(_: void, smith: *Smith) anyerror!void {
var buf: [max_input]u8 = undefined;
const len = smith.sliceWithHash(&buf, primary_hash);
const input = buf[0..len];
// `parsePath` decodes in place, so the raw request line is kept: it is what
// the segmentation has to agree with.
var raw_buf: [max_input]u8 = undefined;
@memcpy(raw_buf[0..len], input);
const raw = raw_buf[0..len];
const path = http_util.parsePath(input) catch return;
try std.testing.expect(path.len <= http_util.max_path_segments);
// Every non-empty run between two `/` in the raw bytes is one segment, in
// order. A non-empty run always decodes to at least one byte, so the two
// sequences are the same length and pair up.
var chunks = std.mem.splitScalar(u8, raw, '/');
var i: usize = 0;
while (chunks.next()) |chunk| {
if (chunk.len == 0) continue;
try std.testing.expect(i < path.len);
const segment = path.segments()[i];
// Decode only shrinks, per segment.
try std.testing.expect(segment.len <= chunk.len);
try expectAliases(segment, input);
i += 1;
}
try std.testing.expectEqual(path.len, i);
}
fn decodeTarget(_: void, smith: *Smith) anyerror!void {
var buf: [max_input]u8 = undefined;
const len = smith.sliceWithHash(&buf, primary_hash);
const input = buf[0..len];
// Both rules are fed the same bytes: `+` is a space in a query string and an
// ordinary character in a path, and neither reading may change the bound.
for ([_]http_util.PlusRule{ .literal_plus, .plus_is_space }) |rule| {
var scratch: [max_input]u8 = undefined;
@memcpy(scratch[0..len], input);
const decoded = http_util.decodeInPlace(scratch[0..len], rule) catch continue;
try std.testing.expect(decoded.len <= len);
try expectPrefixOf(decoded, scratch[0..len]);
}
}
fn queryTarget(_: void, smith: *Smith) anyerror!void {
var query_buf: [max_input]u8 = undefined;
var key_buf: [max_input]u8 = undefined;
const query = query_buf[0..smith.sliceWithHash(&query_buf, primary_hash)];
const key = key_buf[0..smith.sliceWithHash(&key_buf, secondary_hash)];
var out: [http_util.max_query_value_len]u8 = undefined;
const value = (http_util.queryValue(query, key, &out) catch return) orelse return;
// The decoded value lives at the front of the caller's buffer, which is what
// lets a handler keep it for the length of the request.
try expectPrefixOf(value, &out);
// Decode only shrinks, measured against the raw pair the walker found.
var it = http_util.queryPairs(query);
while (it.next()) |pair| {
if (!std.mem.eql(u8, pair.key, key)) continue;
try std.testing.expect(value.len <= pair.value.len);
return;
}
return error.TestValueWithoutPair;
}
/// A result the parsers hand back is a window into the caller's buffer, never a
/// copy and never a pointer into a temporary.
fn expectAliases(result: []const u8, buffer: []const u8) !void {
if (result.len == 0) return;
const start = @intFromPtr(result.ptr);
const base = @intFromPtr(buffer.ptr);
try std.testing.expect(start >= base);
try std.testing.expect(start + result.len <= base + buffer.len);
}
/// The stronger form the in-place decoders owe: the result starts where the
/// input started. A decoder that writes ahead of its read cursor cannot satisfy
/// this and a shorter-but-moved slice would slip past `expectAliases`.
fn expectPrefixOf(result: []const u8, buffer: []const u8) !void {
try std.testing.expect(result.len <= buffer.len);
if (buffer.len == 0) return;
try std.testing.expectEqual(@intFromPtr(buffer.ptr), @intFromPtr(result.ptr));
}
// ---------------------------------------------------------------------------
// corpus
// ---------------------------------------------------------------------------
//
// `Smith` does not consume a corpus entry as raw parser input, so every entry
// below goes through the `smith_encode.zig` encoders. The three targets share one
// corpus: each starts with a slice, and the query target reads a second one that
// falls back to empty when an entry carries only the first.
/// A route the router actually matches, with an id capture.
const api_path = "/api/groups/12";
/// The encoded slash: the byte pattern the split-before-decode rule exists for.
/// It decodes to `a/b/../etc` inside one segment and must stay one segment.
const encoded_slash = "/api/rules/a%2Fb%2F..%2Fetc";
/// One segment past `max_path_segments`.
const deep_path = "/1/2/3/4/5/6/7/8/9";
/// The three malformed escapes `decodeInPlace` refuses rather than passes
/// through, and a plus that means different things under the two rules.
const bad_escapes = "/%2/%/%zz/a+b";
const corpus = [_][]const u8{
sliceInput(api_path),
sliceInput(encoded_slash),
sliceInput(deep_path),
sliceInput(bad_escapes),
sliceInput("/api//groups/"),
// The query shapes the API defines, each with the key that reads it.
pairInput("domain=a+b&limit=250", "domain"),
pairInput("domain=%61%2Fb&blocked=1", "domain"),
pairInput("a=1&&b&c=", "c"),
pairInput("domain=" ++ "x" ** 1024, "domain"),
pairInput("domain=%zz", "domain"),
};