milestone 12: performance bench harness, measured docs, aarch64 tests under qemu and no-dist size assert

This commit is contained in:
2026-08-02 16:08:23 +02:00
parent bdb6ffab7a
commit d522b1f947
5 changed files with 760 additions and 0 deletions
+49
View File
@@ -25,6 +25,27 @@ jobs:
- name: Run test suite (unit + hermetic loopback integration)
run: zig build test -Dintegration
test-aarch64:
runs-on: ubuntu-24.04
steps:
- uses: actions/checkout@v4
- name: Set up Zig
uses: mlugg/setup-zig@v2
with:
version: ${{ env.ZIG_VERSION }}
# qemu-user, not qemu-user-static: Zig execs the bare `qemu-aarch64`
# name, and the -static package only ships `qemu-aarch64-static`.
- name: Install qemu-user
run: |
sudo apt-get update -qq
sudo apt-get install -qq -y --no-install-recommends qemu-user
- name: Run test suite under qemu (plain suite, no -Dintegration)
run: zig build test-aarch64 -fqemu
frontend:
runs-on: ubuntu-24.04
@@ -137,6 +158,34 @@ jobs:
fi
done
# PLAN §18 also budgets the binary without web assets (< 10 MiB). A
# separate prefix keeps the with-assets artifacts above intact.
- name: Build static musl executables without web assets
run: zig build cross -Doptimize=ReleaseSafe --prefix zig-out/nodist
- name: Assert asset-free executables are within the size budget
run: |
set -euo pipefail
size_limit=$((10 * 1024 * 1024))
for triple in x86_64-linux-musl aarch64-linux-musl; do
binary="zig-out/nodist/cross/$triple/nxdns"
if [ ! -f "$binary" ]; then
echo "missing executable: $binary"
exit 1
fi
case "$triple" in
x86_64-*) strip_tool=objcopy ;;
aarch64-*) strip_tool=aarch64-linux-gnu-objcopy ;;
esac
"$strip_tool" --strip-all "$binary" "$binary.stripped"
size=$(stat -c %s "$binary.stripped")
echo "$triple: stripped size without assets $size bytes"
if [ "$size" -ge "$size_limit" ]; then
echo "stripped asset-free executable exceeds the 10 MiB budget: $binary"
exit 1
fi
done
docker:
runs-on: ubuntu-24.04
+69
View File
@@ -109,6 +109,75 @@ pub fn build(b: *std.Build) void {
});
test_step.dependOn(&b.addRunArtifact(blocklist_fuzz_tests).step);
// The bench harness (milestone-12 ruling 1). The measured roots
// (matcher.zig, dns_cache.zig, compiler.zig) share files in their relative
// import closures (model.zig, types.zig, ...), and a file may belong to
// only one module per compilation — separate modules per root cannot link
// into one executable. So one staged module: a copy of src/ plus a
// generated aggregator root, imported by the bench as `core`. No sqlite,
// no mbedTLS: the closure is pure Zig.
const bench_stage = b.addWriteFiles();
_ = bench_stage.addCopyDirectory(b.path("src"), "src", .{});
const bench_core = bench_stage.add("bench_core.zig",
\\pub const matcher = @import("src/filter/matcher.zig");
\\pub const dns_cache = @import("src/cache/dns_cache.zig");
\\pub const compiler = @import("src/filter/compiler.zig");
\\pub const model = @import("src/config/model.zig");
\\pub const dns_name = @import("src/dns/name.zig");
\\pub const dns_types = @import("src/dns/types.zig");
\\pub const packet = @import("src/dns/packet.zig");
\\
);
const bench_core_mod = b.createModule(.{
.root_source_file = bench_core,
.target = target,
.optimize = optimize,
});
const bench_mod = b.createModule(.{
.root_source_file = b.path("tools/bench.zig"),
.target = target,
.optimize = optimize,
});
bench_mod.addImport("core", bench_core_mod);
const bench_exe = b.addExecutable(.{ .name = "bench", .root_module = bench_mod });
const bench_run = b.addRunArtifact(bench_exe);
if (b.args) |args| bench_run.addArgs(args);
b.step("bench", "Run the performance benchmarks (PLAN §18)").dependOn(&bench_run.step);
// aarch64 test execution (milestone-12 ruling 6): the plain suite
// cross-built for the deploy target and run under qemu-user
// (`zig build test-aarch64 -fqemu`). Fuzz artifacts stay native-only, and
// -Dintegration stays out (ruling 7): qemu-user's slowdown makes the
// wall-clock-budgeted loopback TLS tests a flake source.
const aarch64_target = b.resolveTargetQuery(
std.Target.Query.parse(.{ .arch_os_abi = "aarch64-linux-musl" }) catch unreachable,
);
const aarch64_tests = b.addTest(.{
.root_module = b.createModule(.{
.root_source_file = b.path("src/tests.zig"),
.target = aarch64_target,
.optimize = optimize,
.link_libc = true,
}),
});
aarch64_tests.linkage = .static;
aarch64_tests.root_module.addOptions("build_options", options);
aarch64_tests.root_module.linkLibrary(sqliteLibrary(b, aarch64_target, optimize));
aarch64_tests.root_module.linkLibrary(mbedtlsLibrary(b, aarch64_target, optimize));
aarch64_tests.root_module.addCSourceFile(.{ .file = b.path("src/platform/mbedtls_shim.c") });
addMbedtlsThreadingMacros(aarch64_tests.root_module);
aarch64_tests.root_module.addAnonymousImport("test_fixtures", .{
.root_source_file = b.path("tests/fixtures/fixtures.zig"),
});
aarch64_tests.root_module.addAnonymousImport("docs_files", .{
.root_source_file = b.path("docs/docs.zig"),
});
aarch64_tests.root_module.addAnonymousImport("web_assets", .{ .root_source_file = web_assets });
const aarch64_run = b.addRunArtifact(aarch64_tests);
aarch64_run.skip_foreign_checks = true;
b.step("test-aarch64", "Run the test suite for aarch64-linux-musl (use -fqemu)")
.dependOn(&aarch64_run.step);
const cross = b.step("cross", "Build static musl executables for every deploy target");
for (cross_targets) |triple| {
const query = std.Target.Query.parse(.{ .arch_os_abi = triple }) catch |err| {
+81
View File
@@ -0,0 +1,81 @@
# Performance
PLAN §18 sets the targets; `tools/bench.zig` measures the three that are
measurable in-process. Run it with:
```
zig build bench -Doptimize=ReleaseFast
```
Subcommands `filter|cache|compile|all` (default `all`) select a suite; flags
`--domains=N` (default 1,000,000), `--iters=N` (default 200,000) and `--seed=N`
(default 0x5eed) shape the load. The default run is informational; `--assert`
exits non-zero when a target below is exceeded.
## Targets (PLAN §18)
| Target | Where it is checked |
| --- | --- |
| Sustained ≥ 100 qps on Raspberry Pi 5 | End-to-end against the real binary on the Pi (see below); not a harness number |
| Blocklist lookup p95 < 1 ms | `bench filter`: `matcher.normalize` + `Snapshot.evaluate` per op |
| Cached response p95 < 5 ms | `bench cache`: `buildKey` + `DnsCache.get` + `packet.setId` per op |
| Memory with ~1M blocked domains < 100 MB | `bench filter`: VmRSS with the 1M-domain snapshot loaded |
| Stripped static binary < 10 MB per arch (< 15 MB with embedded frontend) | CI size assert on the `cross` artifacts |
## Measured: x86_64 development host (2026-08-02)
Intel Core i7-14700K, Linux 6.18, Zig 0.16.0, `-Doptimize=ReleaseFast`,
defaults (1,000,000 domains, 200,000 iterations per suite, seed 0x5eed).
**This is not the target platform** — the Pi 5's Cortex-A76 is far slower and
these numbers do not transfer; they establish the harness works and set a
baseline for regressions on the machine development happens on.
```
suite ops p50(us) p95(us) p99(us) max(us)
filter 200000 0.11 0.18 0.27 16.41
blocked 66699/200000, Snapshot.memoryBytes 28.0 MiB, VmRSS 31.8 MiB
target p95 < 1ms: PASS
target VmRSS < 100 MiB: PASS
cache 200000 0.10 0.14 0.17 3.53
hits 100000/200000, DnsCache.memoryBytes 4.3 MiB, VmRSS 7.6 MiB
target p95 < 5ms: PASS
compile 1000000 wall 96.025ms, 10413949 lines/s, 1000000 domains kept (informational)
```
Every in-process §18 target passes on this host: the two latency targets by
three-to-four orders of magnitude, the memory target by about 3x.
Two memory figures appear on purpose. `Snapshot.memoryBytes` /
`DnsCache.memoryBytes` are the in-repo accounting of the structures themselves
(the regression guard); VmRSS is what the kernel actually holds resident for
the whole process, allocator slack and code included. The truth sits between
them, and the §18 memory target is judged on VmRSS. The filter suite frees the
generated list source before reading VmRSS, so the number reflects the loaded
snapshot rather than the generator. The cache line's VmRSS is lower because the
filter suite's snapshot has been freed by then.
## Raspberry Pi 5 (target platform)
To be measured on hardware. One command, run on the Pi:
```
zig build bench -Doptimize=ReleaseFast -- --assert
```
| Target | Result |
| --- | --- |
| Blocklist lookup p95 < 1 ms | to be measured on hardware |
| Cached response p95 < 5 ms | to be measured on hardware |
| Memory with ~1M blocked domains < 100 MB | to be measured on hardware |
| Sustained ≥ 100 qps | to be measured on hardware, end-to-end |
The qps target is end-to-end and belongs to the real binary, not the harness:
run `nxdns run` on the Pi and drive it over the LAN with a DNS load generator
(for example `dnsperf`) against real blocklists.
## Why CI does not gate on performance
Required CI stays deterministic (AGENTS.md); latency assertions on shared
runners measure the runner's noisy neighbours, not nxdns, and a perf gate that
flakes trains people to re-run it. The bench exists for hardware you control:
run `--assert` on the Pi, where the numbers mean something.
+179
View File
@@ -0,0 +1,179 @@
# Milestone 12: performance measurement pass + aarch64 test execution
Goal: close the two gaps the post-Phase-10 completeness sweep found — PLAN §18 has no
measurement infrastructure (deferred at specs/milestone-7.md:113 and never delivered),
and aarch64 tests are cross-built but never executed (PLAN.md:145 promised qemu).
## Rulings (binding)
1. **Bench harness = `tools/bench.zig` + `zig build bench`.** Not in src/ — src/ is
the shipped product; tests.zig aggregates everything shippable and the bench must
be a separate compilation anyway (one-file-one-module rule; matcher.zig and
dns_cache.zig already belong to the test compilation). Module wiring per the fuzz
pattern (build.zig:68-92): modules rooted at src/filter/matcher.zig and
src/cache/dns_cache.zig — their relative import closures come along; no
sqlite/mbedTLS linking needed. `if (b.args) |args| run.addArgs(args)`.
2. **What it measures** (PLAN §18 targets):
- `filter`: `matcher.normalize` + `Snapshot.evaluate` per op against a Snapshot
built from a generated, sorted `d{d:0>7}.example.com` list body (~1M lines,
in-memory; DomainSet.max_count 4M and compiler.max_domains 2M leave headroom)
plus a small wild body. Mixed case ratios: hit, miss, parent-walk. Target
p95 < 1 ms.
- `cache`: `buildKey` + `DnsCache.get` + `packet.setId` (the handler's hit path;
TTL aging happens inside get) on a ~10k-entry cache prefilled with the fuzz
corpus response. Mixed hit/miss. Target p95 < 5 ms.
- `compile`: `compiler.compile` over a generated 1M-line hosts body — wall time,
informational (no §18 target; no datapoint exists today).
- Memory: `/proc/self/status` VmRSS (nothing in-repo wraps it; read it directly)
plus the in-repo accounting (`Snapshot.memoryBytes`, `DnsCache.memoryBytes`).
Target < 100 MB RSS with ~1M domains loaded.
- Timing: `std.Io.Clock.awake.now(io)` / `durationTo` — std.time.Timer does not
exist at 0.16. Percentiles: collect per-op nanos in a preallocated []u64, sort
(std.mem.sort), report p50/p95/p99/max. No percentile helper exists; write it
in the tool.
3. **Assertion policy.** Default run is informational (prints a table). `--assert`
exits non-zero when a §18 target is exceeded — for the Pi 5 run, NOT for CI:
required CI stays deterministic (AGENTS.md) and perf assertions on shared runners
are flake generators. CI does not run the bench at all this milestone.
4. **Bench flags**: `--domains=N` (default 1_000_000), `--iters=N` (default 200_000),
`--seed=N` (default fixed), `--assert`, subcommands `filter|cache|compile|all`
(default all). Debug-mode runs print a one-line warning recommending
`-Doptimize=ReleaseFast`.
5. **docs/performance.md**: the §18 targets table; measured numbers from THIS x86_64
host (dated, hardware named honestly, marked as not-the-target-platform); the
in-memory accounting numbers; the one-command Pi 5 recipe
(`zig build bench -Doptimize=ReleaseFast -- --assert`); a sentence on why CI does
not gate on perf. The Pi 5 rows stay "to be measured on hardware".
6. **aarch64 execution = second test artifact, `zig build test-aarch64 -fqemu`.**
Per stdlib evidence: Run steps try qemu only when enable_qemu (the `-fqemu` CLI
flag; a *Build field, no per-step override) and exec `qemu-aarch64` bare from
PATH — so CI installs `qemu-user` (NOT qemu-user-static, which ships the
`-static` name Zig will not find; binfmt is unnecessary). Static musl means no
sysroot/--libc-runtimes. build.zig: resolve aarch64-linux-musl, addTest with the
same wiring as the native tests artifact (sqlite/mbedtls helpers are already
target-parameterized), set `.linkage = .static` on the artifact (TestOptions has
no linkage field), `run.skip_foreign_checks = true`, leave
failing_to_execute_foreign_is_an_error true so a missing qemu is loud. Fuzz
artifacts excluded.
7. **qemu scope: plain suite only, blocking.** No `-Dintegration` under qemu: the
integration tests are multithreaded loopback TLS with wall-clock budgets, and
qemu-user's 5-20x slowdown makes them a flake source — required CI stays
deterministic. The plain suite (the aggregator's tests without the 12
fuzz-artifact tests: 1159 pass + 113 integration-gated skips; all pure
DNS/filter/cache logic included)
is the portable-correctness signal aarch64 needs. Recorded here as the
engineering call closing PLAN.md:145's promise.
8. **CI additions**: one `test-aarch64` job (setup-zig, apt qemu-user,
`zig build test-aarch64 -fqemu`); plus the missing §18 size assert — the cross
job gains a second build WITHOUT `-Dweb-dist` (placeholder dist) and asserts the
stripped copies < 10 MiB per arch (the < 15 MiB with-assets assert already
exists). Job graph otherwise untouched.
9. **No src/ changes.** matcher/dns_cache/compiler public APIs are used as-is; if
the bench needs something they do not expose, report — do not extend them.
## Sessions
V1 then V2 (both edit build.zig — sequenced, not parallel).
## Session V1: bench harness + performance doc
Owns tools/bench.zig, build.zig (the bench step only), docs/performance.md.
Rulings 1-5. Runs the bench on this host (ReleaseFast) and writes the measured
numbers into the doc.
### V1 As built
Delivered tools/bench.zig, the build.zig bench step, and docs/performance.md. One
deviation from ruling 1's wiring: Zig 0.16 rejects separate modules rooted at
matcher.zig and dns_cache.zig ("file exists in multiple modules" — their closures
share model.zig/types.zig; verified with a minimal repro). Instead a b.addWriteFiles
stage copies src/ plus a generated 7-line aggregator root (bench_core.zig) into one
`core` module — the same staging trick the web assets use; a build.zig comment
records why. The fuzz-corpus response is a byte-for-byte copy with provenance
comment (importing corpus.zig recreates the module conflict; the corpus documents
copies as house style). /proc/self/status reads use readerStreaming (procfs stats
size 0, readFileAlloc returns empty). Measured on the build host (i7-14700K,
ReleaseFast, defaults: 1M domains, 200k iters): filter p95 0.18 µs (target < 1 ms),
cache p95 0.14 µs (target < 5 ms), VmRSS 31.8 MiB with the 1M-domain snapshot
(target < 100 MiB), Snapshot.memoryBytes 28.0 MiB, compile 96 ms ≈ 10.4M lines/s
(informational). --assert exit paths verified both ways (a 4M-domain run exceeds
the 1M-scoped RSS target and exits 1; that was an exit-path exercise, not a target
miss). No src/ changes; matcher/dns_cache/compiler APIs sufficed. Cache suite fixes
entries at 10k (the handler-default Config.size); --domains shapes filter and
compile only. Both suites 0 failed after the build.zig edit.
## Session V2: aarch64 test execution + CI (after V1)
Owns build.zig (the test-aarch64 artifact/step only), .gitea/workflows/ci.yml
(the new job + the cross-job size-assert addition). Rulings 6-8. Runs
`zig build test-aarch64 -fqemu` locally if qemu-aarch64 is installable/present;
reports honestly if not.
### V2 As built
build.zig gained the test-aarch64 block after the bench block: aarch64-linux-musl
addTest with wiring identical to the native tests artifact, `.linkage = .static` on
the artifact, `run.skip_foreign_checks = true`, failing-to-execute left loud. Stdlib
mechanics re-verified (Run.zig:71/78/226, Build.zig:73, system.zig:111 — bare
`qemu-aarch64` from PATH). ci.yml gained the test-aarch64 job (apt qemu-user,
`zig build test-aarch64 -fqemu`) and the cross job builds a second no-dist pair to
`--prefix zig-out/nodist` (existing with-assets steps byte-identical) with a
< 10 MiB stripped assert; the new assert skips the static-linkage recheck (proven
on the same-config with-assets build). Local verification: real qemu execution —
1159 pass / 113 skip / 0 failed (native count minus the 12 excluded fuzz tests);
no-dist stripped sizes 5,880,336 (x86_64) and 5,212,640 (aarch64) bytes. No
-Dintegration under qemu per ruling 7.
## Review (Codex, as built)
Two rounds on one thread; round 2 returned "No findings."
Round 1 (2 important, 2 minor): bench.zig's explicit body.deinit left the earlier
errdefer armed on an undefined list (double-free on any later error) → clearAndFree
keeps the list valid. Ubuntu's qemu-user recommends qemu-user-binfmt, which apt
installs by default — binfmt registration would bypass the -fqemu path CI intends
to exercise → --no-install-recommends. performance.md's headroom claim overstated
the memory margin (3x, not orders of magnitude) → corrected. Ruling 7 quoted the
native aggregate test count instead of the aarch64 artifact's own (1159 + 113
skips, fuzz excluded) → corrected.
Final gates: plain 1171/1284 passed, 113 skipped (integration-gated), 0 failed;
integration 1280/1284, 4 skipped (live-network by design), 0 failed; qemu aarch64
run 1159 passed, 113 skipped, 0 failed; cross ReleaseSafe with dist 18/18; no-dist
stripped sizes 5.9/5.2 MB under the 10 MiB assert.
## Module layout (new)
tools/bench.zig, docs/performance.md.
## File ownership
V1 tools/bench.zig + docs/performance.md + build.zig(bench); V2 build.zig(aarch64) +
ci.yml. Orchestrator: spec sync, review, commit.
## Acceptance (milestone complete)
- [ ] `zig build bench -Doptimize=ReleaseFast` runs all three suites and prints
p50/p95/p99/max + memory; `--assert` enforces the §18 targets.
- [ ] docs/performance.md holds dated x86_64 numbers + the Pi 5 recipe.
- [ ] `zig build test-aarch64 -fqemu` passes locally under qemu (or its
unavailability is recorded with the exact CI-equivalent command).
- [ ] CI: test-aarch64 job green-by-construction (same command CI runs); cross job
asserts < 10 MiB stripped without assets, < 15 MiB with (existing).
- [ ] Both existing suites 0 failed; no src/ changes.
## Anti-requirements
- No perf assertions in CI; no bench in the test step.
- No qemu integration suite; no binfmt setup; no qemu-user-static.
- No new pub API on matcher/dns_cache/compiler; no synthetic-load DNS server
benchmark (the ≥100 qps target is end-to-end on the Pi — the operator recipe
covers it via the real binary, not a harness).
+382
View File
@@ -0,0 +1,382 @@
//! Performance bench harness (PLAN §18, milestone-12 rulings 1-4).
//!
//! `zig build bench -Doptimize=ReleaseFast -- [filter|cache|compile|all] [flags]`
//!
//! Flags: `--domains=N` (default 1_000_000), `--iters=N` (default 200_000),
//! `--seed=N` (default 0x5eed), `--assert`. The default run is informational;
//! `--assert` exits non-zero when a §18 target is exceeded — meant for the
//! Pi 5 acceptance run, never for CI (required CI stays deterministic).
//!
//! What each suite measures:
//! - `filter`: `matcher.normalize` + `Snapshot.evaluate` per op — the handler's
//! filtering work — against a snapshot built from `--domains` generated exact
//! entries plus a small wildcard body. Query mix cycles hit, miss and
//! parent-walk. Target p95 < 1 ms; VmRSS < 100 MiB with the list loaded.
//! - `cache`: `buildKey` + `DnsCache.get` + `packet.setId` — the handler's
//! cache-hit path, TTL aging included — on a 10k-entry cache prefilled with a
//! realistic response. Query mix alternates hit and miss. Target p95 < 5 ms.
//! - `compile`: `compiler.compile` over `--domains` generated hosts lines.
//! Wall time, informational (no §18 target).
const std = @import("std");
const builtin = @import("builtin");
const core = @import("core");
const matcher = core.matcher;
const dns_cache = core.dns_cache;
const compiler = core.compiler;
const model = core.model;
const dns_name = core.dns_name;
const dns_types = core.dns_types;
const packet = core.packet;
const Allocator = std.mem.Allocator;
const Writer = std.Io.Writer;
const filter_p95_target_ns: u64 = 1 * std.time.ns_per_ms;
const cache_p95_target_ns: u64 = 5 * std.time.ns_per_ms;
const rss_target_bytes: usize = 100 * 1024 * 1024;
const cache_entries: u32 = 10_000;
/// Byte-for-byte copy of `response` in tests/fuzz/corpus.zig (a copy on
/// purpose, same as the corpus itself: a bench input that changes whenever a
/// test fixture is edited is a benchmark that silently shifts). A CNAME to
/// www.example.com (TTL 300) plus its A record (TTL 60) and an OPT record,
/// so `classify` stores it as a positive entry with a 60 s lifetime.
const cached_response =
"\x12\x34\x81\x80\x00\x01\x00\x02\x00\x00\x00\x01" ++
"\x07example\x03com\x00\x00\x01\x00\x01" ++
"\xc0\x0c\x00\x05\x00\x01\x00\x00\x01\x2c\x00\x06\x03www\xc0\x0c" ++
"\xc0\x29\x00\x01\x00\x01\x00\x00\x00\x3c\x00\x04\x5d\xb8\xd8\x22" ++
"\x00\x00\x29\x10\x00\x00\x00\x00\x00\x00\x00";
/// Strictly ascending, so `DomainSet.build` accepts it. Never matched by the
/// generated queries: the wildcards exist to be walked past, the way a real
/// snapshot's wildcard set is on most queries.
const wild_body = "ads.bench.invalid\nmetrics.bench.invalid\ntelemetry.bench.invalid\n";
const usage =
"usage: zig build bench -Doptimize=ReleaseFast -- " ++
"[filter|cache|compile|all] [--domains=N] [--iters=N] [--seed=N] [--assert]";
const Suite = enum { filter, cache, compile, all };
const Options = struct {
suite: Suite = .all,
domains: u32 = 1_000_000,
iters: u32 = 200_000,
seed: u64 = 0x5eed,
assert: bool = false,
};
pub fn main(init: std.process.Init) !u8 {
const arena = init.arena.allocator();
const gpa = init.gpa;
const io = init.io;
const args = try init.minimal.args.toSlice(arena);
const opts = parseOptions(args);
var out_buffer: [4096]u8 = undefined;
var out = std.Io.File.stdout().writer(io, &out_buffer);
const w = &out.interface;
if (builtin.mode == .Debug) {
try w.print("warning: Debug build; run with -Doptimize=ReleaseFast for meaningful numbers\n", .{});
}
try w.print("nxdns bench suite={t} domains={d} iters={d} seed=0x{x} optimize={t}\n\n", .{
opts.suite, opts.domains, opts.iters, opts.seed, builtin.mode,
});
try w.print("{s:<9}{s:>10}{s:>12}{s:>12}{s:>12}{s:>12}\n", .{
"suite", "ops", "p50(us)", "p95(us)", "p99(us)", "max(us)",
});
var exceeded: u32 = 0;
if (opts.suite == .filter or opts.suite == .all) exceeded += try runFilter(io, gpa, opts, w);
if (opts.suite == .cache or opts.suite == .all) exceeded += try runCache(io, gpa, opts, w);
if (opts.suite == .compile or opts.suite == .all) try runCompile(io, gpa, opts, w);
if (exceeded > 0) try w.print("\n{d} target(s) exceeded\n", .{exceeded});
try w.flush();
return if (opts.assert and exceeded > 0) 1 else 0;
}
fn parseOptions(args: []const [:0]const u8) Options {
var opts: Options = .{};
for (args[1..]) |arg| {
if (std.mem.eql(u8, arg, "--assert")) {
opts.assert = true;
} else if (std.mem.startsWith(u8, arg, "--domains=")) {
opts.domains = parseNumber(u32, arg, "--domains=");
} else if (std.mem.startsWith(u8, arg, "--iters=")) {
opts.iters = parseNumber(u32, arg, "--iters=");
} else if (std.mem.startsWith(u8, arg, "--seed=")) {
opts.seed = parseNumber(u64, arg, "--seed=");
} else if (std.meta.stringToEnum(Suite, arg)) |suite| {
opts.suite = suite;
} else {
std.process.fatal("unknown argument '{s}'\n{s}", .{ arg, usage });
}
}
if (opts.iters == 0) std.process.fatal("--iters must be at least 1", .{});
if (opts.domains == 0) std.process.fatal("--domains must be at least 1", .{});
// Seven zero-padded digits keep generation order equal to sorted order;
// the DomainSet cap is lower anyway.
if (opts.domains > 4_000_000) std.process.fatal("--domains must be at most 4000000", .{});
return opts;
}
fn parseNumber(comptime T: type, arg: []const u8, prefix: []const u8) T {
return std.fmt.parseInt(T, arg[prefix.len..], 10) catch {
std.process.fatal("bad value in '{s}'\n{s}", .{ arg, usage });
};
}
// ---------------------------------------------------------------------------
// Suites
// ---------------------------------------------------------------------------
/// Returns how many §18 targets the suite exceeded.
fn runFilter(io: std.Io, gpa: Allocator, opts: Options, w: *Writer) !u32 {
var body: std.ArrayList(u8) = .empty;
errdefer body.deinit(gpa);
try body.ensureTotalCapacity(gpa, @as(usize, opts.domains) * 20);
var line: [64]u8 = undefined;
for (0..opts.domains) |i| {
const text = std.fmt.bufPrint(&line, "d{d:0>7}.example.com\n", .{i}) catch unreachable;
try body.appendSlice(gpa, text);
}
const sources = [_]model.BlocklistSource{.{ .url = "bench://list", .name = "bench" }};
const links = [_]model.GroupSource{.{ .group = "default", .source_url = "bench://list" }};
var snapshot = try matcher.Snapshot.build(gpa, .{
.groups = &.{.{ .name = "default" }},
.group_ids = &.{1},
.group_sources = &links,
.sources = &sources,
.source_ids = &.{1},
.rules = &.{},
.clients = &.{},
.prefixes = &.{},
.compiled = &.{.{ .list_body = body.items, .wild_body = wild_body }},
.seed = opts.seed,
.generation = 1,
});
defer snapshot.deinit();
// The snapshot copied everything it needs; freeing the source body before
// the RSS read keeps the memory number about the loaded snapshot.
// clearAndFree leaves the list valid so the errdefer above stays safe.
body.clearAndFree(gpa);
var prng = std.Random.DefaultPrng.init(opts.seed);
const random = prng.random();
const pool = try gpa.alloc(dns_name.Name, 4096);
defer gpa.free(pool);
for (pool, 0..) |*entry, i| {
const r = random.uintLessThan(u32, opts.domains);
const text = switch (i % 3) {
0 => std.fmt.bufPrint(&line, "d{d:0>7}.example.com", .{r}),
1 => std.fmt.bufPrint(&line, "m{d:0>7}.example.org", .{r}),
else => std.fmt.bufPrint(&line, "a.b.d{d:0>7}.example.com", .{r}),
} catch unreachable;
entry.* = dns_name.fromText(text) catch unreachable;
}
const samples = try gpa.alloc(u64, opts.iters);
defer gpa.free(samples);
var buf: [dns_types.max_name_len]u8 = undefined;
for (pool) |qname| {
std.mem.doNotOptimizeAway(snapshot.evaluate(0, matcher.normalize(qname, &buf)).blocked);
}
var blocked: u64 = 0;
for (samples, 0..) |*sample, i| {
const qname = pool[i % pool.len];
const t0 = std.Io.Clock.awake.now(io);
const domain = matcher.normalize(qname, &buf);
const decision = snapshot.evaluate(0, domain);
const t1 = std.Io.Clock.awake.now(io);
sample.* = @intCast(@max(0, t0.durationTo(t1).toNanoseconds()));
if (decision.blocked) blocked += 1;
}
if (blocked == 0) std.process.fatal("filter bench blocked nothing; the suite is broken", .{});
const pct = percentiles(samples);
const rss = vmRssBytes(io);
try printRow(w, "filter", opts.iters, pct);
try w.print(" blocked {d}/{d}, Snapshot.memoryBytes {d:.1} MiB, VmRSS {d:.1} MiB\n", .{
blocked, opts.iters, mib(snapshot.memoryBytes()), mib(rss),
});
var exceeded: u32 = 0;
exceeded += try printTarget(w, "p95 < 1ms", pct.p95 < filter_p95_target_ns);
exceeded += try printTarget(w, "VmRSS < 100 MiB", rss < rss_target_bytes);
return exceeded;
}
fn runCache(io: std.Io, gpa: Allocator, opts: Options, w: *Writer) !u32 {
var cache = try dns_cache.DnsCache.init(gpa, .{
.size = cache_entries,
.negative_ttl_max = 3600,
});
defer cache.deinit();
const class = dns_cache.classify(cached_response, 3600) orelse {
std.process.fatal("cache bench response is not cacheable; the suite is broken", .{});
};
const filled_at: i64 = 1_000_000;
var key_buf: [dns_cache.max_key_len]u8 = undefined;
var text_buf: [64]u8 = undefined;
for (0..cache_entries) |i| {
const qname = std.fmt.bufPrint(&text_buf, "c{d:0>5}.example.com", .{i}) catch unreachable;
const key = dns_cache.buildKey(&key_buf, qname, 1, 1, false, null);
try cache.put(filled_at, key, cached_response, class);
}
const samples = try gpa.alloc(u64, opts.iters);
defer gpa.free(samples);
var prng = std.Random.DefaultPrng.init(opts.seed);
const random = prng.random();
// Inside the entry's 60 s lifetime, far enough in to make `get` age TTLs.
const queried_at = filled_at + 30;
var out_buf: [512]u8 = undefined;
var hits: u64 = 0;
for (samples, 0..) |*sample, i| {
const r = random.uintLessThan(u32, cache_entries);
const qname = if (i % 2 == 0)
std.fmt.bufPrint(&text_buf, "c{d:0>5}.example.com", .{r}) catch unreachable
else
std.fmt.bufPrint(&text_buf, "x{d:0>5}.example.org", .{r}) catch unreachable;
const t0 = std.Io.Clock.awake.now(io);
const key = dns_cache.buildKey(&key_buf, qname, 1, 1, false, null);
const found = cache.get(queried_at, key, &out_buf);
if (found) |bytes| packet.setId(bytes, @truncate(i));
const t1 = std.Io.Clock.awake.now(io);
sample.* = @intCast(@max(0, t0.durationTo(t1).toNanoseconds()));
if (found != null) hits += 1;
}
if (hits == 0) std.process.fatal("cache bench hit nothing; the suite is broken", .{});
const pct = percentiles(samples);
const rss = vmRssBytes(io);
try printRow(w, "cache", opts.iters, pct);
try w.print(" hits {d}/{d}, DnsCache.memoryBytes {d:.1} MiB, VmRSS {d:.1} MiB\n", .{
hits, opts.iters, mib(cache.memoryBytes()), mib(rss),
});
return try printTarget(w, "p95 < 5ms", pct.p95 < cache_p95_target_ns);
}
fn runCompile(io: std.Io, gpa: Allocator, opts: Options, w: *Writer) !void {
if (opts.domains > compiler.max_domains) {
std.process.fatal("compile suite needs --domains <= {d}", .{compiler.max_domains});
}
var body: std.ArrayList(u8) = .empty;
defer body.deinit(gpa);
try body.ensureTotalCapacity(gpa, @as(usize, opts.domains) * 28);
var line: [64]u8 = undefined;
for (0..opts.domains) |i| {
const text = std.fmt.bufPrint(&line, "0.0.0.0 d{d:0>7}.example.com\n", .{i}) catch unreachable;
try body.appendSlice(gpa, text);
}
var reader = std.Io.Reader.fixed(body.items);
var list_buf: [4096]u8 = undefined;
var wild_buf: [4096]u8 = undefined;
var list_out: Writer.Discarding = .init(&list_buf);
var wild_out: Writer.Discarding = .init(&wild_buf);
const t0 = std.Io.Clock.awake.now(io);
const result = compiler.compile(gpa, &reader, .hosts, &list_out.writer, &wild_out.writer) catch |err| {
std.process.fatal("compiler.compile failed: {t}", .{err});
};
const t1 = std.Io.Clock.awake.now(io);
if (result.counts.domains != opts.domains) {
std.process.fatal("compile kept {d} of {d} domains; the suite is broken", .{
result.counts.domains, opts.domains,
});
}
const elapsed_ns: u64 = @intCast(@max(1, t0.durationTo(t1).toNanoseconds()));
const lines_per_s = @as(u64, opts.domains) * std.time.ns_per_s / elapsed_ns;
try w.print("{s:<9}{d:>10} wall {f}, {d} lines/s, {d} domains kept (informational)\n", .{
"compile", opts.domains, std.Io.Duration.fromNanoseconds(@intCast(elapsed_ns)), lines_per_s,
result.counts.domains,
});
}
// ---------------------------------------------------------------------------
// Reporting helpers
// ---------------------------------------------------------------------------
const Percentiles = struct { p50: u64, p95: u64, p99: u64, max: u64 };
/// Nearest-rank percentiles over per-op nanoseconds. Sorts `samples` in place.
fn percentiles(samples: []u64) Percentiles {
std.debug.assert(samples.len > 0);
std.mem.sort(u64, samples, {}, std.sort.asc(u64));
return .{
.p50 = atRank(samples, 50),
.p95 = atRank(samples, 95),
.p99 = atRank(samples, 99),
.max = samples[samples.len - 1],
};
}
fn atRank(sorted: []const u64, pct: usize) u64 {
const rank = (sorted.len * pct + 99) / 100;
return sorted[@max(rank, 1) - 1];
}
fn printRow(w: *Writer, suite: []const u8, ops: u32, pct: Percentiles) !void {
try w.print("{s:<9}{d:>10}{d:>12.2}{d:>12.2}{d:>12.2}{d:>12.2}\n", .{
suite, ops, us(pct.p50), us(pct.p95), us(pct.p99), us(pct.max),
});
}
fn printTarget(w: *Writer, target: []const u8, ok: bool) !u32 {
try w.print(" target {s}: {s}\n", .{ target, if (ok) "PASS" else "FAIL" });
return @intFromBool(!ok);
}
fn us(ns: u64) f64 {
return @as(f64, @floatFromInt(ns)) / @as(f64, std.time.ns_per_us);
}
fn mib(bytes: usize) f64 {
return @as(f64, @floatFromInt(bytes)) / (1024.0 * 1024.0);
}
/// The kernel's resident-set figure, since nothing in-repo wraps it. In-repo
/// accounting (`Snapshot.memoryBytes`, `DnsCache.memoryBytes`) is reported
/// alongside; the two bracket the truth from below and above.
fn vmRssBytes(io: std.Io) usize {
var file = std.Io.Dir.cwd().openFile(io, "/proc/self/status", .{}) catch |err| {
std.process.fatal("cannot open /proc/self/status: {t}", .{err});
};
defer file.close(io);
// procfs reports a zero size to stat, so the size-aware alloc readers see
// an instant end-of-stream; a plain streaming read does not.
var reader_buf: [64]u8 = undefined;
var reader = file.readerStreaming(io, &reader_buf);
var status_buf: [8192]u8 = undefined;
const len = reader.interface.readSliceShort(&status_buf) catch |err| {
std.process.fatal("cannot read /proc/self/status: {t}", .{err});
};
const status = status_buf[0..len];
var lines = std.mem.splitScalar(u8, status, '\n');
while (lines.next()) |status_line| {
if (!std.mem.startsWith(u8, status_line, "VmRSS:")) continue;
var fields = std.mem.tokenizeAny(u8, status_line["VmRSS:".len..], " \t");
const kib = fields.next() orelse break;
return 1024 * (std.fmt.parseInt(usize, kib, 10) catch break);
}
std.process.fatal("no VmRSS line in /proc/self/status", .{});
}