diff --git a/.gitea/workflows/ci.yml b/.gitea/workflows/ci.yml index 7e3990f..f32179f 100644 --- a/.gitea/workflows/ci.yml +++ b/.gitea/workflows/ci.yml @@ -25,6 +25,27 @@ jobs: - name: Run test suite (unit + hermetic loopback integration) run: zig build test -Dintegration + test-aarch64: + runs-on: ubuntu-24.04 + + steps: + - uses: actions/checkout@v4 + + - name: Set up Zig + uses: mlugg/setup-zig@v2 + with: + version: ${{ env.ZIG_VERSION }} + + # qemu-user, not qemu-user-static: Zig execs the bare `qemu-aarch64` + # name, and the -static package only ships `qemu-aarch64-static`. + - name: Install qemu-user + run: | + sudo apt-get update -qq + sudo apt-get install -qq -y --no-install-recommends qemu-user + + - name: Run test suite under qemu (plain suite, no -Dintegration) + run: zig build test-aarch64 -fqemu + frontend: runs-on: ubuntu-24.04 @@ -137,6 +158,34 @@ jobs: fi done + # PLAN §18 also budgets the binary without web assets (< 10 MiB). A + # separate prefix keeps the with-assets artifacts above intact. + - name: Build static musl executables without web assets + run: zig build cross -Doptimize=ReleaseSafe --prefix zig-out/nodist + + - name: Assert asset-free executables are within the size budget + run: | + set -euo pipefail + size_limit=$((10 * 1024 * 1024)) + for triple in x86_64-linux-musl aarch64-linux-musl; do + binary="zig-out/nodist/cross/$triple/nxdns" + if [ ! -f "$binary" ]; then + echo "missing executable: $binary" + exit 1 + fi + case "$triple" in + x86_64-*) strip_tool=objcopy ;; + aarch64-*) strip_tool=aarch64-linux-gnu-objcopy ;; + esac + "$strip_tool" --strip-all "$binary" "$binary.stripped" + size=$(stat -c %s "$binary.stripped") + echo "$triple: stripped size without assets $size bytes" + if [ "$size" -ge "$size_limit" ]; then + echo "stripped asset-free executable exceeds the 10 MiB budget: $binary" + exit 1 + fi + done + docker: runs-on: ubuntu-24.04 diff --git a/build.zig b/build.zig index 6667877..31cecf5 100644 --- a/build.zig +++ b/build.zig @@ -109,6 +109,75 @@ pub fn build(b: *std.Build) void { }); test_step.dependOn(&b.addRunArtifact(blocklist_fuzz_tests).step); + // The bench harness (milestone-12 ruling 1). The measured roots + // (matcher.zig, dns_cache.zig, compiler.zig) share files in their relative + // import closures (model.zig, types.zig, ...), and a file may belong to + // only one module per compilation — separate modules per root cannot link + // into one executable. So one staged module: a copy of src/ plus a + // generated aggregator root, imported by the bench as `core`. No sqlite, + // no mbedTLS: the closure is pure Zig. + const bench_stage = b.addWriteFiles(); + _ = bench_stage.addCopyDirectory(b.path("src"), "src", .{}); + const bench_core = bench_stage.add("bench_core.zig", + \\pub const matcher = @import("src/filter/matcher.zig"); + \\pub const dns_cache = @import("src/cache/dns_cache.zig"); + \\pub const compiler = @import("src/filter/compiler.zig"); + \\pub const model = @import("src/config/model.zig"); + \\pub const dns_name = @import("src/dns/name.zig"); + \\pub const dns_types = @import("src/dns/types.zig"); + \\pub const packet = @import("src/dns/packet.zig"); + \\ + ); + const bench_core_mod = b.createModule(.{ + .root_source_file = bench_core, + .target = target, + .optimize = optimize, + }); + const bench_mod = b.createModule(.{ + .root_source_file = b.path("tools/bench.zig"), + .target = target, + .optimize = optimize, + }); + bench_mod.addImport("core", bench_core_mod); + const bench_exe = b.addExecutable(.{ .name = "bench", .root_module = bench_mod }); + const bench_run = b.addRunArtifact(bench_exe); + if (b.args) |args| bench_run.addArgs(args); + b.step("bench", "Run the performance benchmarks (PLAN §18)").dependOn(&bench_run.step); + + // aarch64 test execution (milestone-12 ruling 6): the plain suite + // cross-built for the deploy target and run under qemu-user + // (`zig build test-aarch64 -fqemu`). Fuzz artifacts stay native-only, and + // -Dintegration stays out (ruling 7): qemu-user's slowdown makes the + // wall-clock-budgeted loopback TLS tests a flake source. + const aarch64_target = b.resolveTargetQuery( + std.Target.Query.parse(.{ .arch_os_abi = "aarch64-linux-musl" }) catch unreachable, + ); + const aarch64_tests = b.addTest(.{ + .root_module = b.createModule(.{ + .root_source_file = b.path("src/tests.zig"), + .target = aarch64_target, + .optimize = optimize, + .link_libc = true, + }), + }); + aarch64_tests.linkage = .static; + aarch64_tests.root_module.addOptions("build_options", options); + aarch64_tests.root_module.linkLibrary(sqliteLibrary(b, aarch64_target, optimize)); + aarch64_tests.root_module.linkLibrary(mbedtlsLibrary(b, aarch64_target, optimize)); + aarch64_tests.root_module.addCSourceFile(.{ .file = b.path("src/platform/mbedtls_shim.c") }); + addMbedtlsThreadingMacros(aarch64_tests.root_module); + aarch64_tests.root_module.addAnonymousImport("test_fixtures", .{ + .root_source_file = b.path("tests/fixtures/fixtures.zig"), + }); + aarch64_tests.root_module.addAnonymousImport("docs_files", .{ + .root_source_file = b.path("docs/docs.zig"), + }); + aarch64_tests.root_module.addAnonymousImport("web_assets", .{ .root_source_file = web_assets }); + const aarch64_run = b.addRunArtifact(aarch64_tests); + aarch64_run.skip_foreign_checks = true; + b.step("test-aarch64", "Run the test suite for aarch64-linux-musl (use -fqemu)") + .dependOn(&aarch64_run.step); + const cross = b.step("cross", "Build static musl executables for every deploy target"); for (cross_targets) |triple| { const query = std.Target.Query.parse(.{ .arch_os_abi = triple }) catch |err| { diff --git a/docs/performance.md b/docs/performance.md new file mode 100644 index 0000000..8a34df8 --- /dev/null +++ b/docs/performance.md @@ -0,0 +1,81 @@ +# Performance + +PLAN §18 sets the targets; `tools/bench.zig` measures the three that are +measurable in-process. Run it with: + +``` +zig build bench -Doptimize=ReleaseFast +``` + +Subcommands `filter|cache|compile|all` (default `all`) select a suite; flags +`--domains=N` (default 1,000,000), `--iters=N` (default 200,000) and `--seed=N` +(default 0x5eed) shape the load. The default run is informational; `--assert` +exits non-zero when a target below is exceeded. + +## Targets (PLAN §18) + +| Target | Where it is checked | +| --- | --- | +| Sustained ≥ 100 qps on Raspberry Pi 5 | End-to-end against the real binary on the Pi (see below); not a harness number | +| Blocklist lookup p95 < 1 ms | `bench filter`: `matcher.normalize` + `Snapshot.evaluate` per op | +| Cached response p95 < 5 ms | `bench cache`: `buildKey` + `DnsCache.get` + `packet.setId` per op | +| Memory with ~1M blocked domains < 100 MB | `bench filter`: VmRSS with the 1M-domain snapshot loaded | +| Stripped static binary < 10 MB per arch (< 15 MB with embedded frontend) | CI size assert on the `cross` artifacts | + +## Measured: x86_64 development host (2026-08-02) + +Intel Core i7-14700K, Linux 6.18, Zig 0.16.0, `-Doptimize=ReleaseFast`, +defaults (1,000,000 domains, 200,000 iterations per suite, seed 0x5eed). +**This is not the target platform** — the Pi 5's Cortex-A76 is far slower and +these numbers do not transfer; they establish the harness works and set a +baseline for regressions on the machine development happens on. + +``` +suite ops p50(us) p95(us) p99(us) max(us) +filter 200000 0.11 0.18 0.27 16.41 + blocked 66699/200000, Snapshot.memoryBytes 28.0 MiB, VmRSS 31.8 MiB + target p95 < 1ms: PASS + target VmRSS < 100 MiB: PASS +cache 200000 0.10 0.14 0.17 3.53 + hits 100000/200000, DnsCache.memoryBytes 4.3 MiB, VmRSS 7.6 MiB + target p95 < 5ms: PASS +compile 1000000 wall 96.025ms, 10413949 lines/s, 1000000 domains kept (informational) +``` + +Every in-process §18 target passes on this host: the two latency targets by +three-to-four orders of magnitude, the memory target by about 3x. + +Two memory figures appear on purpose. `Snapshot.memoryBytes` / +`DnsCache.memoryBytes` are the in-repo accounting of the structures themselves +(the regression guard); VmRSS is what the kernel actually holds resident for +the whole process, allocator slack and code included. The truth sits between +them, and the §18 memory target is judged on VmRSS. The filter suite frees the +generated list source before reading VmRSS, so the number reflects the loaded +snapshot rather than the generator. The cache line's VmRSS is lower because the +filter suite's snapshot has been freed by then. + +## Raspberry Pi 5 (target platform) + +To be measured on hardware. One command, run on the Pi: + +``` +zig build bench -Doptimize=ReleaseFast -- --assert +``` + +| Target | Result | +| --- | --- | +| Blocklist lookup p95 < 1 ms | to be measured on hardware | +| Cached response p95 < 5 ms | to be measured on hardware | +| Memory with ~1M blocked domains < 100 MB | to be measured on hardware | +| Sustained ≥ 100 qps | to be measured on hardware, end-to-end | + +The qps target is end-to-end and belongs to the real binary, not the harness: +run `nxdns run` on the Pi and drive it over the LAN with a DNS load generator +(for example `dnsperf`) against real blocklists. + +## Why CI does not gate on performance + +Required CI stays deterministic (AGENTS.md); latency assertions on shared +runners measure the runner's noisy neighbours, not nxdns, and a perf gate that +flakes trains people to re-run it. The bench exists for hardware you control: +run `--assert` on the Pi, where the numbers mean something. diff --git a/specs/milestone-12.md b/specs/milestone-12.md new file mode 100644 index 0000000..3576280 --- /dev/null +++ b/specs/milestone-12.md @@ -0,0 +1,179 @@ +# Milestone 12: performance measurement pass + aarch64 test execution + +Goal: close the two gaps the post-Phase-10 completeness sweep found — PLAN §18 has no +measurement infrastructure (deferred at specs/milestone-7.md:113 and never delivered), +and aarch64 tests are cross-built but never executed (PLAN.md:145 promised qemu). + +## Rulings (binding) + +1. **Bench harness = `tools/bench.zig` + `zig build bench`.** Not in src/ — src/ is + the shipped product; tests.zig aggregates everything shippable and the bench must + be a separate compilation anyway (one-file-one-module rule; matcher.zig and + dns_cache.zig already belong to the test compilation). Module wiring per the fuzz + pattern (build.zig:68-92): modules rooted at src/filter/matcher.zig and + src/cache/dns_cache.zig — their relative import closures come along; no + sqlite/mbedTLS linking needed. `if (b.args) |args| run.addArgs(args)`. + +2. **What it measures** (PLAN §18 targets): + - `filter`: `matcher.normalize` + `Snapshot.evaluate` per op against a Snapshot + built from a generated, sorted `d{d:0>7}.example.com` list body (~1M lines, + in-memory; DomainSet.max_count 4M and compiler.max_domains 2M leave headroom) + plus a small wild body. Mixed case ratios: hit, miss, parent-walk. Target + p95 < 1 ms. + - `cache`: `buildKey` + `DnsCache.get` + `packet.setId` (the handler's hit path; + TTL aging happens inside get) on a ~10k-entry cache prefilled with the fuzz + corpus response. Mixed hit/miss. Target p95 < 5 ms. + - `compile`: `compiler.compile` over a generated 1M-line hosts body — wall time, + informational (no §18 target; no datapoint exists today). + - Memory: `/proc/self/status` VmRSS (nothing in-repo wraps it; read it directly) + plus the in-repo accounting (`Snapshot.memoryBytes`, `DnsCache.memoryBytes`). + Target < 100 MB RSS with ~1M domains loaded. + - Timing: `std.Io.Clock.awake.now(io)` / `durationTo` — std.time.Timer does not + exist at 0.16. Percentiles: collect per-op nanos in a preallocated []u64, sort + (std.mem.sort), report p50/p95/p99/max. No percentile helper exists; write it + in the tool. + +3. **Assertion policy.** Default run is informational (prints a table). `--assert` + exits non-zero when a §18 target is exceeded — for the Pi 5 run, NOT for CI: + required CI stays deterministic (AGENTS.md) and perf assertions on shared runners + are flake generators. CI does not run the bench at all this milestone. + +4. **Bench flags**: `--domains=N` (default 1_000_000), `--iters=N` (default 200_000), + `--seed=N` (default fixed), `--assert`, subcommands `filter|cache|compile|all` + (default all). Debug-mode runs print a one-line warning recommending + `-Doptimize=ReleaseFast`. + +5. **docs/performance.md**: the §18 targets table; measured numbers from THIS x86_64 + host (dated, hardware named honestly, marked as not-the-target-platform); the + in-memory accounting numbers; the one-command Pi 5 recipe + (`zig build bench -Doptimize=ReleaseFast -- --assert`); a sentence on why CI does + not gate on perf. The Pi 5 rows stay "to be measured on hardware". + +6. **aarch64 execution = second test artifact, `zig build test-aarch64 -fqemu`.** + Per stdlib evidence: Run steps try qemu only when enable_qemu (the `-fqemu` CLI + flag; a *Build field, no per-step override) and exec `qemu-aarch64` bare from + PATH — so CI installs `qemu-user` (NOT qemu-user-static, which ships the + `-static` name Zig will not find; binfmt is unnecessary). Static musl means no + sysroot/--libc-runtimes. build.zig: resolve aarch64-linux-musl, addTest with the + same wiring as the native tests artifact (sqlite/mbedtls helpers are already + target-parameterized), set `.linkage = .static` on the artifact (TestOptions has + no linkage field), `run.skip_foreign_checks = true`, leave + failing_to_execute_foreign_is_an_error true so a missing qemu is loud. Fuzz + artifacts excluded. + +7. **qemu scope: plain suite only, blocking.** No `-Dintegration` under qemu: the + integration tests are multithreaded loopback TLS with wall-clock budgets, and + qemu-user's 5-20x slowdown makes them a flake source — required CI stays + deterministic. The plain suite (the aggregator's tests without the 12 + fuzz-artifact tests: 1159 pass + 113 integration-gated skips; all pure + DNS/filter/cache logic included) + is the portable-correctness signal aarch64 needs. Recorded here as the + engineering call closing PLAN.md:145's promise. + +8. **CI additions**: one `test-aarch64` job (setup-zig, apt qemu-user, + `zig build test-aarch64 -fqemu`); plus the missing §18 size assert — the cross + job gains a second build WITHOUT `-Dweb-dist` (placeholder dist) and asserts the + stripped copies < 10 MiB per arch (the < 15 MiB with-assets assert already + exists). Job graph otherwise untouched. + +9. **No src/ changes.** matcher/dns_cache/compiler public APIs are used as-is; if + the bench needs something they do not expose, report — do not extend them. + +## Sessions + +V1 then V2 (both edit build.zig — sequenced, not parallel). + +## Session V1: bench harness + performance doc + +Owns tools/bench.zig, build.zig (the bench step only), docs/performance.md. +Rulings 1-5. Runs the bench on this host (ReleaseFast) and writes the measured +numbers into the doc. + +### V1 As built + +Delivered tools/bench.zig, the build.zig bench step, and docs/performance.md. One +deviation from ruling 1's wiring: Zig 0.16 rejects separate modules rooted at +matcher.zig and dns_cache.zig ("file exists in multiple modules" — their closures +share model.zig/types.zig; verified with a minimal repro). Instead a b.addWriteFiles +stage copies src/ plus a generated 7-line aggregator root (bench_core.zig) into one +`core` module — the same staging trick the web assets use; a build.zig comment +records why. The fuzz-corpus response is a byte-for-byte copy with provenance +comment (importing corpus.zig recreates the module conflict; the corpus documents +copies as house style). /proc/self/status reads use readerStreaming (procfs stats +size 0, readFileAlloc returns empty). Measured on the build host (i7-14700K, +ReleaseFast, defaults: 1M domains, 200k iters): filter p95 0.18 µs (target < 1 ms), +cache p95 0.14 µs (target < 5 ms), VmRSS 31.8 MiB with the 1M-domain snapshot +(target < 100 MiB), Snapshot.memoryBytes 28.0 MiB, compile 96 ms ≈ 10.4M lines/s +(informational). --assert exit paths verified both ways (a 4M-domain run exceeds +the 1M-scoped RSS target and exits 1; that was an exit-path exercise, not a target +miss). No src/ changes; matcher/dns_cache/compiler APIs sufficed. Cache suite fixes +entries at 10k (the handler-default Config.size); --domains shapes filter and +compile only. Both suites 0 failed after the build.zig edit. + +## Session V2: aarch64 test execution + CI (after V1) + +Owns build.zig (the test-aarch64 artifact/step only), .gitea/workflows/ci.yml +(the new job + the cross-job size-assert addition). Rulings 6-8. Runs +`zig build test-aarch64 -fqemu` locally if qemu-aarch64 is installable/present; +reports honestly if not. + +### V2 As built + +build.zig gained the test-aarch64 block after the bench block: aarch64-linux-musl +addTest with wiring identical to the native tests artifact, `.linkage = .static` on +the artifact, `run.skip_foreign_checks = true`, failing-to-execute left loud. Stdlib +mechanics re-verified (Run.zig:71/78/226, Build.zig:73, system.zig:111 — bare +`qemu-aarch64` from PATH). ci.yml gained the test-aarch64 job (apt qemu-user, +`zig build test-aarch64 -fqemu`) and the cross job builds a second no-dist pair to +`--prefix zig-out/nodist` (existing with-assets steps byte-identical) with a +< 10 MiB stripped assert; the new assert skips the static-linkage recheck (proven +on the same-config with-assets build). Local verification: real qemu execution — +1159 pass / 113 skip / 0 failed (native count minus the 12 excluded fuzz tests); +no-dist stripped sizes 5,880,336 (x86_64) and 5,212,640 (aarch64) bytes. No +-Dintegration under qemu per ruling 7. + +## Review (Codex, as built) + +Two rounds on one thread; round 2 returned "No findings." + +Round 1 (2 important, 2 minor): bench.zig's explicit body.deinit left the earlier +errdefer armed on an undefined list (double-free on any later error) → clearAndFree +keeps the list valid. Ubuntu's qemu-user recommends qemu-user-binfmt, which apt +installs by default — binfmt registration would bypass the -fqemu path CI intends +to exercise → --no-install-recommends. performance.md's headroom claim overstated +the memory margin (3x, not orders of magnitude) → corrected. Ruling 7 quoted the +native aggregate test count instead of the aarch64 artifact's own (1159 + 113 +skips, fuzz excluded) → corrected. + +Final gates: plain 1171/1284 passed, 113 skipped (integration-gated), 0 failed; +integration 1280/1284, 4 skipped (live-network by design), 0 failed; qemu aarch64 +run 1159 passed, 113 skipped, 0 failed; cross ReleaseSafe with dist 18/18; no-dist +stripped sizes 5.9/5.2 MB under the 10 MiB assert. + +## Module layout (new) + +tools/bench.zig, docs/performance.md. + +## File ownership + +V1 tools/bench.zig + docs/performance.md + build.zig(bench); V2 build.zig(aarch64) + +ci.yml. Orchestrator: spec sync, review, commit. + +## Acceptance (milestone complete) + +- [ ] `zig build bench -Doptimize=ReleaseFast` runs all three suites and prints + p50/p95/p99/max + memory; `--assert` enforces the §18 targets. +- [ ] docs/performance.md holds dated x86_64 numbers + the Pi 5 recipe. +- [ ] `zig build test-aarch64 -fqemu` passes locally under qemu (or its + unavailability is recorded with the exact CI-equivalent command). +- [ ] CI: test-aarch64 job green-by-construction (same command CI runs); cross job + asserts < 10 MiB stripped without assets, < 15 MiB with (existing). +- [ ] Both existing suites 0 failed; no src/ changes. + +## Anti-requirements + +- No perf assertions in CI; no bench in the test step. +- No qemu integration suite; no binfmt setup; no qemu-user-static. +- No new pub API on matcher/dns_cache/compiler; no synthetic-load DNS server + benchmark (the ≥100 qps target is end-to-end on the Pi — the operator recipe + covers it via the real binary, not a harness). diff --git a/tools/bench.zig b/tools/bench.zig new file mode 100644 index 0000000..b70cd16 --- /dev/null +++ b/tools/bench.zig @@ -0,0 +1,382 @@ +//! Performance bench harness (PLAN §18, milestone-12 rulings 1-4). +//! +//! `zig build bench -Doptimize=ReleaseFast -- [filter|cache|compile|all] [flags]` +//! +//! Flags: `--domains=N` (default 1_000_000), `--iters=N` (default 200_000), +//! `--seed=N` (default 0x5eed), `--assert`. The default run is informational; +//! `--assert` exits non-zero when a §18 target is exceeded — meant for the +//! Pi 5 acceptance run, never for CI (required CI stays deterministic). +//! +//! What each suite measures: +//! - `filter`: `matcher.normalize` + `Snapshot.evaluate` per op — the handler's +//! filtering work — against a snapshot built from `--domains` generated exact +//! entries plus a small wildcard body. Query mix cycles hit, miss and +//! parent-walk. Target p95 < 1 ms; VmRSS < 100 MiB with the list loaded. +//! - `cache`: `buildKey` + `DnsCache.get` + `packet.setId` — the handler's +//! cache-hit path, TTL aging included — on a 10k-entry cache prefilled with a +//! realistic response. Query mix alternates hit and miss. Target p95 < 5 ms. +//! - `compile`: `compiler.compile` over `--domains` generated hosts lines. +//! Wall time, informational (no §18 target). + +const std = @import("std"); +const builtin = @import("builtin"); +const core = @import("core"); +const matcher = core.matcher; +const dns_cache = core.dns_cache; +const compiler = core.compiler; +const model = core.model; +const dns_name = core.dns_name; +const dns_types = core.dns_types; +const packet = core.packet; + +const Allocator = std.mem.Allocator; +const Writer = std.Io.Writer; + +const filter_p95_target_ns: u64 = 1 * std.time.ns_per_ms; +const cache_p95_target_ns: u64 = 5 * std.time.ns_per_ms; +const rss_target_bytes: usize = 100 * 1024 * 1024; + +const cache_entries: u32 = 10_000; + +/// Byte-for-byte copy of `response` in tests/fuzz/corpus.zig (a copy on +/// purpose, same as the corpus itself: a bench input that changes whenever a +/// test fixture is edited is a benchmark that silently shifts). A CNAME to +/// www.example.com (TTL 300) plus its A record (TTL 60) and an OPT record, +/// so `classify` stores it as a positive entry with a 60 s lifetime. +const cached_response = + "\x12\x34\x81\x80\x00\x01\x00\x02\x00\x00\x00\x01" ++ + "\x07example\x03com\x00\x00\x01\x00\x01" ++ + "\xc0\x0c\x00\x05\x00\x01\x00\x00\x01\x2c\x00\x06\x03www\xc0\x0c" ++ + "\xc0\x29\x00\x01\x00\x01\x00\x00\x00\x3c\x00\x04\x5d\xb8\xd8\x22" ++ + "\x00\x00\x29\x10\x00\x00\x00\x00\x00\x00\x00"; + +/// Strictly ascending, so `DomainSet.build` accepts it. Never matched by the +/// generated queries: the wildcards exist to be walked past, the way a real +/// snapshot's wildcard set is on most queries. +const wild_body = "ads.bench.invalid\nmetrics.bench.invalid\ntelemetry.bench.invalid\n"; + +const usage = + "usage: zig build bench -Doptimize=ReleaseFast -- " ++ + "[filter|cache|compile|all] [--domains=N] [--iters=N] [--seed=N] [--assert]"; + +const Suite = enum { filter, cache, compile, all }; + +const Options = struct { + suite: Suite = .all, + domains: u32 = 1_000_000, + iters: u32 = 200_000, + seed: u64 = 0x5eed, + assert: bool = false, +}; + +pub fn main(init: std.process.Init) !u8 { + const arena = init.arena.allocator(); + const gpa = init.gpa; + const io = init.io; + + const args = try init.minimal.args.toSlice(arena); + const opts = parseOptions(args); + + var out_buffer: [4096]u8 = undefined; + var out = std.Io.File.stdout().writer(io, &out_buffer); + const w = &out.interface; + + if (builtin.mode == .Debug) { + try w.print("warning: Debug build; run with -Doptimize=ReleaseFast for meaningful numbers\n", .{}); + } + try w.print("nxdns bench suite={t} domains={d} iters={d} seed=0x{x} optimize={t}\n\n", .{ + opts.suite, opts.domains, opts.iters, opts.seed, builtin.mode, + }); + try w.print("{s:<9}{s:>10}{s:>12}{s:>12}{s:>12}{s:>12}\n", .{ + "suite", "ops", "p50(us)", "p95(us)", "p99(us)", "max(us)", + }); + + var exceeded: u32 = 0; + if (opts.suite == .filter or opts.suite == .all) exceeded += try runFilter(io, gpa, opts, w); + if (opts.suite == .cache or opts.suite == .all) exceeded += try runCache(io, gpa, opts, w); + if (opts.suite == .compile or opts.suite == .all) try runCompile(io, gpa, opts, w); + + if (exceeded > 0) try w.print("\n{d} target(s) exceeded\n", .{exceeded}); + try w.flush(); + return if (opts.assert and exceeded > 0) 1 else 0; +} + +fn parseOptions(args: []const [:0]const u8) Options { + var opts: Options = .{}; + for (args[1..]) |arg| { + if (std.mem.eql(u8, arg, "--assert")) { + opts.assert = true; + } else if (std.mem.startsWith(u8, arg, "--domains=")) { + opts.domains = parseNumber(u32, arg, "--domains="); + } else if (std.mem.startsWith(u8, arg, "--iters=")) { + opts.iters = parseNumber(u32, arg, "--iters="); + } else if (std.mem.startsWith(u8, arg, "--seed=")) { + opts.seed = parseNumber(u64, arg, "--seed="); + } else if (std.meta.stringToEnum(Suite, arg)) |suite| { + opts.suite = suite; + } else { + std.process.fatal("unknown argument '{s}'\n{s}", .{ arg, usage }); + } + } + if (opts.iters == 0) std.process.fatal("--iters must be at least 1", .{}); + if (opts.domains == 0) std.process.fatal("--domains must be at least 1", .{}); + // Seven zero-padded digits keep generation order equal to sorted order; + // the DomainSet cap is lower anyway. + if (opts.domains > 4_000_000) std.process.fatal("--domains must be at most 4000000", .{}); + return opts; +} + +fn parseNumber(comptime T: type, arg: []const u8, prefix: []const u8) T { + return std.fmt.parseInt(T, arg[prefix.len..], 10) catch { + std.process.fatal("bad value in '{s}'\n{s}", .{ arg, usage }); + }; +} + +// --------------------------------------------------------------------------- +// Suites +// --------------------------------------------------------------------------- + +/// Returns how many §18 targets the suite exceeded. +fn runFilter(io: std.Io, gpa: Allocator, opts: Options, w: *Writer) !u32 { + var body: std.ArrayList(u8) = .empty; + errdefer body.deinit(gpa); + try body.ensureTotalCapacity(gpa, @as(usize, opts.domains) * 20); + var line: [64]u8 = undefined; + for (0..opts.domains) |i| { + const text = std.fmt.bufPrint(&line, "d{d:0>7}.example.com\n", .{i}) catch unreachable; + try body.appendSlice(gpa, text); + } + + const sources = [_]model.BlocklistSource{.{ .url = "bench://list", .name = "bench" }}; + const links = [_]model.GroupSource{.{ .group = "default", .source_url = "bench://list" }}; + var snapshot = try matcher.Snapshot.build(gpa, .{ + .groups = &.{.{ .name = "default" }}, + .group_ids = &.{1}, + .group_sources = &links, + .sources = &sources, + .source_ids = &.{1}, + .rules = &.{}, + .clients = &.{}, + .prefixes = &.{}, + .compiled = &.{.{ .list_body = body.items, .wild_body = wild_body }}, + .seed = opts.seed, + .generation = 1, + }); + defer snapshot.deinit(); + // The snapshot copied everything it needs; freeing the source body before + // the RSS read keeps the memory number about the loaded snapshot. + // clearAndFree leaves the list valid so the errdefer above stays safe. + body.clearAndFree(gpa); + + var prng = std.Random.DefaultPrng.init(opts.seed); + const random = prng.random(); + const pool = try gpa.alloc(dns_name.Name, 4096); + defer gpa.free(pool); + for (pool, 0..) |*entry, i| { + const r = random.uintLessThan(u32, opts.domains); + const text = switch (i % 3) { + 0 => std.fmt.bufPrint(&line, "d{d:0>7}.example.com", .{r}), + 1 => std.fmt.bufPrint(&line, "m{d:0>7}.example.org", .{r}), + else => std.fmt.bufPrint(&line, "a.b.d{d:0>7}.example.com", .{r}), + } catch unreachable; + entry.* = dns_name.fromText(text) catch unreachable; + } + + const samples = try gpa.alloc(u64, opts.iters); + defer gpa.free(samples); + + var buf: [dns_types.max_name_len]u8 = undefined; + for (pool) |qname| { + std.mem.doNotOptimizeAway(snapshot.evaluate(0, matcher.normalize(qname, &buf)).blocked); + } + + var blocked: u64 = 0; + for (samples, 0..) |*sample, i| { + const qname = pool[i % pool.len]; + const t0 = std.Io.Clock.awake.now(io); + const domain = matcher.normalize(qname, &buf); + const decision = snapshot.evaluate(0, domain); + const t1 = std.Io.Clock.awake.now(io); + sample.* = @intCast(@max(0, t0.durationTo(t1).toNanoseconds())); + if (decision.blocked) blocked += 1; + } + if (blocked == 0) std.process.fatal("filter bench blocked nothing; the suite is broken", .{}); + + const pct = percentiles(samples); + const rss = vmRssBytes(io); + try printRow(w, "filter", opts.iters, pct); + try w.print(" blocked {d}/{d}, Snapshot.memoryBytes {d:.1} MiB, VmRSS {d:.1} MiB\n", .{ + blocked, opts.iters, mib(snapshot.memoryBytes()), mib(rss), + }); + + var exceeded: u32 = 0; + exceeded += try printTarget(w, "p95 < 1ms", pct.p95 < filter_p95_target_ns); + exceeded += try printTarget(w, "VmRSS < 100 MiB", rss < rss_target_bytes); + return exceeded; +} + +fn runCache(io: std.Io, gpa: Allocator, opts: Options, w: *Writer) !u32 { + var cache = try dns_cache.DnsCache.init(gpa, .{ + .size = cache_entries, + .negative_ttl_max = 3600, + }); + defer cache.deinit(); + + const class = dns_cache.classify(cached_response, 3600) orelse { + std.process.fatal("cache bench response is not cacheable; the suite is broken", .{}); + }; + + const filled_at: i64 = 1_000_000; + var key_buf: [dns_cache.max_key_len]u8 = undefined; + var text_buf: [64]u8 = undefined; + for (0..cache_entries) |i| { + const qname = std.fmt.bufPrint(&text_buf, "c{d:0>5}.example.com", .{i}) catch unreachable; + const key = dns_cache.buildKey(&key_buf, qname, 1, 1, false, null); + try cache.put(filled_at, key, cached_response, class); + } + + const samples = try gpa.alloc(u64, opts.iters); + defer gpa.free(samples); + + var prng = std.Random.DefaultPrng.init(opts.seed); + const random = prng.random(); + // Inside the entry's 60 s lifetime, far enough in to make `get` age TTLs. + const queried_at = filled_at + 30; + var out_buf: [512]u8 = undefined; + var hits: u64 = 0; + for (samples, 0..) |*sample, i| { + const r = random.uintLessThan(u32, cache_entries); + const qname = if (i % 2 == 0) + std.fmt.bufPrint(&text_buf, "c{d:0>5}.example.com", .{r}) catch unreachable + else + std.fmt.bufPrint(&text_buf, "x{d:0>5}.example.org", .{r}) catch unreachable; + + const t0 = std.Io.Clock.awake.now(io); + const key = dns_cache.buildKey(&key_buf, qname, 1, 1, false, null); + const found = cache.get(queried_at, key, &out_buf); + if (found) |bytes| packet.setId(bytes, @truncate(i)); + const t1 = std.Io.Clock.awake.now(io); + sample.* = @intCast(@max(0, t0.durationTo(t1).toNanoseconds())); + if (found != null) hits += 1; + } + if (hits == 0) std.process.fatal("cache bench hit nothing; the suite is broken", .{}); + + const pct = percentiles(samples); + const rss = vmRssBytes(io); + try printRow(w, "cache", opts.iters, pct); + try w.print(" hits {d}/{d}, DnsCache.memoryBytes {d:.1} MiB, VmRSS {d:.1} MiB\n", .{ + hits, opts.iters, mib(cache.memoryBytes()), mib(rss), + }); + return try printTarget(w, "p95 < 5ms", pct.p95 < cache_p95_target_ns); +} + +fn runCompile(io: std.Io, gpa: Allocator, opts: Options, w: *Writer) !void { + if (opts.domains > compiler.max_domains) { + std.process.fatal("compile suite needs --domains <= {d}", .{compiler.max_domains}); + } + + var body: std.ArrayList(u8) = .empty; + defer body.deinit(gpa); + try body.ensureTotalCapacity(gpa, @as(usize, opts.domains) * 28); + var line: [64]u8 = undefined; + for (0..opts.domains) |i| { + const text = std.fmt.bufPrint(&line, "0.0.0.0 d{d:0>7}.example.com\n", .{i}) catch unreachable; + try body.appendSlice(gpa, text); + } + + var reader = std.Io.Reader.fixed(body.items); + var list_buf: [4096]u8 = undefined; + var wild_buf: [4096]u8 = undefined; + var list_out: Writer.Discarding = .init(&list_buf); + var wild_out: Writer.Discarding = .init(&wild_buf); + + const t0 = std.Io.Clock.awake.now(io); + const result = compiler.compile(gpa, &reader, .hosts, &list_out.writer, &wild_out.writer) catch |err| { + std.process.fatal("compiler.compile failed: {t}", .{err}); + }; + const t1 = std.Io.Clock.awake.now(io); + + if (result.counts.domains != opts.domains) { + std.process.fatal("compile kept {d} of {d} domains; the suite is broken", .{ + result.counts.domains, opts.domains, + }); + } + + const elapsed_ns: u64 = @intCast(@max(1, t0.durationTo(t1).toNanoseconds())); + const lines_per_s = @as(u64, opts.domains) * std.time.ns_per_s / elapsed_ns; + try w.print("{s:<9}{d:>10} wall {f}, {d} lines/s, {d} domains kept (informational)\n", .{ + "compile", opts.domains, std.Io.Duration.fromNanoseconds(@intCast(elapsed_ns)), lines_per_s, + result.counts.domains, + }); +} + +// --------------------------------------------------------------------------- +// Reporting helpers +// --------------------------------------------------------------------------- + +const Percentiles = struct { p50: u64, p95: u64, p99: u64, max: u64 }; + +/// Nearest-rank percentiles over per-op nanoseconds. Sorts `samples` in place. +fn percentiles(samples: []u64) Percentiles { + std.debug.assert(samples.len > 0); + std.mem.sort(u64, samples, {}, std.sort.asc(u64)); + return .{ + .p50 = atRank(samples, 50), + .p95 = atRank(samples, 95), + .p99 = atRank(samples, 99), + .max = samples[samples.len - 1], + }; +} + +fn atRank(sorted: []const u64, pct: usize) u64 { + const rank = (sorted.len * pct + 99) / 100; + return sorted[@max(rank, 1) - 1]; +} + +fn printRow(w: *Writer, suite: []const u8, ops: u32, pct: Percentiles) !void { + try w.print("{s:<9}{d:>10}{d:>12.2}{d:>12.2}{d:>12.2}{d:>12.2}\n", .{ + suite, ops, us(pct.p50), us(pct.p95), us(pct.p99), us(pct.max), + }); +} + +fn printTarget(w: *Writer, target: []const u8, ok: bool) !u32 { + try w.print(" target {s}: {s}\n", .{ target, if (ok) "PASS" else "FAIL" }); + return @intFromBool(!ok); +} + +fn us(ns: u64) f64 { + return @as(f64, @floatFromInt(ns)) / @as(f64, std.time.ns_per_us); +} + +fn mib(bytes: usize) f64 { + return @as(f64, @floatFromInt(bytes)) / (1024.0 * 1024.0); +} + +/// The kernel's resident-set figure, since nothing in-repo wraps it. In-repo +/// accounting (`Snapshot.memoryBytes`, `DnsCache.memoryBytes`) is reported +/// alongside; the two bracket the truth from below and above. +fn vmRssBytes(io: std.Io) usize { + var file = std.Io.Dir.cwd().openFile(io, "/proc/self/status", .{}) catch |err| { + std.process.fatal("cannot open /proc/self/status: {t}", .{err}); + }; + defer file.close(io); + + // procfs reports a zero size to stat, so the size-aware alloc readers see + // an instant end-of-stream; a plain streaming read does not. + var reader_buf: [64]u8 = undefined; + var reader = file.readerStreaming(io, &reader_buf); + var status_buf: [8192]u8 = undefined; + const len = reader.interface.readSliceShort(&status_buf) catch |err| { + std.process.fatal("cannot read /proc/self/status: {t}", .{err}); + }; + const status = status_buf[0..len]; + + var lines = std.mem.splitScalar(u8, status, '\n'); + while (lines.next()) |status_line| { + if (!std.mem.startsWith(u8, status_line, "VmRSS:")) continue; + var fields = std.mem.tokenizeAny(u8, status_line["VmRSS:".len..], " \t"); + const kib = fields.next() orelse break; + return 1024 * (std.fmt.parseInt(usize, kib, 10) catch break); + } + std.process.fatal("no VmRSS line in /proc/self/status", .{}); +}