db-mode config changes apply live in-process
settings and upstream writes now follow a prepare, commit, publish, retire contract: candidates are built and validated before the database transaction, published as infallible pointer swaps, and old generations retire after their readers drain. per-query policy values snapshot once per query; upstream pool, cache, rate limiter, sessions, api limiter, log sink, blocklist scheduler and the query-log queue each gained one named live operation. restart_required shrinks from every scalar key to the bind keys and web.enabled; the admin ui drops its restart notices for everything else. file mode is unchanged.
This commit is contained in:
@@ -33,11 +33,65 @@ pub fn classify(free_bytes: u64, cfg: model.Disk) State {
|
||||
return .ok;
|
||||
}
|
||||
|
||||
/// `min_free_mb` and `warn_free_mb` are one invariant pair — `classify` reads
|
||||
/// both and reports the more severe verdict — so they live in one atomic word
|
||||
/// and a reader unpacks a single load. Two atomics would let a sample land
|
||||
/// between the two stores and classify against half of one configuration and
|
||||
/// half of another.
|
||||
fn packThresholds(d: model.Disk) u64 {
|
||||
return (@as(u64, d.min_free_mb) << 32) | d.warn_free_mb;
|
||||
}
|
||||
|
||||
fn unpackThresholds(bits: u64) model.Disk {
|
||||
return .{
|
||||
.min_free_mb = @truncate(bits >> 32),
|
||||
.warn_free_mb = @truncate(bits),
|
||||
};
|
||||
}
|
||||
|
||||
/// Where the measured log directory comes from. `sample` borrows the path
|
||||
/// across a directory scan, so the path cannot simply be replaced under it: a
|
||||
/// reader pins a generation for the whole borrow and `setLogDir` retires the
|
||||
/// old one, which is freed by whichever of the two — the last reader or the
|
||||
/// setter — finds it retired with no refs.
|
||||
pub const LogDirSource = union(enum) {
|
||||
/// The path `init` was given, borrowed from the config. It outlives the
|
||||
/// process, so a reader still holding it after a swap is safe and it needs
|
||||
/// no pin. Null means logs do not go to a file.
|
||||
boot: ?[:0]const u8,
|
||||
/// Every generation `setLogDir` installs. Null means the same as above.
|
||||
installed: ?*LogDir,
|
||||
};
|
||||
|
||||
/// A reader's hold on the log directory for the length of one scan. `pinned`
|
||||
/// is null for the boot source, which nothing frees.
|
||||
pub const LogDirBorrow = struct {
|
||||
path: ?[:0]const u8,
|
||||
pinned: ?*LogDir,
|
||||
};
|
||||
|
||||
pub const LogDir = struct {
|
||||
path: [:0]const u8,
|
||||
refs: u32 = 0,
|
||||
retired: bool = false,
|
||||
/// Non-null exactly for heap generations, and the allocator that frees
|
||||
/// them.
|
||||
gpa: ?std.mem.Allocator = null,
|
||||
|
||||
fn destroy(self: *LogDir) void {
|
||||
const gpa = self.gpa orelse return;
|
||||
gpa.free(self.path);
|
||||
gpa.destroy(self);
|
||||
}
|
||||
};
|
||||
|
||||
pub const Monitor = struct {
|
||||
cfg: model.Disk,
|
||||
thresholds_packed: std.atomic.Value(u64),
|
||||
data_dir: std.Io.Dir,
|
||||
data_path: [:0]const u8,
|
||||
log_dir_path: ?[:0]const u8,
|
||||
/// Guards `log_dir` and every generation's `refs`/`retired`.
|
||||
log_dir_mutex: std.Io.Mutex,
|
||||
log_dir: LogDirSource,
|
||||
|
||||
state_raw: std.atomic.Value(u8),
|
||||
free_bytes: std.atomic.Value(u64),
|
||||
@@ -57,10 +111,11 @@ pub const Monitor = struct {
|
||||
log_dir_path: ?[:0]const u8,
|
||||
) Monitor {
|
||||
return .{
|
||||
.cfg = cfg,
|
||||
.thresholds_packed = .init(packThresholds(cfg)),
|
||||
.data_dir = data_dir,
|
||||
.data_path = data_path,
|
||||
.log_dir_path = log_dir_path,
|
||||
.log_dir_mutex = .init,
|
||||
.log_dir = .{ .boot = log_dir_path },
|
||||
.state_raw = .init(@intFromEnum(State.ok)),
|
||||
.free_bytes = .init(0),
|
||||
.db_bytes = .init(0),
|
||||
@@ -69,6 +124,90 @@ pub const Monitor = struct {
|
||||
};
|
||||
}
|
||||
|
||||
/// The live threshold pair, from one load: `warn >= min` holds for every
|
||||
/// value this ever returns, whatever a concurrent `setThresholds` does.
|
||||
pub fn thresholds(self: *const Monitor) model.Disk {
|
||||
return unpackThresholds(self.thresholds_packed.load(.monotonic));
|
||||
}
|
||||
|
||||
pub fn setThresholds(self: *Monitor, d: model.Disk) void {
|
||||
self.thresholds_packed.store(packThresholds(d), .monotonic);
|
||||
}
|
||||
|
||||
/// Pins the log directory for one scan. Every borrow is matched by a
|
||||
/// `releaseLogDir`, which is what lets `setLogDir` free a generation the
|
||||
/// moment no scan is reading its path.
|
||||
pub fn acquireLogDir(self: *Monitor, io: std.Io) LogDirBorrow {
|
||||
self.log_dir_mutex.lockUncancelable(io);
|
||||
defer self.log_dir_mutex.unlock(io);
|
||||
|
||||
switch (self.log_dir) {
|
||||
.boot => |path| return .{ .path = path, .pinned = null },
|
||||
.installed => |maybe| {
|
||||
const gen = maybe orelse return .{ .path = null, .pinned = null };
|
||||
gen.refs += 1;
|
||||
return .{ .path = gen.path, .pinned = gen };
|
||||
},
|
||||
}
|
||||
}
|
||||
|
||||
pub fn releaseLogDir(self: *Monitor, io: std.Io, borrow: LogDirBorrow) void {
|
||||
const gen = borrow.pinned orelse return;
|
||||
|
||||
self.log_dir_mutex.lockUncancelable(io);
|
||||
std.debug.assert(gen.refs > 0);
|
||||
gen.refs -= 1;
|
||||
const free_it = gen.retired and gen.refs == 0;
|
||||
self.log_dir_mutex.unlock(io);
|
||||
|
||||
if (free_it) gen.destroy();
|
||||
}
|
||||
|
||||
/// Prepare half of a log-directory change: allocates the owned path and
|
||||
/// its generation node before any commit, so publish cannot fail. `path`
|
||||
/// null means logs no longer go to a file and nothing is measured.
|
||||
pub fn prepareLogDir(
|
||||
gpa: std.mem.Allocator,
|
||||
path: ?[]const u8,
|
||||
) std.mem.Allocator.Error!?*LogDir {
|
||||
const p = path orelse return null;
|
||||
const owned = try gpa.dupeZ(u8, p);
|
||||
errdefer gpa.free(owned);
|
||||
const gen = try gpa.create(LogDir);
|
||||
gen.* = .{ .path = owned, .gpa = gpa };
|
||||
return gen;
|
||||
}
|
||||
|
||||
/// Discards a generation `prepareLogDir` built that will not be published.
|
||||
pub fn destroyPreparedLogDir(prepared: ?*LogDir) void {
|
||||
if (prepared) |gen| gen.destroy();
|
||||
}
|
||||
|
||||
/// Publish half: infallible and I/O-free. The old generation is retired
|
||||
/// and freed here when no scan holds it, or by the last release otherwise.
|
||||
pub fn setLogDir(self: *Monitor, io: std.Io, prepared: ?*LogDir) void {
|
||||
self.log_dir_mutex.lockUncancelable(io);
|
||||
const old: ?*LogDir = switch (self.log_dir) {
|
||||
.boot => null,
|
||||
.installed => |maybe| maybe,
|
||||
};
|
||||
self.log_dir = .{ .installed = prepared };
|
||||
var free_old = false;
|
||||
if (old) |gen| {
|
||||
gen.retired = true;
|
||||
free_old = gen.refs == 0;
|
||||
}
|
||||
self.log_dir_mutex.unlock(io);
|
||||
|
||||
if (free_old) old.?.destroy();
|
||||
}
|
||||
|
||||
/// Frees any installed log-directory generation. Every scan must have
|
||||
/// released first, which shutdown ordering guarantees.
|
||||
pub fn deinit(self: *Monitor, io: std.Io) void {
|
||||
self.setLogDir(io, null);
|
||||
}
|
||||
|
||||
pub fn state(self: *const Monitor) State {
|
||||
return @enumFromInt(self.state_raw.load(.monotonic));
|
||||
}
|
||||
@@ -109,7 +248,9 @@ pub const Monitor = struct {
|
||||
probeFailed(store, io, now_s, "data_dir", "sizing the data directory failed", err);
|
||||
}
|
||||
|
||||
if (self.log_dir_path) |path| {
|
||||
const borrow = self.acquireLogDir(io);
|
||||
defer self.releaseLogDir(io, borrow);
|
||||
if (borrow.path) |path| {
|
||||
if (self.sumLogDir(io, path)) |bytes| {
|
||||
self.log_bytes.store(bytes, .monotonic);
|
||||
if (store) |s| s.resolve(io, now_s, .disk_probe, "log_dir");
|
||||
@@ -120,7 +261,7 @@ pub const Monitor = struct {
|
||||
}
|
||||
}
|
||||
|
||||
self.publish(io, store, now_s, classify(free, self.cfg), free);
|
||||
self.publish(io, store, now_s, classify(free, self.thresholds()), free);
|
||||
}
|
||||
|
||||
/// Sample first, then sleep: a process that starts on a full disk must not
|
||||
@@ -466,7 +607,7 @@ test "a threshold above the real free space drives the state to critical" {
|
||||
try testing.expectEqual(State.critical, monitor.state());
|
||||
try testing.expect(!monitor.writesAllowed());
|
||||
|
||||
monitor.cfg = .{ .min_free_mb = 0, .warn_free_mb = 0 };
|
||||
monitor.setThresholds(.{ .min_free_mb = 0, .warn_free_mb = 0 });
|
||||
monitor.sample(io, null, 0);
|
||||
try testing.expectEqual(State.ok, monitor.state());
|
||||
try testing.expect(monitor.writesAllowed());
|
||||
@@ -509,7 +650,7 @@ test "a disk transition records an episode per severity and closes it on recover
|
||||
monitor.sample(io, &fx.store, 1060);
|
||||
try testing.expectEqual(@as(i64, 1), try fx.count("SELECT count(*) FROM operational_events"));
|
||||
|
||||
monitor.cfg = .{ .min_free_mb = 0, .warn_free_mb = 0 };
|
||||
monitor.setThresholds(.{ .min_free_mb = 0, .warn_free_mb = 0 });
|
||||
monitor.sample(io, &fx.store, 1120);
|
||||
try testing.expectEqual(State.ok, monitor.state());
|
||||
try testing.expectEqual(
|
||||
@@ -551,7 +692,9 @@ test "a failed probe opens an episode the next clean pass closes" {
|
||||
|
||||
try tmp.dir.createDirPath(io, "logs");
|
||||
var path_buf: [256]u8 = undefined;
|
||||
monitor.log_dir_path = try std.fmt.bufPrintZ(&path_buf, ".zig-cache/tmp/{s}/logs", .{tmp.sub_path});
|
||||
const good_dir = try std.fmt.bufPrint(&path_buf, ".zig-cache/tmp/{s}/logs", .{tmp.sub_path});
|
||||
monitor.setLogDir(io, try Monitor.prepareLogDir(testing.allocator, good_dir));
|
||||
defer monitor.deinit(io);
|
||||
monitor.sample(io, &fx.store, 1100);
|
||||
|
||||
try testing.expectEqual(
|
||||
@@ -560,6 +703,37 @@ test "a failed probe opens an episode the next clean pass closes" {
|
||||
);
|
||||
}
|
||||
|
||||
/// Alternates between two pairs that each satisfy `warn >= min`, so any
|
||||
/// observed pair violating it can only have been torn out of two stores.
|
||||
fn storeThresholdPairs(monitor: *Monitor, rounds: usize) void {
|
||||
for (0..rounds) |i| {
|
||||
monitor.setThresholds(if (i % 2 == 0)
|
||||
.{ .min_free_mb = 1, .warn_free_mb = 2 }
|
||||
else
|
||||
.{ .min_free_mb = 3_000_000, .warn_free_mb = 4_000_000 });
|
||||
}
|
||||
}
|
||||
|
||||
test "a threshold reader never observes a pair from two different stores" {
|
||||
var threaded: std.Io.Threaded = .init(testing.allocator, .{});
|
||||
defer threaded.deinit();
|
||||
const io = threaded.io();
|
||||
|
||||
var monitor: Monitor = .init(.{ .min_free_mb = 1, .warn_free_mb = 2 }, std.Io.Dir.cwd(), ".", null);
|
||||
|
||||
const rounds = 20_000;
|
||||
var writer = try io.concurrent(storeThresholdPairs, .{ &monitor, rounds });
|
||||
// Recorded, not asserted, while the writer runs: an assertion that returned
|
||||
// here would leave `Threaded.deinit` joining a task nothing ends.
|
||||
var torn = false;
|
||||
for (0..rounds) |_| {
|
||||
const pair = monitor.thresholds();
|
||||
if (pair.warn_free_mb < pair.min_free_mb) torn = true;
|
||||
}
|
||||
writer.await(io);
|
||||
try testing.expect(!torn);
|
||||
}
|
||||
|
||||
test "every emit site is inert when the store is absent" {
|
||||
var threaded: std.Io.Threaded = .init(testing.allocator, .{});
|
||||
defer threaded.deinit();
|
||||
@@ -574,3 +748,89 @@ test "every emit site is inert when the store is absent" {
|
||||
monitor.sample(io, null, 0);
|
||||
try testing.expectEqual(@as(u64, 1), monitor.sample_failures.load(.monotonic));
|
||||
}
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// setLogDir (milestone-34 S3.5)
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
test "setLogDir re-points the measurement and frees the retired generation" {
|
||||
var threaded: std.Io.Threaded = .init(testing.allocator, .{});
|
||||
defer threaded.deinit();
|
||||
const io = threaded.io();
|
||||
|
||||
var tmp = testing.tmpDir(.{ .iterate = true });
|
||||
defer tmp.cleanup();
|
||||
try tmp.dir.createDirPath(io, "first");
|
||||
try tmp.dir.createDirPath(io, "second");
|
||||
try tmp.dir.writeFile(io, .{ .sub_path = "first/nxdns.log", .data = "aaaa" });
|
||||
try tmp.dir.writeFile(io, .{ .sub_path = "second/nxdns.log", .data = "bbbbbbbb" });
|
||||
|
||||
var first_buf: [160]u8 = undefined;
|
||||
var second_buf: [160]u8 = undefined;
|
||||
const first = try std.fmt.bufPrint(&first_buf, ".zig-cache/tmp/{s}/first", .{tmp.sub_path});
|
||||
const second = try std.fmt.bufPrint(&second_buf, ".zig-cache/tmp/{s}/second", .{tmp.sub_path});
|
||||
|
||||
var monitor: Monitor = .init(.{ .min_free_mb = 0, .warn_free_mb = 0 }, tmp.dir, ".", null);
|
||||
defer monitor.deinit(io);
|
||||
|
||||
// Boot measures nothing.
|
||||
monitor.sample(io, null, 1_000);
|
||||
try testing.expectEqual(@as(u64, 0), monitor.gauges().log_bytes);
|
||||
|
||||
monitor.setLogDir(io, try Monitor.prepareLogDir(testing.allocator, first));
|
||||
monitor.sample(io, null, 1_100);
|
||||
try testing.expectEqual(@as(u64, 4), monitor.gauges().log_bytes);
|
||||
|
||||
// The retired generation is freed here; the testing allocator says so.
|
||||
monitor.setLogDir(io, try Monitor.prepareLogDir(testing.allocator, second));
|
||||
monitor.sample(io, null, 1_200);
|
||||
try testing.expectEqual(@as(u64, 8), monitor.gauges().log_bytes);
|
||||
|
||||
// Output moved away from file: nothing is measured, and the gauge keeps
|
||||
// its last reading rather than claiming zero bytes of logs.
|
||||
monitor.setLogDir(io, null);
|
||||
monitor.sample(io, null, 1_300);
|
||||
try testing.expectEqual(@as(u64, 8), monitor.gauges().log_bytes);
|
||||
}
|
||||
|
||||
test "a prepared log directory that is never published is freed by the caller" {
|
||||
const prepared = try Monitor.prepareLogDir(testing.allocator, "/var/log/nxdns");
|
||||
Monitor.destroyPreparedLogDir(prepared);
|
||||
try testing.expectEqual(@as(?*LogDir, null), try Monitor.prepareLogDir(testing.allocator, null));
|
||||
}
|
||||
|
||||
test "a sample borrowing a log directory survives a concurrent setLogDir" {
|
||||
var threaded: std.Io.Threaded = .init(testing.allocator, .{});
|
||||
defer threaded.deinit();
|
||||
const io = threaded.io();
|
||||
|
||||
var tmp = testing.tmpDir(.{ .iterate = true });
|
||||
defer tmp.cleanup();
|
||||
try tmp.dir.createDirPath(io, "logs");
|
||||
try tmp.dir.writeFile(io, .{ .sub_path = "logs/nxdns.log", .data = "aaaa" });
|
||||
|
||||
var path_buf: [160]u8 = undefined;
|
||||
const logs = try std.fmt.bufPrint(&path_buf, ".zig-cache/tmp/{s}/logs", .{tmp.sub_path});
|
||||
|
||||
var monitor: Monitor = .init(.{ .min_free_mb = 0, .warn_free_mb = 0 }, tmp.dir, ".", null);
|
||||
defer monitor.deinit(io);
|
||||
monitor.setLogDir(io, try Monitor.prepareLogDir(testing.allocator, logs));
|
||||
|
||||
const Racer = struct {
|
||||
fn sample(m: *Monitor, sio: std.Io) void {
|
||||
for (0..200) |i| m.sample(sio, null, @intCast(1_000 + i));
|
||||
}
|
||||
fn repoint(m: *Monitor, sio: std.Io, p: []const u8) void {
|
||||
for (0..200) |i| {
|
||||
const prepared = Monitor.prepareLogDir(testing.allocator, if (i % 2 == 0) p else null) catch return;
|
||||
m.setLogDir(sio, prepared);
|
||||
}
|
||||
}
|
||||
};
|
||||
|
||||
var group: std.Io.Group = .init;
|
||||
defer group.cancel(io);
|
||||
try group.concurrent(io, Racer.sample, .{ &monitor, io });
|
||||
try group.concurrent(io, Racer.repoint, .{ &monitor, io, logs });
|
||||
try group.await(io);
|
||||
}
|
||||
|
||||
Reference in New Issue
Block a user