keventd: defer pidfile until coldplug queue is drained

When started with -c (the default when keventd is the device manager),
gate the pidfile on the kernel's uevent_seqnum having been stable for
200ms.  Up to now ready signaling with the pidfile was done right after
coldplug() triggered the kernel to re-emit events, but before uev_run()
had drained any of them, so <pid/keventd> really only meant "listening
on netlink".

With the gate, services that depend on <pid/keventd> can now assume /dev
is populated and persistent symlinks are live.

Signed-off-by: Joachim Wiberg <troglobit@gmail.com>
This commit is contained in:
Joachim Wiberg
2026-08-16 22:03:34 +02:00
parent c1c68054ea
commit 5a3972714a
3 changed files with 93 additions and 11 deletions
+6
View File
@@ -154,6 +154,12 @@ kernel to re-emit add events for all existing devices.
This replaces the separate `coldplug` script previously used with mdev.
When `-c` is used, keventd defers its pidfile (and the `pid/keventd`
condition Finit asserts from it) until the coldplug event queue has
been fully drained. Services that depend on `<pid/keventd>` can
therefore assume `/dev` is populated and persistent symlinks are live,
without needing a separate `settle` step.
Netlink Rebroadcast
-------------------
+80 -11
View File
@@ -76,6 +76,13 @@
/* Default netlink group for uevent rebroadcast (libudev-zero convention) */
#define REBC_DEFAULT_NLGROUP 4
/* Used by settle and coldplug at startup */
struct coldplug_gate {
unsigned long long last_seq;
struct timespec last_change;
int primed;
};
static int num_ac_online;
static int num_ac;
@@ -537,6 +544,55 @@ static void uevent_cb(uev_t *w, void *arg, int events)
rebc_event(rebc_buf, len);
}
/*
* Pidfile gate used with -c: defer pidfile() until the kernel's
* uevent_seqnum has been stable for 200ms, so that <pid/keventd>
* means "/dev is populated, persistent symlinks are live" rather
* than just "listening on netlink".
*/
static void coldplug_pidfile_cb(uev_t *w, void *arg, int events)
{
struct coldplug_gate *cg = arg;
unsigned long long cur = 0;
struct timespec now;
FILE *fp;
long dt_ms;
(void)events;
fp = fopen("/sys/kernel/uevent_seqnum", "r");
if (!fp)
return;
if (fscanf(fp, "%llu", &cur) != 1)
cur = cg->last_seq;
fclose(fp);
clock_gettime(CLOCK_MONOTONIC, &now);
if (!cg->primed) {
cg->last_seq = cur;
cg->last_change = now;
cg->primed = 1;
return;
}
if (cur != cg->last_seq) {
cg->last_seq = cur;
cg->last_change = now;
return;
}
dt_ms = (now.tv_sec - cg->last_change.tv_sec) * 1000 +
(now.tv_nsec - cg->last_change.tv_nsec) / 1000000;
if (dt_ms < 200)
return;
pidfile(NULL);
logit(LOG_NOTICE, "keventd ready, coldplug queue drained");
uev_timer_stop(w);
}
/*
* Stop-gap "settle" mode: poll /sys/kernel/uevent_seqnum until it has
* been stable for `stable_ms` (default 200ms), or until `timeout_s`
@@ -547,14 +603,15 @@ static void uevent_cb(uev_t *w, void *arg, int events)
*/
static int cmd_settle(int timeout_s, int stable_ms)
{
struct timespec start, last_change, now;
unsigned long long last_seq = 0, cur_seq = 0;
FILE *fp;
struct timespec start, last_change, now;
clock_gettime(CLOCK_MONOTONIC, &start);
last_change = start;
while (1) {
FILE *fp;
fp = fopen("/sys/kernel/uevent_seqnum", "r");
if (!fp) {
fprintf(stderr, "keventd: cannot read uevent_seqnum: %s\n",
@@ -580,7 +637,7 @@ static int cmd_settle(int timeout_s, int stable_ms)
if (now.tv_sec - start.tv_sec >= timeout_s)
return 1;
usleep(50000); /* 50ms */
usleep(50000);
}
}
@@ -620,16 +677,17 @@ static int usage(int rc)
*/
int main(int argc, char *argv[])
{
static char uevent_buf[UEVENT_BUFFER_SIZE];
uev_t netlink_watcher, sigusr1_watcher, sigterm_watcher, sighup_watcher, sigchld_watcher;
unsigned int nlgroups = REBC_DEFAULT_NLGROUP;
static char uevent_buf[UEVENT_BUFFER_SIZE];
static struct coldplug_gate cg;
struct sockaddr_nl nls = { 0 };
uev_t sigusr1_w, sigterm_w, sighup_w, sigchld_watcher;
uev_t netlink_watcher;
uev_ctx_t ctx;
int do_coldplug = 0;
int do_settle = 0;
static uev_t coldplug_timer;
int settle_timeout = 30;
int do_coldplug = 0;
int foreground = 0;
int do_settle = 0;
uev_ctx_t ctx;
int nlfd;
int c;
@@ -737,8 +795,19 @@ int main(int argc, char *argv[])
if (!passive)
rules_load_all(&rules, rules_dir);
pidfile(NULL);
logit(LOG_NOTICE, "keventd v%s started, waiting for events...", KEVENTD_VERSION);
/*
* With -c, defer pidfile until coldplug events have actually been
* drained, so <pid/keventd> means "/dev populated", not just
* "listening on netlink". Without -c, nothing is queued, drop
* the pidfile immediately as before.
*/
if (do_coldplug) {
uev_timer_init(&ctx, &coldplug_timer, coldplug_pidfile_cb, &cg, 100, 100);
logit(LOG_NOTICE, "keventd v%s started, draining coldplug queue...", KEVENTD_VERSION);
} else {
pidfile(NULL);
logit(LOG_NOTICE, "keventd v%s started, waiting for events...", KEVENTD_VERSION);
}
uev_run(&ctx, 0);
+7
View File
@@ -41,6 +41,13 @@ texec initctl reload >/dev/null \
|| fail "initctl reload returned non-zero"
assert "initctl reload ok" 0 -eq 0
# keventd was restarted above and only writes its pidfile once the
# coldplug queue has drained, so <pid/keventd> asserts a good while
# after the restart returns. Let it land before subscribing below,
# or the monitor catches keventd's ConditionChanged instead of ours
# and stops, since it waits for exactly one signal.
retry 'assert_cond pid/keventd' 100 0.1
# A ConditionChanged signal can only originate from Cond1.Set going
# through finit (the legacy filesystem path doesn't emit signals).
# So if the monitor sees one, we know initctl cond set was routed