From 567c69595486a7c70309e611487e4499a3341695 Mon Sep 17 00:00:00 2001 From: Joachim Wiberg Date: Mon, 15 Dec 2025 16:43:43 +0100 Subject: [PATCH] Start services with clone3() and place in cgroup directly This commit activates the use of clone3() for service_fork(), to allow Linux to create the new process directly in the correct cgroup instead of later moving it there -- much cheaper and less error prone. To facilitate this a few new helper cgroup functions have been added and two new configuration directives introduced: delegate and name:leafname. The delegate option is for running, e.g., container runtimes that want to create their own cgroup v2 structur, and the name:leafname allows a user to change the name of the subgroup under user/system/init. Signed-off-by: Joachim Wiberg --- src/cgroup.c | 202 +++++++++++++++++++++++++++++++++++++++++++++----- src/cgroup.h | 26 +++++-- src/conf.c | 37 ++++++++- src/conf.h | 2 + src/helpers.c | 2 +- src/service.c | 123 ++++++++++++++++++------------ 6 files changed, 316 insertions(+), 76 deletions(-) diff --git a/src/cgroup.c b/src/cgroup.c index 20d9f93f..0759335e 100644 --- a/src/cgroup.c +++ b/src/cgroup.c @@ -61,6 +61,33 @@ static uev_t cgw; static int avail; +/* + * Set up inotify watch on cgroup for automatic cleanup when empty. + * Should be called after cgroup_prepare() but before process starts. + */ +int cgroup_watch(const char *group, const char *name) +{ + char path[256]; + int rc; + + if (!avail) + return 0; + + /* Special case: root cgroup has no separate events file to watch */ + if (!strcmp(group, "root")) + return 0; + + snprintf(path, sizeof(path), "/sys/fs/cgroup/%s/%s/cgroup.events", group, name); + rc = iwatch_add(&iw_cgroup, path, 0); + if (rc < 0) { + warn("Failed setting up inotify watch on %s", path); + return -1; + } + + dbg("Watching %s for automatic cleanup", path); + return 0; +} + static void cgset(const char *path, char *ctrl, char *prop) { char *val; @@ -148,40 +175,99 @@ static void group_init(char *path, int leaf, const char *cfg) } } -static int cgroup_leaf_init(char *group, char *name, int pid, const char *cfg) +static int cgroup_create(const char *group, const char *name, const char *cfg, + int delegate, const char *username, const char *grpname, + char *pathbuf, size_t pathlen) { char path[256]; - dbg("group %s, name %s, pid %d, cfg %s", group, name, pid, cfg ?: "NIL"); - if (pid < 0 || pid == 1) { - errno = EINVAL; - return 1; + snprintf(path, sizeof(path), "/sys/fs/cgroup/%s/%s", group, name); + + if (delegate) { + char initpath[PATH_MAX]; + + /* For delegation, create as domain group (not leaf) */ + group_init(path, 0, cfg); + + /* Enable controllers for delegation */ + if (fnwrite(controllers, "%s/cgroup.subtree_control", path)) + warn("Failed enabling controllers in %s for delegation", path); + + /* Change ownership of delegation files */ + if (username && username[0] && grpname && grpname[0]) { + uid_t uid = getuser(username, NULL); + gid_t gid = getgroup(grpname); + + if (uid != (uid_t)-1 && gid != (gid_t)-1) { + char filepath[PATH_MAX]; + char *files[] = { + "cgroup.procs", + "cgroup.subtree_control", + "cgroup.threads", + "cgroup.type", + NULL + }; + + for (int i = 0; files[i]; i++) { + snprintf(filepath, sizeof(filepath), "%s/%s", path, files[i]); + if (chown(filepath, uid, gid)) + warn("Failed chown %s to %d:%d", filepath, uid, gid); + } + } + } + + snprintf(initpath, sizeof(initpath), "%s/supervisor", path); + group_init(initpath, 1, NULL); + strlcpy(path, initpath, sizeof(path)); + } else { + /* Normal leaf cgroup */ + group_init(path, 1, cfg); } - /* create and initialize new group */ - snprintf(path, sizeof(path), "/sys/fs/cgroup/%s/%s", group, name); - group_init(path, 1, cfg); + if (!fisdir(path)) { + warn("Cgroup directory %s doesn't exist after creation", path); + return -1; + } - /* move process to new group */ - if (fnwrite(str("%d", pid), "%s/cgroup.procs", path)) - err(1, "Failed moving pid %d to group %s", pid, path); + /* Return the path to caller if they provided a buffer */ + if (pathbuf && pathlen > 0) + strlcpy(pathbuf, path, pathlen); - strlcat(path, "/cgroup.events", sizeof(path)); - - return iwatch_add(&iw_cgroup, path, 0); + return cgroup_watch(group, name); } -int cgroup_user(char *name, int pid) +static int cgroup_leaf_init(const char *group, const char *name, int pid, const char *cfg, + int delegate, const char *username, const char *grpname) +{ + dbg("group %s, name %s, pid %d, cfg %s, delegate %d", group, name, pid, cfg ?: "NIL", delegate); + if (cgroup_create(group, name, cfg, delegate, username, grpname, NULL, 0)) + return -1; + + dbg("Assigning PID %d to cgroup %s/%s", pid, group, name); + + /* Special case: "root" cgroup means the actual cgroup root */ + if (!strcmp(group, "root")) + return fnwrite(str("%d", pid), FINIT_CGPATH "/cgroup.procs"); + + if (cgroup_move_pid(group, name, pid, delegate)) + err(1, "Failed moving pid %d to group %s/%s", pid, group, name); + + /* Set up inotify watch for cgroup cleanup */ + return cgroup_watch(group, name); +} + +int cgroup_user(const char *name, int pid) { if (!avail) return 0; - return cgroup_leaf_init("user", name, pid, NULL); + return cgroup_leaf_init("user", name, pid, NULL, 0, NULL, NULL); } -int cgroup_service(char *name, int pid, struct cgroup *cg) +int cgroup_service(const char *name, int pid, struct cgroup *cg, char *username, char *grpname) { char *group = "system"; + int delegate = 0; if (!avail) return 0; @@ -198,9 +284,64 @@ int cgroup_service(char *name, int pid, struct cgroup *cg) snprintf(path, sizeof(path), "/sys/fs/cgroup/%s", cg->name); if (fisdir(path)) group = cg->name; + + delegate = cg->delegate; } - return cgroup_leaf_init(group, name, pid, cg ? cg->cfg : NULL); + return cgroup_leaf_init(group, name, pid, cg ? cg->cfg : NULL, delegate, username, grpname); +} + +/* Create cgroup for the requested type of service, return fd to cgroup for clone3() */ +int cgroup_prepare(svc_t *svc, const char *name) +{ + const char *group; + const char *cfg = NULL; + const char *username = NULL; + const char *grpname = NULL; + int delegate = 0; + char path[256]; + int fd = -1; + + if (!avail) + return -1; + + if (!name) { + errno = EINVAL; + return -1; + } + + if (!svc) { + /* Helper process (e.g., networking) */ + group = "system"; + } else if (svc_is_tty(svc)) { + /* TTY/getty services go in user cgroup */ + group = "user"; + } else if (svc->cgroup.name[0] && !strcmp(svc->cgroup.name, "root")) { + /* Special case: SCHED_RR processes go in root cgroup */ + fd = open("/sys/fs/cgroup", O_RDONLY | O_DIRECTORY | O_CLOEXEC); + if (fd < 0) + warn("Failed opening root cgroup"); + + return fd; + } else { + /* Regular service */ + group = svc->cgroup.name[0] ? svc->cgroup.name : "system"; + cfg = svc->cgroup.cfg; + delegate = svc->cgroup.delegate; + username = svc->username; + grpname = svc->group; + } + + /* Create the cgroup and get the path back */ + if (cgroup_create(group, name, cfg, delegate, username, grpname, path, sizeof(path))) + return -1; + + /* Open and return fd for clone3() */ + fd = open(path, O_RDONLY | O_DIRECTORY | O_CLOEXEC); + if (fd < 0) + warn("Failed opening cgroup %s", path); + + return fd; } static void append_ctrl(char *ctrl) @@ -442,6 +583,31 @@ int cgroup_del(char *dir) return 0; } +/* + * Delete cgroup for a service (convenience wrapper) + */ +int cgroup_del_svc(svc_t *svc, const char *name) +{ + const char *group; + char path[256]; + + if (!avail) + return 0; + + /* Determine group from service */ + if (svc_is_tty(svc)) { + group = "user"; + } else if (svc->cgroup.name[0] && !strcmp(svc->cgroup.name, "root")) { + /* Root cgroup - nothing to delete */ + return 0; + } else { + group = svc->cgroup.name[0] ? svc->cgroup.name : "system"; + } + + snprintf(path, sizeof(path), "/sys/fs/cgroup/%s/%s", group, name); + return cgroup_del(path); +} + /* the top-level init cgroup is a leaf, that's ensured in cgroup_init() */ void cgroup_config(void) { diff --git a/src/cgroup.h b/src/cgroup.h index 538b759f..d0ac05af 100644 --- a/src/cgroup.h +++ b/src/cgroup.h @@ -26,21 +26,31 @@ #include +/* Forward declaration */ +typedef struct svc svc_t; + struct cgroup { char name[16]; char cfg[128]; + char leafname[128]; + char delegate; }; -void cgroup_mark_all(void); -void cgroup_cleanup (void); +void cgroup_mark_all(void); +void cgroup_cleanup (void); -int cgroup_add (char *name, char *cfg, int is_protected); -int cgroup_del (char *dir); -void cgroup_config (void); +int cgroup_add (char *name, char *cfg, int is_protected); +int cgroup_del (char *dir); +int cgroup_del_svc (svc_t *svc, const char *name); +void cgroup_config (void); -void cgroup_init (uev_ctx_t *ctx); +void cgroup_init (uev_ctx_t *ctx); -int cgroup_user (char *name, int pid); -int cgroup_service (char *name, int pid, struct cgroup *cg); +int cgroup_user (const char *name, int pid); +int cgroup_service (const char *name, int pid, struct cgroup *cg, char *username, char *group); + +char *cgroup_svc_name(svc_t *svc, char *buf, size_t len); +int cgroup_prepare (svc_t *svc, const char *name); +int cgroup_watch (const char *group, const char *name); #endif /* FINIT_CGROUP_H_ */ diff --git a/src/conf.c b/src/conf.c index 0d8f2ffb..b4393e65 100644 --- a/src/conf.c +++ b/src/conf.c @@ -106,7 +106,9 @@ char *runparts = NULL; int runparts_progress; int runparts_sysv; -char cgroup_current[16]; /* cgroup.NAME sets current cgroup for a set of services */ +char cgroup_current[16]; /* cgroup.NAME sets current cgroup for a set of services */ +char cgroup_settings_current[128]; /* cgroup.system,cpu.weight:500 - cgroup settings */ +int cgroup_delegate_current; /* cgroup.system,delegate - delegation flag */ struct conf_change { TAILQ_ENTRY(conf_change) link; @@ -1105,7 +1107,38 @@ static int parse_dynamic(char *line, struct rlimit rlimit[], char *file) /* Set current cgroup for the following services/run/tasks */ if (MATCH_CMD(line, "cgroup.", x)) { - strlcpy(cgroup_current, x, sizeof(cgroup_current)); + char *group, *token, *saveptr; + char settings[128] = {0}; + + /* Reset cgroup state */ + cgroup_settings_current[0] = '\0'; + cgroup_delegate_current = 0; + + /* First token is the group name (system/user/init) */ + group = strtok_r(x, ",", &saveptr); + if (group) + strlcpy(cgroup_current, group, sizeof(cgroup_current)); + else + strlcpy(cgroup_current, x, sizeof(cgroup_current)); + + /* Parse any remaining comma-separated options */ + while ((token = strtok_r(NULL, ",", &saveptr)) != NULL) { + if (strncmp(token, "name:", 5) == 0) { + /* Cgroup leaf name override - only valid in per-service directives */ + warnx("cgroup %s: name: option not valid in global directive!", group); + } else if (strcmp(token, "delegate") == 0) { + cgroup_delegate_current = 1; + } else { + /* Other settings (cpu.weight:500, memory.max:1G, etc.) */ + if (settings[0]) + strlcat(settings, ",", sizeof(settings)); + strlcat(settings, token, sizeof(settings)); + } + } + + if (settings[0]) + strlcpy(cgroup_settings_current, settings, sizeof(cgroup_settings_current)); + return 0; } diff --git a/src/conf.h b/src/conf.h index ab2e82e6..a86d60f4 100644 --- a/src/conf.h +++ b/src/conf.h @@ -51,6 +51,8 @@ extern int logfile_count_max; extern struct rlimit global_rlimit[]; extern char cgroup_current[]; +extern char cgroup_settings_current[]; +extern int cgroup_delegate_current; int str2rlim(char *str); char *rlim2str(int rlim); diff --git a/src/helpers.c b/src/helpers.c index cab93c89..96049954 100644 --- a/src/helpers.c +++ b/src/helpers.c @@ -463,7 +463,7 @@ void networking(int updown) _exit(EX_OSERR); _exit(WEXITSTATUS(rc)); } - cgroup_service("network", pid, NULL); + cgroup_service("network", pid, NULL, NULL, NULL); print(pid > 0 ? 0 : 1, "%s network interfaces ...", updown ? "Bringing up" : "Taking down"); diff --git a/src/service.c b/src/service.c index 6df698b3..978f7c8f 100644 --- a/src/service.c +++ b/src/service.c @@ -44,6 +44,7 @@ #endif #include "cgroup.h" +#include "clone3.h" #include "client.h" #include "conf.h" #include "cond.h" @@ -477,30 +478,6 @@ static int is_norespawn(void) fexist("/tmp/norespawn"); } -/* used for process group name, derived from originating filename, - * so to group multiple services, place them in the same .conf - */ -static char *group_name(svc_t *svc, char *buf, size_t len) -{ - char *ptr; - - if (!svc->file[0]) - return svc_ident(svc, buf, len); - - ptr = strrchr(svc->file, '/'); - if (ptr) - ptr++; - else - ptr = svc->file; - - strlcpy(buf, ptr, len); - ptr = strstr(buf, ".conf"); - if (ptr) - *ptr = 0; - - return buf; -} - static void compose_cmdline(svc_t *svc, char *buf, size_t len) { size_t i; @@ -547,9 +524,23 @@ static void set_uid(uid_t uid, svc_t *svc) static pid_t service_fork(svc_t *svc) { + const char *cgnm; + char grnam[128]; + int cgfd = -1; pid_t pid; - pid = fork(); + cgnm = cgroup_svc_name(svc, grnam, sizeof(grnam)); + cgfd = cgroup_prepare(svc, cgnm); + + pid = call_clone3(0, cgfd); + if (cgfd >= 0) + close(cgfd); + + if (pid < 0) { + cgroup_del_svc(svc, cgnm); + return pid; + } + if (pid == 0) { char *home = NULL; #ifdef ENABLE_STATIC @@ -594,13 +585,11 @@ static pid_t service_fork(svc_t *svc) source_env(svc); } - if (pid > 1) { - char grnam[80]; - + if (pid > 0 && !has_clone3()) { if (svc_is_tty(svc)) cgroup_user("getty", pid); else - cgroup_service(group_name(svc, grnam, sizeof(grnam)), pid, &svc->cgroup); + cgroup_service(cgnm, pid, &svc->cgroup, svc->username, svc->group); } return pid; @@ -1396,36 +1385,68 @@ static void parse_env(svc_t *svc, char *env) } /* - * the @cgroup argument can be, e.g., .system:mem.max:1234 or just the + * the @cgroup argument can be, e.g., .system,mem.max:1234 or just the * default group with some cfg, e.g., :mem.max:1234 as a side effect, * cgroupinit also work, selecting group init. */ static void parse_cgroup(svc_t *svc, char *cgroup) { - char *ptr = NULL; - if (!cgroup) return; + /* Per-service cgroup directive completely overrides global settings */ + svc->cgroup.delegate = 0; + svc->cgroup.leafname[0] = 0; + svc->cgroup.cfg[0] = 0; + + /* Detect syntax: old colon-based vs new comma-separated */ if (cgroup[0] == '.') { - ptr = strchr(cgroup, ':'); - if (ptr) - *ptr++ = 0; - group: - strlcpy(svc->cgroup.name, &cgroup[1], sizeof(svc->cgroup.name)); - if (!ptr) - return; + char *token, *ptr, *group; + char settings[128] = {0}; + char cgcopy[256]; + + /* New comma-separated syntax: cgroup.system,name:udevd,delegate,cpu.max:10000 */ + strlcpy(cgcopy, cgroup, sizeof(cgcopy)); + group = strtok_r(cgcopy, ",", &ptr); + if (group && group[0] == '.') + strlcpy(svc->cgroup.name, &group[1], sizeof(svc->cgroup.name)); + + /* Parse remaining comma-separated options */ + while ((token = strtok_r(NULL, ",", &ptr)) != NULL) { + if (strncmp(token, "name:", 5) == 0) { + /* Cgroup leaf name override */ + strlcpy(svc->cgroup.leafname, token + 5, sizeof(svc->cgroup.leafname)); + } else if (strcmp(token, "delegate") == 0) { + svc->cgroup.delegate = 1; + } else { + /* Other settings (cpu.weight:500, memory.max:1G, etc.) */ + if (settings[0]) + strlcat(settings, ",", sizeof(settings)); + strlcat(settings, token, sizeof(settings)); + } + } + + if (settings[0]) + strlcpy(svc->cgroup.cfg, settings, sizeof(svc->cgroup.cfg)); + + dbg("%s: cgroup name:%s leafname:%s cfg:%s delegate:%s", svc_ident(svc, NULL, 0), + svc->cgroup.name, svc->cgroup.leafname, svc->cgroup.cfg, + svc->cgroup.delegate ? "yes" : "no"); } else if (cgroup[0] == ':') { + char *ptr; + + /* Old syntax: cgroup:settings (keeps current group) */ ptr = &cgroup[1]; - } else - goto group; - - if (strlen(ptr) >= sizeof(svc->cgroup)) { - errx(1, "%s: cgroup settings too long (>%zu chars)", svc_ident(svc, NULL, 0), sizeof(svc->cgroup)); - return; + if (strlen(ptr) >= sizeof(svc->cgroup.cfg)) { + errx(1, "%s: cgroup settings too long (>%zu chars)", svc_ident(svc, NULL, 0), + sizeof(svc->cgroup.cfg)); + return; + } + strlcpy(svc->cgroup.cfg, ptr, sizeof(svc->cgroup.cfg)); + } else { + /* Just a group name without dot or colon */ + strlcpy(svc->cgroup.name, cgroup, sizeof(svc->cgroup.name)); } - - strlcpy(svc->cgroup.cfg, ptr, sizeof(svc->cgroup.cfg)); } static void parse_sighalt(svc_t *svc, char *arg) @@ -2085,6 +2106,14 @@ int service_register(int type, char *cfg, struct rlimit rlimit[], char *file) /* Seed with currently active group, may be empty */ strlcpy(svc->cgroup.name, cgroup_current, sizeof(svc->cgroup.name)); + + /* Apply cgroup settings if specified */ + if (cgroup_settings_current[0]) + strlcpy(svc->cgroup.cfg, cgroup_settings_current, sizeof(svc->cgroup.cfg)); + + /* Apply delegation flag */ + svc->cgroup.delegate = cgroup_delegate_current; + if (cgroup) parse_cgroup(svc, cgroup);