Start services with clone3() and place in cgroup directly

This commit activates the use of clone3() for service_fork(), to allow
Linux to create the new process directly in the correct cgroup instead
of later moving it there -- much cheaper and less error prone.

To facilitate this a few new helper cgroup functions have been added and
two new configuration directives introduced: delegate and name:leafname.
The delegate option is for running, e.g., container runtimes that want
to create their own cgroup v2 structur, and the name:leafname allows a
user to change the name of the subgroup under user/system/init.

Signed-off-by: Joachim Wiberg <troglobit@gmail.com>
This commit is contained in:
Joachim Wiberg
2025-12-15 22:04:31 +01:00
parent 4877aecc1e
commit 567c695954
6 changed files with 316 additions and 76 deletions
+184 -18
View File
@@ -61,6 +61,33 @@ static uev_t cgw;
static int avail;
/*
* Set up inotify watch on cgroup for automatic cleanup when empty.
* Should be called after cgroup_prepare() but before process starts.
*/
int cgroup_watch(const char *group, const char *name)
{
char path[256];
int rc;
if (!avail)
return 0;
/* Special case: root cgroup has no separate events file to watch */
if (!strcmp(group, "root"))
return 0;
snprintf(path, sizeof(path), "/sys/fs/cgroup/%s/%s/cgroup.events", group, name);
rc = iwatch_add(&iw_cgroup, path, 0);
if (rc < 0) {
warn("Failed setting up inotify watch on %s", path);
return -1;
}
dbg("Watching %s for automatic cleanup", path);
return 0;
}
static void cgset(const char *path, char *ctrl, char *prop)
{
char *val;
@@ -148,40 +175,99 @@ static void group_init(char *path, int leaf, const char *cfg)
}
}
static int cgroup_leaf_init(char *group, char *name, int pid, const char *cfg)
static int cgroup_create(const char *group, const char *name, const char *cfg,
int delegate, const char *username, const char *grpname,
char *pathbuf, size_t pathlen)
{
char path[256];
dbg("group %s, name %s, pid %d, cfg %s", group, name, pid, cfg ?: "NIL");
if (pid < 0 || pid == 1) {
errno = EINVAL;
return 1;
snprintf(path, sizeof(path), "/sys/fs/cgroup/%s/%s", group, name);
if (delegate) {
char initpath[PATH_MAX];
/* For delegation, create as domain group (not leaf) */
group_init(path, 0, cfg);
/* Enable controllers for delegation */
if (fnwrite(controllers, "%s/cgroup.subtree_control", path))
warn("Failed enabling controllers in %s for delegation", path);
/* Change ownership of delegation files */
if (username && username[0] && grpname && grpname[0]) {
uid_t uid = getuser(username, NULL);
gid_t gid = getgroup(grpname);
if (uid != (uid_t)-1 && gid != (gid_t)-1) {
char filepath[PATH_MAX];
char *files[] = {
"cgroup.procs",
"cgroup.subtree_control",
"cgroup.threads",
"cgroup.type",
NULL
};
for (int i = 0; files[i]; i++) {
snprintf(filepath, sizeof(filepath), "%s/%s", path, files[i]);
if (chown(filepath, uid, gid))
warn("Failed chown %s to %d:%d", filepath, uid, gid);
}
}
}
snprintf(initpath, sizeof(initpath), "%s/supervisor", path);
group_init(initpath, 1, NULL);
strlcpy(path, initpath, sizeof(path));
} else {
/* Normal leaf cgroup */
group_init(path, 1, cfg);
}
/* create and initialize new group */
snprintf(path, sizeof(path), "/sys/fs/cgroup/%s/%s", group, name);
group_init(path, 1, cfg);
if (!fisdir(path)) {
warn("Cgroup directory %s doesn't exist after creation", path);
return -1;
}
/* move process to new group */
if (fnwrite(str("%d", pid), "%s/cgroup.procs", path))
err(1, "Failed moving pid %d to group %s", pid, path);
/* Return the path to caller if they provided a buffer */
if (pathbuf && pathlen > 0)
strlcpy(pathbuf, path, pathlen);
strlcat(path, "/cgroup.events", sizeof(path));
return iwatch_add(&iw_cgroup, path, 0);
return cgroup_watch(group, name);
}
int cgroup_user(char *name, int pid)
static int cgroup_leaf_init(const char *group, const char *name, int pid, const char *cfg,
int delegate, const char *username, const char *grpname)
{
dbg("group %s, name %s, pid %d, cfg %s, delegate %d", group, name, pid, cfg ?: "NIL", delegate);
if (cgroup_create(group, name, cfg, delegate, username, grpname, NULL, 0))
return -1;
dbg("Assigning PID %d to cgroup %s/%s", pid, group, name);
/* Special case: "root" cgroup means the actual cgroup root */
if (!strcmp(group, "root"))
return fnwrite(str("%d", pid), FINIT_CGPATH "/cgroup.procs");
if (cgroup_move_pid(group, name, pid, delegate))
err(1, "Failed moving pid %d to group %s/%s", pid, group, name);
/* Set up inotify watch for cgroup cleanup */
return cgroup_watch(group, name);
}
int cgroup_user(const char *name, int pid)
{
if (!avail)
return 0;
return cgroup_leaf_init("user", name, pid, NULL);
return cgroup_leaf_init("user", name, pid, NULL, 0, NULL, NULL);
}
int cgroup_service(char *name, int pid, struct cgroup *cg)
int cgroup_service(const char *name, int pid, struct cgroup *cg, char *username, char *grpname)
{
char *group = "system";
int delegate = 0;
if (!avail)
return 0;
@@ -198,9 +284,64 @@ int cgroup_service(char *name, int pid, struct cgroup *cg)
snprintf(path, sizeof(path), "/sys/fs/cgroup/%s", cg->name);
if (fisdir(path))
group = cg->name;
delegate = cg->delegate;
}
return cgroup_leaf_init(group, name, pid, cg ? cg->cfg : NULL);
return cgroup_leaf_init(group, name, pid, cg ? cg->cfg : NULL, delegate, username, grpname);
}
/* Create cgroup for the requested type of service, return fd to cgroup for clone3() */
int cgroup_prepare(svc_t *svc, const char *name)
{
const char *group;
const char *cfg = NULL;
const char *username = NULL;
const char *grpname = NULL;
int delegate = 0;
char path[256];
int fd = -1;
if (!avail)
return -1;
if (!name) {
errno = EINVAL;
return -1;
}
if (!svc) {
/* Helper process (e.g., networking) */
group = "system";
} else if (svc_is_tty(svc)) {
/* TTY/getty services go in user cgroup */
group = "user";
} else if (svc->cgroup.name[0] && !strcmp(svc->cgroup.name, "root")) {
/* Special case: SCHED_RR processes go in root cgroup */
fd = open("/sys/fs/cgroup", O_RDONLY | O_DIRECTORY | O_CLOEXEC);
if (fd < 0)
warn("Failed opening root cgroup");
return fd;
} else {
/* Regular service */
group = svc->cgroup.name[0] ? svc->cgroup.name : "system";
cfg = svc->cgroup.cfg;
delegate = svc->cgroup.delegate;
username = svc->username;
grpname = svc->group;
}
/* Create the cgroup and get the path back */
if (cgroup_create(group, name, cfg, delegate, username, grpname, path, sizeof(path)))
return -1;
/* Open and return fd for clone3() */
fd = open(path, O_RDONLY | O_DIRECTORY | O_CLOEXEC);
if (fd < 0)
warn("Failed opening cgroup %s", path);
return fd;
}
static void append_ctrl(char *ctrl)
@@ -442,6 +583,31 @@ int cgroup_del(char *dir)
return 0;
}
/*
* Delete cgroup for a service (convenience wrapper)
*/
int cgroup_del_svc(svc_t *svc, const char *name)
{
const char *group;
char path[256];
if (!avail)
return 0;
/* Determine group from service */
if (svc_is_tty(svc)) {
group = "user";
} else if (svc->cgroup.name[0] && !strcmp(svc->cgroup.name, "root")) {
/* Root cgroup - nothing to delete */
return 0;
} else {
group = svc->cgroup.name[0] ? svc->cgroup.name : "system";
}
snprintf(path, sizeof(path), "/sys/fs/cgroup/%s/%s", group, name);
return cgroup_del(path);
}
/* the top-level init cgroup is a leaf, that's ensured in cgroup_init() */
void cgroup_config(void)
{
+18 -8
View File
@@ -26,21 +26,31 @@
#include <uev/uev.h>
/* Forward declaration */
typedef struct svc svc_t;
struct cgroup {
char name[16];
char cfg[128];
char leafname[128];
char delegate;
};
void cgroup_mark_all(void);
void cgroup_cleanup (void);
void cgroup_mark_all(void);
void cgroup_cleanup (void);
int cgroup_add (char *name, char *cfg, int is_protected);
int cgroup_del (char *dir);
void cgroup_config (void);
int cgroup_add (char *name, char *cfg, int is_protected);
int cgroup_del (char *dir);
int cgroup_del_svc (svc_t *svc, const char *name);
void cgroup_config (void);
void cgroup_init (uev_ctx_t *ctx);
void cgroup_init (uev_ctx_t *ctx);
int cgroup_user (char *name, int pid);
int cgroup_service (char *name, int pid, struct cgroup *cg);
int cgroup_user (const char *name, int pid);
int cgroup_service (const char *name, int pid, struct cgroup *cg, char *username, char *group);
char *cgroup_svc_name(svc_t *svc, char *buf, size_t len);
int cgroup_prepare (svc_t *svc, const char *name);
int cgroup_watch (const char *group, const char *name);
#endif /* FINIT_CGROUP_H_ */
+35 -2
View File
@@ -106,7 +106,9 @@ char *runparts = NULL;
int runparts_progress;
int runparts_sysv;
char cgroup_current[16]; /* cgroup.NAME sets current cgroup for a set of services */
char cgroup_current[16]; /* cgroup.NAME sets current cgroup for a set of services */
char cgroup_settings_current[128]; /* cgroup.system,cpu.weight:500 - cgroup settings */
int cgroup_delegate_current; /* cgroup.system,delegate - delegation flag */
struct conf_change {
TAILQ_ENTRY(conf_change) link;
@@ -1105,7 +1107,38 @@ static int parse_dynamic(char *line, struct rlimit rlimit[], char *file)
/* Set current cgroup for the following services/run/tasks */
if (MATCH_CMD(line, "cgroup.", x)) {
strlcpy(cgroup_current, x, sizeof(cgroup_current));
char *group, *token, *saveptr;
char settings[128] = {0};
/* Reset cgroup state */
cgroup_settings_current[0] = '\0';
cgroup_delegate_current = 0;
/* First token is the group name (system/user/init) */
group = strtok_r(x, ",", &saveptr);
if (group)
strlcpy(cgroup_current, group, sizeof(cgroup_current));
else
strlcpy(cgroup_current, x, sizeof(cgroup_current));
/* Parse any remaining comma-separated options */
while ((token = strtok_r(NULL, ",", &saveptr)) != NULL) {
if (strncmp(token, "name:", 5) == 0) {
/* Cgroup leaf name override - only valid in per-service directives */
warnx("cgroup %s: name: option not valid in global directive!", group);
} else if (strcmp(token, "delegate") == 0) {
cgroup_delegate_current = 1;
} else {
/* Other settings (cpu.weight:500, memory.max:1G, etc.) */
if (settings[0])
strlcat(settings, ",", sizeof(settings));
strlcat(settings, token, sizeof(settings));
}
}
if (settings[0])
strlcpy(cgroup_settings_current, settings, sizeof(cgroup_settings_current));
return 0;
}
+2
View File
@@ -51,6 +51,8 @@ extern int logfile_count_max;
extern struct rlimit global_rlimit[];
extern char cgroup_current[];
extern char cgroup_settings_current[];
extern int cgroup_delegate_current;
int str2rlim(char *str);
char *rlim2str(int rlim);
+1 -1
View File
@@ -463,7 +463,7 @@ void networking(int updown)
_exit(EX_OSERR);
_exit(WEXITSTATUS(rc));
}
cgroup_service("network", pid, NULL);
cgroup_service("network", pid, NULL, NULL, NULL);
print(pid > 0 ? 0 : 1, "%s network interfaces ...",
updown ? "Bringing up" : "Taking down");
+76 -47
View File
@@ -44,6 +44,7 @@
#endif
#include "cgroup.h"
#include "clone3.h"
#include "client.h"
#include "conf.h"
#include "cond.h"
@@ -477,30 +478,6 @@ static int is_norespawn(void)
fexist("/tmp/norespawn");
}
/* used for process group name, derived from originating filename,
* so to group multiple services, place them in the same .conf
*/
static char *group_name(svc_t *svc, char *buf, size_t len)
{
char *ptr;
if (!svc->file[0])
return svc_ident(svc, buf, len);
ptr = strrchr(svc->file, '/');
if (ptr)
ptr++;
else
ptr = svc->file;
strlcpy(buf, ptr, len);
ptr = strstr(buf, ".conf");
if (ptr)
*ptr = 0;
return buf;
}
static void compose_cmdline(svc_t *svc, char *buf, size_t len)
{
size_t i;
@@ -547,9 +524,23 @@ static void set_uid(uid_t uid, svc_t *svc)
static pid_t service_fork(svc_t *svc)
{
const char *cgnm;
char grnam[128];
int cgfd = -1;
pid_t pid;
pid = fork();
cgnm = cgroup_svc_name(svc, grnam, sizeof(grnam));
cgfd = cgroup_prepare(svc, cgnm);
pid = call_clone3(0, cgfd);
if (cgfd >= 0)
close(cgfd);
if (pid < 0) {
cgroup_del_svc(svc, cgnm);
return pid;
}
if (pid == 0) {
char *home = NULL;
#ifdef ENABLE_STATIC
@@ -594,13 +585,11 @@ static pid_t service_fork(svc_t *svc)
source_env(svc);
}
if (pid > 1) {
char grnam[80];
if (pid > 0 && !has_clone3()) {
if (svc_is_tty(svc))
cgroup_user("getty", pid);
else
cgroup_service(group_name(svc, grnam, sizeof(grnam)), pid, &svc->cgroup);
cgroup_service(cgnm, pid, &svc->cgroup, svc->username, svc->group);
}
return pid;
@@ -1396,36 +1385,68 @@ static void parse_env(svc_t *svc, char *env)
}
/*
* the @cgroup argument can be, e.g., .system:mem.max:1234 or just the
* the @cgroup argument can be, e.g., .system,mem.max:1234 or just the
* default group with some cfg, e.g., :mem.max:1234 as a side effect,
* cgroupinit also work, selecting group init.
*/
static void parse_cgroup(svc_t *svc, char *cgroup)
{
char *ptr = NULL;
if (!cgroup)
return;
/* Per-service cgroup directive completely overrides global settings */
svc->cgroup.delegate = 0;
svc->cgroup.leafname[0] = 0;
svc->cgroup.cfg[0] = 0;
/* Detect syntax: old colon-based vs new comma-separated */
if (cgroup[0] == '.') {
ptr = strchr(cgroup, ':');
if (ptr)
*ptr++ = 0;
group:
strlcpy(svc->cgroup.name, &cgroup[1], sizeof(svc->cgroup.name));
if (!ptr)
return;
char *token, *ptr, *group;
char settings[128] = {0};
char cgcopy[256];
/* New comma-separated syntax: cgroup.system,name:udevd,delegate,cpu.max:10000 */
strlcpy(cgcopy, cgroup, sizeof(cgcopy));
group = strtok_r(cgcopy, ",", &ptr);
if (group && group[0] == '.')
strlcpy(svc->cgroup.name, &group[1], sizeof(svc->cgroup.name));
/* Parse remaining comma-separated options */
while ((token = strtok_r(NULL, ",", &ptr)) != NULL) {
if (strncmp(token, "name:", 5) == 0) {
/* Cgroup leaf name override */
strlcpy(svc->cgroup.leafname, token + 5, sizeof(svc->cgroup.leafname));
} else if (strcmp(token, "delegate") == 0) {
svc->cgroup.delegate = 1;
} else {
/* Other settings (cpu.weight:500, memory.max:1G, etc.) */
if (settings[0])
strlcat(settings, ",", sizeof(settings));
strlcat(settings, token, sizeof(settings));
}
}
if (settings[0])
strlcpy(svc->cgroup.cfg, settings, sizeof(svc->cgroup.cfg));
dbg("%s: cgroup name:%s leafname:%s cfg:%s delegate:%s", svc_ident(svc, NULL, 0),
svc->cgroup.name, svc->cgroup.leafname, svc->cgroup.cfg,
svc->cgroup.delegate ? "yes" : "no");
} else if (cgroup[0] == ':') {
char *ptr;
/* Old syntax: cgroup:settings (keeps current group) */
ptr = &cgroup[1];
} else
goto group;
if (strlen(ptr) >= sizeof(svc->cgroup)) {
errx(1, "%s: cgroup settings too long (>%zu chars)", svc_ident(svc, NULL, 0), sizeof(svc->cgroup));
return;
if (strlen(ptr) >= sizeof(svc->cgroup.cfg)) {
errx(1, "%s: cgroup settings too long (>%zu chars)", svc_ident(svc, NULL, 0),
sizeof(svc->cgroup.cfg));
return;
}
strlcpy(svc->cgroup.cfg, ptr, sizeof(svc->cgroup.cfg));
} else {
/* Just a group name without dot or colon */
strlcpy(svc->cgroup.name, cgroup, sizeof(svc->cgroup.name));
}
strlcpy(svc->cgroup.cfg, ptr, sizeof(svc->cgroup.cfg));
}
static void parse_sighalt(svc_t *svc, char *arg)
@@ -2085,6 +2106,14 @@ int service_register(int type, char *cfg, struct rlimit rlimit[], char *file)
/* Seed with currently active group, may be empty */
strlcpy(svc->cgroup.name, cgroup_current, sizeof(svc->cgroup.name));
/* Apply cgroup settings if specified */
if (cgroup_settings_current[0])
strlcpy(svc->cgroup.cfg, cgroup_settings_current, sizeof(svc->cgroup.cfg));
/* Apply delegation flag */
svc->cgroup.delegate = cgroup_delegate_current;
if (cgroup)
parse_cgroup(svc, cgroup);