From 665d212bd68d93f303d5b5349b5cd59f041f5281 Mon Sep 17 00:00:00 2001 From: Joachim Wiberg Date: Tue, 30 Mar 2021 10:11:10 +0200 Subject: [PATCH] Add support for configuring cgroups and their settings on services This patch adds support for modifying settings for the default cgroups; init, user, and system, as well as adding up to a total of eight groups for the system. Services can now be assigned to a cgroup, with optional extra settings for that particular process group. The syntax is slightly contrivied but follows the overall Finit syntax of prop:value,prop':value', e.g. cgroup maint cpu.weight:123,mem.max:10000 Starting with the introduction of rlimits, a group of services sharing the same .conf file can share the same (locally "global") rlimits, and now also the same cgroup, e.g. cgroup.maint service foo service bar cgroup:mem.max:1000 This puts foo and bar in the same top-level cgroup 'maint', with an extra memory restriction on bar for max 1000 bytes memory. NOTE: 'mem.' is a Finit extension, a shorthand for cgroups2 'memory.' Signed-off-by: Joachim Wiberg --- contrib/finit.conf | 31 +++++++++++- src/cgroup.c | 119 +++++++++++++++++++++++++++++++++++++++------ src/cgroup.h | 8 ++- src/conf.c | 98 +++++++++++++++++++++++++++++++++++-- src/conf.h | 6 +++ src/service.c | 39 +++++++++++++-- src/svc.h | 2 + src/util.c | 7 +-- 8 files changed, 282 insertions(+), 28 deletions(-) diff --git a/contrib/finit.conf b/contrib/finit.conf index 53b4dc89..f5823d11 100644 --- a/contrib/finit.conf +++ b/contrib/finit.conf @@ -8,6 +8,18 @@ module evdev module loop module psmouse +# Top-level cgroups and their default settings. All groups mandatory +# but more can be added, max 8 groups in total currently. The cgroup +# 'root' is also available, reserved for RT processes. Settings are +# as-is, only one shorthand 'mem.' exists, other than that it's the +# cgroup v2 controller default names. +cgroup init cpu.weight:100 +cgroup user cpu.weight:100 +cgroup system cpu.weight:9700 cpu.max:50000 + +# Example extra cgroup +cgroup maint cpu.weight:100 + # Runlevel to start after bootstrap, runlevel 'S' runlevel 2 @@ -16,7 +28,7 @@ network service networking start # Max file size for each log file: 100 kiB, rotate max 4 copies: # log => log.1 => log.2.gz => log.3.gz => log.4.gz -log size=100k count=4 +log size:100k count:4 # Virtual consoles to start getty on tty /dev/tty1 @@ -30,9 +42,24 @@ tty /dev/tty3 #run [2] /etc/init.d/networking start -- Start networking # Services to be monitored and respawned as needed + +# klgod and syslogd are placed in the maint cgroup +cgroup.maint service [2345] klogd -n -- Kernel logging server service [2345] syslogd -n -- Syslog server -service [3] gdm -- GNOME Display Manager + +# gdm in the system cgroup (default when read from it's own .conf file) +cgroup.system +service [3] cgroup:cpu.weight:250,mem.max:10M gdm -- GNOME Display Manager + +# Start SSH daemon with opts from /etc/default/ssh (if available), move +# to user cgroup with cpu.weight:250. Default for services is system +# group. Default weight in user group is 100 (above). Max length of +# cgroup argument is (currently) 63 chars. +# +# Notice alternative syntax where current cgroup is only set for this +# particular service. +service [2345] cgroup.user:cpu.weight:250,mem.max:10M env:-/etc/default/ssh /usr/sbin/sshd -D $SSHD_OPTS -- OpenSSH daemon # The BusyBox ntpd does not use syslog when running in the foreground # So we use this trick to redirect stdout/stderr to a log file. The diff --git a/src/cgroup.c b/src/cgroup.c index bc201844..f2ac392b 100644 --- a/src/cgroup.c +++ b/src/cgroup.c @@ -28,6 +28,7 @@ #include #include /* get_nprocs_conf() */ +#include "cgroup.h" #include "finit.h" #include "iwatch.h" #include "log.h" @@ -39,8 +40,52 @@ static struct iwatch iw_cgroup; static uev_t cgw; -static void group_init(char *path, int leaf, int weight) +static void cgset(const char *path, char *ctrl, char *prop) { + char *val; + + if (!path || !ctrl) { + _e("Missing path or controller, skipping!"); + return; + } + + if (!prop) { + prop = strchr(ctrl, '.'); + if (!prop) { + _e("Invalid cgroup ctrl syntax: %s", ctrl); + return; + } + + *prop++ = 0; + } + + val = strchr(prop, ':'); + if (!val) { + _e("Missing cgroup ctrl value, prop %s", prop); + return; + } + *val++ = 0; + + /* disallow sneaky relative paths */ + if (strstr(ctrl, "..") || strstr(prop, "..")) { + _e("Possible security violation; '..' not allowed in cgroup config!"); + return; + } + + _d("%s/%s.%s <= %s", path, ctrl, prop, val); + if (fnwrite(val, "%s/%s.%s", path, ctrl, prop)) + _pe("Failed setting %s/%s.%s = %s", path, ctrl, prop, val); +} + +/* + * Settings for a cgroup are on the form: cpu.weight:1234,mem.max:4321,... + * Finit supports the short-form 'mem.', replacing it with 'memory.' when + * writing the setting to the file system. + */ +static void group_init(char *path, int leaf, const char *cfg) +{ + char *ptr, *s; + if (mkdir(path, 0755) && EEXIST != errno) { _pe("Failed creating cgroup %s", path); return; @@ -48,14 +93,33 @@ static void group_init(char *path, int leaf, int weight) /* enable detected controllers on domain groups */ if (!leaf && fnwrite(controllers, "%s/cgroup.subtree_control", path)) - _pe("Failed %s for %s", controllers, path); + _pe("Failed enabling %s for %s", controllers, path); - /* set initial group weight */ - if (weight && fnwrite(str("%d", weight), "%s/cpu.weight", path)) - _pe("Failed setting cpu.weight %d for %s", weight, path); + if (!cfg) + return; + + s = strdup(cfg); + if (!s) { + _pe("Failed activating cgroup cfg for %s", path); + return; + } + + _d("%s <=> %s", path, s); + ptr = strtok(s, ","); + while (ptr) { + _d("ptr: %s", ptr); + if (!strncmp("mem.", ptr, 4)) + cgset(path, "memory", &ptr[4]); + else + cgset(path, ptr, NULL); + + ptr = strtok(NULL, ","); + } + + free(s); } -static int cgroup_leaf_init(char *group, char *name, int pid) +static int cgroup_leaf_init(char *group, char *name, int pid, const char *cfg) { char path[256]; @@ -66,11 +130,11 @@ static int cgroup_leaf_init(char *group, char *name, int pid) /* create and initialize new group */ snprintf(path, sizeof(path), "/sys/fs/cgroup/%s/%s", group, name); - group_init(path, 1, 0); + group_init(path, 1, cfg); /* move process to new group */ if (fnwrite(str("%d", pid), "%s/cgroup.procs", path)) - _pe("Failed moving pid %d to %s", pid, path); + _pe("Failed moving pid %d to group %s", pid, path); strlcat(path, "/cgroup.events", sizeof(path)); @@ -79,12 +143,22 @@ static int cgroup_leaf_init(char *group, char *name, int pid) int cgroup_user(char *name, int pid) { - return cgroup_leaf_init("user", name, pid); + return cgroup_leaf_init("user", name, pid, NULL); } -int cgroup_service(char *name, int pid) +int cgroup_service(char *name, int pid, struct cgroup *cg) { - return cgroup_leaf_init("system", name, pid); + char *group = "system"; + + if (cg && cg->name[0]) { + char path[256]; + + snprintf(path, sizeof(path), "/sys/fs/cgroup/%s", cg->name); + if (fisdir(path)) + group = cg->name; + } + + return cgroup_leaf_init(group, name, pid, cg->cfg); } static void append_ctrl(char *ctrl) @@ -214,14 +288,14 @@ void cgroup_init(uev_ctx_t *ctx) /* Enable all controllers */ if (fnwrite(controllers, FINIT_CGPATH "/cgroup.subtree_control")) - _pe("Failed %s for %s", controllers, FINIT_CGPATH "/cgroup.subtree_control"); + _pe("Failed enabling %s for %s", controllers, FINIT_CGPATH "/cgroup.subtree_control"); /* Default groups, PID 1, services, and user/login processes */ - group_init(FINIT_CGPATH "/init", 1, 100); - group_init(FINIT_CGPATH "/user", 0, 100); - group_init(FINIT_CGPATH "/system", 0, 9800); + group_init(FINIT_CGPATH "/init", 1, "cpu.weight:100"); + group_init(FINIT_CGPATH "/user", 0, "cpu.weight:100"); + group_init(FINIT_CGPATH "/system", 0, "cpu.weight:9800"); - /* Move ourselves to init */ + /* Move ourselves to init (best effort, otherwise run in 'root' group */ fnwrite("1", FINIT_CGPATH "/init/cgroup.procs"); /* prepare cgroup.events watcher */ @@ -232,6 +306,19 @@ void cgroup_init(uev_ctx_t *ctx) } } +/* the top-level init cgroup is a leaf, that's ensured in cgroup_init() */ +void cgroup_config(size_t num, struct cgroup cg[]) +{ + size_t i; + + for (i = 0; i < num; i++) { + char path[256]; + + snprintf(path, sizeof(path), "%s/%s", FINIT_CGPATH, cg[i].name); + group_init(path, 0, cg[i].cfg); + } +} + /** * Local Variables: * indent-tabs-mode: t diff --git a/src/cgroup.h b/src/cgroup.h index 352ee3a4..58e8b45a 100644 --- a/src/cgroup.h +++ b/src/cgroup.h @@ -26,9 +26,15 @@ #include +struct cgroup { + char name[16]; + char cfg[128]; +}; + void cgroup_init (uev_ctx_t *ctx); +void cgroup_config (size_t num, struct cgroup cg[]); int cgroup_user (char *name, int pid); -int cgroup_service (char *name, int pid); +int cgroup_service (char *name, int pid, struct cgroup *cg); #endif /* FINIT_CGROUP_H_ */ diff --git a/src/conf.c b/src/conf.c index 5750aac3..7461c2f5 100644 --- a/src/conf.c +++ b/src/conf.c @@ -23,6 +23,7 @@ #include "config.h" /* Generated by configure script */ +#include #include #include #include @@ -50,6 +51,10 @@ int logfile_count_max = 5; struct rlimit initial_rlimit[RLIMIT_NLIMITS]; struct rlimit global_rlimit[RLIMIT_NLIMITS]; +struct cgroup cgroups[8]; +size_t cgroups_num; +char cgroup_current[16]; /* cgroup.NAME sets current cgroup for a set of services */ + struct conf_change { TAILQ_ENTRY(conf_change) link; char *name; @@ -423,6 +428,67 @@ error: logit(LOG_WARNING, "rlimit: parse error"); } +/* reset defaults */ +static void init_cgroups(void) +{ + struct cgroup init[3] = { + { .name = "init", .cfg = "cpu.weight:100" }, + { .name = "user", .cfg = "cpu.weight:100" }, + { .name = "system", .cfg = "cpu.weight:9800" }, + }; + size_t i; + + for (i = 0; i < NELEMS(init); i++) + memcpy(&cgroups[i], &init[i], sizeof(struct cgroup)); + + cgroups_num = i; +} + +struct cgroup *conf_cgfind(char *name) +{ + size_t i; + + for (i = 0; i < cgroups_num; i++) { + if (strcmp(cgroups[i].name, name)) + continue; + + return &cgroups[i]; + } + + return NULL; +} + +/* cgroup NAME ctrl.prop:value,ctrl.prop:value ... */ +static void conf_parse_cgroup(char *line) +{ + struct cgroup *cg; + char *ptr, *name; + + name = strtok(line, " \t"); + if (!name) + return; + + cg = conf_cgfind(name); + if (!cg) { + if (strstr(name, "..") || strchr(name, '/')) + return; /* illegal */ + if (cgroups_num + 1 == NELEMS(cgroups)) + return; /* exhausted */ + + cg = &cgroups[cgroups_num++]; + strlcpy(cg->name, name, sizeof(cg->name)); + } + + cg->cfg[0] = 0; + while ((ptr = strtok(NULL, " \t"))) { + if (cg->cfg[0]) + strlcat(cg->cfg, ",", sizeof(cg->cfg)); + strlcat(cg->cfg, ptr, sizeof(cg->cfg)); + } + + _d("cg->cfg: %s", cg->cfg); +} + static void parse_static(char *line, int is_rcsd) { char cmd[CMD_SIZE]; @@ -552,6 +618,18 @@ static void parse_dynamic(char *line, struct rlimit rlimit[], char *file) return; } + /* Read contrl group limits */ + if (MATCH_CMD(line, "cgroup ", x)) { + conf_parse_cgroup(x); + return; + } + + /* Set current cgroup for the following services/run/tasks */ + if (MATCH_CMD(line, "cgroup.", x)) { + strlcpy(cgroup_current, x, sizeof(cgroup_current)); + return; + } + /* Regular or serial TTYs to run getty */ if (MATCH_CMD(line, "tty ", x)) { tty_register(strip_line(x), rlimit, file); @@ -580,9 +658,11 @@ static int parse_conf(char *file, int is_rcsd) if (!fp) return 1; - /* Prepare default limits for each service in /etc/finit.d/ */ - if (is_rcsd) + /* Prepare default limits and group for each service in /etc/finit.d/ */ + if (is_rcsd) { memcpy(rlimit, global_rlimit, sizeof(rlimit)); + memset(cgroup_current, 0, sizeof(cgroup_current)); + } _d("*** Parsing %s", file); while (!feof(fp)) { @@ -630,6 +710,9 @@ int conf_reload(void) */ memcpy(global_rlimit, initial_rlimit, sizeof(global_rlimit)); + /* Initialize default cgroups */ + init_cgroups(); + if (rescue) { int rc; char line[80] = "tty [12345] @console noclear nologin"; @@ -699,6 +782,9 @@ int conf_reload(void) /* Mark any reverse deps as chenaged. */ service_update_rdeps(); + + /* Set up top-level cgroups */ + cgroup_config(cgroups_num, cgroups); done: /* Drop record of all .conf changes */ drop_changes(); @@ -888,7 +974,7 @@ int conf_monitor(void) } /* - * Prepare .conf parser and load all .conf files + * Prepare .conf parser and load /etc/finit.conf for global settings */ int conf_init(uev_ctx_t *ctx) { @@ -911,6 +997,12 @@ int conf_init(uev_ctx_t *ctx) /* Initialize global rlimits, e.g. for built-in services */ memcpy(global_rlimit, initial_rlimit, sizeof(global_rlimit)); + /* Initialize default cgroups */ + init_cgroups(); + + /* Read global rlimits and global cgroup setup from /etc/finit.conf */ + parse_conf(FINIT_CONF, 0); + /* prepare /etc watcher */ fd = iwatch_init(&iw_conf); if (fd < 0) diff --git a/src/conf.h b/src/conf.h index 40d882d7..0b217fb6 100644 --- a/src/conf.h +++ b/src/conf.h @@ -24,16 +24,22 @@ #ifndef FINIT_CONF_H_ #define FINIT_CONF_H_ +#include "cgroup.h" #include "svc.h" extern int logfile_size_max; extern int logfile_count_max; extern struct rlimit global_rlimit[]; +extern struct cgroup cgroups[]; +extern size_t cgroup_num; +extern char cgroup_current[]; int str2rlim(char *str); char *rlim2str(int rlim); +struct cgroup *conf_cgfind(char *name); + int conf_init (uev_ctx_t *ctx); void conf_reload (void); int conf_any_change (void); diff --git a/src/service.c b/src/service.c index 674463f3..3c164d66 100644 --- a/src/service.c +++ b/src/service.c @@ -570,7 +570,7 @@ static int service_start(svc_t *svc) _d("Starting %s %s", svc->cmd, buf); } - cgroup_service(group_name(svc, grnam, sizeof(grnam)), pid); + cgroup_service(group_name(svc, grnam, sizeof(grnam)), pid, &svc->cgroup); logit(LOG_CONSOLE | LOG_NOTICE, "Starting %s[%d]", svc_ident(svc, NULL, 0), pid); @@ -857,6 +857,30 @@ static void parse_env(svc_t *svc, char *env) strlcpy(svc->env, env, sizeof(svc->env)); } +static void parse_cgroup(svc_t *svc, char *cgroup) +{ + char *ptr = cgroup; + + if (!cgroup) + return; + + if (cgroup[0] == '.') { + ptr = strchr(cgroup, ':'); + if (ptr) + *ptr++ = 0; + strlcpy(svc->cgroup.name, &cgroup[1], sizeof(svc->cgroup.name)); + if (!ptr) + return; + } + + if (strlen(ptr) >= sizeof(svc->cgroup)) { + _e("%s: cgroup settings too long (>%d chars)", svc->cmd, sizeof(svc->cgroup)); + return; + } + + strlcpy(svc->cgroup.cfg, ptr, sizeof(svc->cgroup.cfg)); +} + static void parse_sighalt(svc_t *svc, char *arg) { int signo; @@ -1026,7 +1050,7 @@ int service_register(int type, char *cfg, struct rlimit rlimit[], char *file) char *cmd, *desc, *runlevels = NULL, *cond = NULL; char *username = NULL, *log = NULL, *pid = NULL; char *name = NULL, *halt = NULL, *delay = NULL; - char *id = NULL, *env = NULL; + char *id = NULL, *env = NULL, *cgroup = NULL; int levels = 0; int manual = 0; char *line; @@ -1090,6 +1114,10 @@ int service_register(int type, char *cfg, struct rlimit rlimit[], char *file) delay = &cmd[5]; else if (!strncasecmp(cmd, "env:", 4)) env = &cmd[4]; + else if (!strncasecmp(cmd, "cgroup:", 7)) + cgroup = &cmd[7]; /* only settings */ + else if (!strncasecmp(cmd, "cgroup.", 7)) + cgroup = &cmd[6]; /* with group */ else break; @@ -1148,7 +1176,7 @@ int service_register(int type, char *cfg, struct rlimit rlimit[], char *file) parse_cmdline_args(svc, cmd); svc->runlevels = levels; - _d("Service %s runlevel 0x%2x", svc->cmd, svc->runlevels); + _d("Service %s runlevel 0x%02x", svc->cmd, svc->runlevels); conf_parse_cond(svc, cond); @@ -1169,6 +1197,11 @@ int service_register(int type, char *cfg, struct rlimit rlimit[], char *file) /* Set configured limits */ memcpy(svc->rlimit, rlimit, sizeof(svc->rlimit)); + /* Seed with currently active group, may be empty */ + strlcpy(svc->cgroup.name, cgroup_current, sizeof(svc->cgroup.name)); + if (cgroup) + parse_cgroup(svc, cgroup); + /* New, recently modified or unchanged ... used on reload. */ if ((file && conf_changed(file)) || conf_changed(svc_getenv(svc))) svc_mark_dirty(svc); diff --git a/src/svc.h b/src/svc.h index 60486c1f..a54f9579 100644 --- a/src/svc.h +++ b/src/svc.h @@ -32,6 +32,7 @@ #include /* BSD sys/queue.h API */ #include +#include "cgroup.h" #include "helpers.h" typedef int svc_cmd_t; @@ -94,6 +95,7 @@ typedef struct svc { /* Limits and scoping */ struct rlimit rlimit[RLIMIT_NLIMITS]; + struct cgroup cgroup; /* Service details */ int sighalt; /* Signal to stop prorcess, default: SIGTERM */ diff --git a/src/util.c b/src/util.c index 4dfac658..745e6d2a 100644 --- a/src/util.c +++ b/src/util.c @@ -88,9 +88,10 @@ int fnwrite(char *value, char *fmt, ...) return -1; /* echo(1) always adds a newline */ - fputs(value, fp); - fputs("\n", fp); - fclose(fp); + if (fputs(value, fp) == EOF || + fputs("\n", fp) == EOF || + fclose(fp)) + return -1; return 0; }