Add support for configuring cgroups and their settings on services

This patch adds support for modifying settings for the default cgroups;
init, user, and system, as well as adding up to a total of eight groups
for the system.

Services can now be assigned to a cgroup, with optional extra settings
for that particular process group.  The syntax is slightly contrivied
but follows the overall Finit syntax of prop:value,prop':value', e.g.

   cgroup maint cpu.weight:123,mem.max:10000

Starting with the introduction of rlimits, a group of services sharing
the same .conf file can share the same (locally "global") rlimits, and
now also the same cgroup, e.g.

    cgroup.maint
    service foo
    service bar cgroup:mem.max:1000

This puts foo and bar in the same top-level cgroup 'maint', with an
extra memory restriction on bar for max 1000 bytes memory.

NOTE: 'mem.' is a Finit extension, a shorthand for cgroups2 'memory.'

Signed-off-by: Joachim Wiberg <troglobit@gmail.com>
This commit is contained in:
Joachim Wiberg
2021-03-30 10:21:01 +02:00
parent 9e6e653c57
commit 665d212bd6
8 changed files with 282 additions and 28 deletions
+29 -2
View File
@@ -8,6 +8,18 @@ module evdev
module loop
module psmouse
# Top-level cgroups and their default settings. All groups mandatory
# but more can be added, max 8 groups in total currently. The cgroup
# 'root' is also available, reserved for RT processes. Settings are
# as-is, only one shorthand 'mem.' exists, other than that it's the
# cgroup v2 controller default names.
cgroup init cpu.weight:100
cgroup user cpu.weight:100
cgroup system cpu.weight:9700 cpu.max:50000
# Example extra cgroup
cgroup maint cpu.weight:100
# Runlevel to start after bootstrap, runlevel 'S'
runlevel 2
@@ -16,7 +28,7 @@ network service networking start
# Max file size for each log file: 100 kiB, rotate max 4 copies:
# log => log.1 => log.2.gz => log.3.gz => log.4.gz
log size=100k count=4
log size:100k count:4
# Virtual consoles to start getty on
tty /dev/tty1
@@ -30,9 +42,24 @@ tty /dev/tty3
#run [2] /etc/init.d/networking start -- Start networking
# Services to be monitored and respawned as needed
# klgod and syslogd are placed in the maint cgroup
cgroup.maint
service [2345] <pid/syslogd> klogd -n -- Kernel logging server
service [2345] syslogd -n -- Syslog server
service [3] gdm -- GNOME Display Manager
# gdm in the system cgroup (default when read from it's own .conf file)
cgroup.system
service [3] cgroup:cpu.weight:250,mem.max:10M gdm -- GNOME Display Manager
# Start SSH daemon with opts from /etc/default/ssh (if available), move
# to user cgroup with cpu.weight:250. Default for services is system
# group. Default weight in user group is 100 (above). Max length of
# cgroup argument is (currently) 63 chars.
#
# Notice alternative syntax where current cgroup is only set for this
# particular service.
service [2345] cgroup.user:cpu.weight:250,mem.max:10M env:-/etc/default/ssh /usr/sbin/sshd -D $SSHD_OPTS -- OpenSSH daemon
# The BusyBox ntpd does not use syslog when running in the foreground
# So we use this trick to redirect stdout/stderr to a log file. The
+103 -16
View File
@@ -28,6 +28,7 @@
#include <sys/mount.h>
#include <sys/sysinfo.h> /* get_nprocs_conf() */
#include "cgroup.h"
#include "finit.h"
#include "iwatch.h"
#include "log.h"
@@ -39,8 +40,52 @@ static struct iwatch iw_cgroup;
static uev_t cgw;
static void group_init(char *path, int leaf, int weight)
static void cgset(const char *path, char *ctrl, char *prop)
{
char *val;
if (!path || !ctrl) {
_e("Missing path or controller, skipping!");
return;
}
if (!prop) {
prop = strchr(ctrl, '.');
if (!prop) {
_e("Invalid cgroup ctrl syntax: %s", ctrl);
return;
}
*prop++ = 0;
}
val = strchr(prop, ':');
if (!val) {
_e("Missing cgroup ctrl value, prop %s", prop);
return;
}
*val++ = 0;
/* disallow sneaky relative paths */
if (strstr(ctrl, "..") || strstr(prop, "..")) {
_e("Possible security violation; '..' not allowed in cgroup config!");
return;
}
_d("%s/%s.%s <= %s", path, ctrl, prop, val);
if (fnwrite(val, "%s/%s.%s", path, ctrl, prop))
_pe("Failed setting %s/%s.%s = %s", path, ctrl, prop, val);
}
/*
* Settings for a cgroup are on the form: cpu.weight:1234,mem.max:4321,...
* Finit supports the short-form 'mem.', replacing it with 'memory.' when
* writing the setting to the file system.
*/
static void group_init(char *path, int leaf, const char *cfg)
{
char *ptr, *s;
if (mkdir(path, 0755) && EEXIST != errno) {
_pe("Failed creating cgroup %s", path);
return;
@@ -48,14 +93,33 @@ static void group_init(char *path, int leaf, int weight)
/* enable detected controllers on domain groups */
if (!leaf && fnwrite(controllers, "%s/cgroup.subtree_control", path))
_pe("Failed %s for %s", controllers, path);
_pe("Failed enabling %s for %s", controllers, path);
/* set initial group weight */
if (weight && fnwrite(str("%d", weight), "%s/cpu.weight", path))
_pe("Failed setting cpu.weight %d for %s", weight, path);
if (!cfg)
return;
s = strdup(cfg);
if (!s) {
_pe("Failed activating cgroup cfg for %s", path);
return;
}
_d("%s <=> %s", path, s);
ptr = strtok(s, ",");
while (ptr) {
_d("ptr: %s", ptr);
if (!strncmp("mem.", ptr, 4))
cgset(path, "memory", &ptr[4]);
else
cgset(path, ptr, NULL);
ptr = strtok(NULL, ",");
}
free(s);
}
static int cgroup_leaf_init(char *group, char *name, int pid)
static int cgroup_leaf_init(char *group, char *name, int pid, const char *cfg)
{
char path[256];
@@ -66,11 +130,11 @@ static int cgroup_leaf_init(char *group, char *name, int pid)
/* create and initialize new group */
snprintf(path, sizeof(path), "/sys/fs/cgroup/%s/%s", group, name);
group_init(path, 1, 0);
group_init(path, 1, cfg);
/* move process to new group */
if (fnwrite(str("%d", pid), "%s/cgroup.procs", path))
_pe("Failed moving pid %d to %s", pid, path);
_pe("Failed moving pid %d to group %s", pid, path);
strlcat(path, "/cgroup.events", sizeof(path));
@@ -79,12 +143,22 @@ static int cgroup_leaf_init(char *group, char *name, int pid)
int cgroup_user(char *name, int pid)
{
return cgroup_leaf_init("user", name, pid);
return cgroup_leaf_init("user", name, pid, NULL);
}
int cgroup_service(char *name, int pid)
int cgroup_service(char *name, int pid, struct cgroup *cg)
{
return cgroup_leaf_init("system", name, pid);
char *group = "system";
if (cg && cg->name[0]) {
char path[256];
snprintf(path, sizeof(path), "/sys/fs/cgroup/%s", cg->name);
if (fisdir(path))
group = cg->name;
}
return cgroup_leaf_init(group, name, pid, cg->cfg);
}
static void append_ctrl(char *ctrl)
@@ -214,14 +288,14 @@ void cgroup_init(uev_ctx_t *ctx)
/* Enable all controllers */
if (fnwrite(controllers, FINIT_CGPATH "/cgroup.subtree_control"))
_pe("Failed %s for %s", controllers, FINIT_CGPATH "/cgroup.subtree_control");
_pe("Failed enabling %s for %s", controllers, FINIT_CGPATH "/cgroup.subtree_control");
/* Default groups, PID 1, services, and user/login processes */
group_init(FINIT_CGPATH "/init", 1, 100);
group_init(FINIT_CGPATH "/user", 0, 100);
group_init(FINIT_CGPATH "/system", 0, 9800);
group_init(FINIT_CGPATH "/init", 1, "cpu.weight:100");
group_init(FINIT_CGPATH "/user", 0, "cpu.weight:100");
group_init(FINIT_CGPATH "/system", 0, "cpu.weight:9800");
/* Move ourselves to init */
/* Move ourselves to init (best effort, otherwise run in 'root' group */
fnwrite("1", FINIT_CGPATH "/init/cgroup.procs");
/* prepare cgroup.events watcher */
@@ -232,6 +306,19 @@ void cgroup_init(uev_ctx_t *ctx)
}
}
/* the top-level init cgroup is a leaf, that's ensured in cgroup_init() */
void cgroup_config(size_t num, struct cgroup cg[])
{
size_t i;
for (i = 0; i < num; i++) {
char path[256];
snprintf(path, sizeof(path), "%s/%s", FINIT_CGPATH, cg[i].name);
group_init(path, 0, cg[i].cfg);
}
}
/**
* Local Variables:
* indent-tabs-mode: t
+7 -1
View File
@@ -26,9 +26,15 @@
#include <uev/uev.h>
struct cgroup {
char name[16];
char cfg[128];
};
void cgroup_init (uev_ctx_t *ctx);
void cgroup_config (size_t num, struct cgroup cg[]);
int cgroup_user (char *name, int pid);
int cgroup_service (char *name, int pid);
int cgroup_service (char *name, int pid, struct cgroup *cg);
#endif /* FINIT_CGROUP_H_ */
+95 -3
View File
@@ -23,6 +23,7 @@
#include "config.h" /* Generated by configure script */
#include <ctype.h>
#include <dirent.h>
#include <string.h>
#include <sys/inotify.h>
@@ -50,6 +51,10 @@ int logfile_count_max = 5;
struct rlimit initial_rlimit[RLIMIT_NLIMITS];
struct rlimit global_rlimit[RLIMIT_NLIMITS];
struct cgroup cgroups[8];
size_t cgroups_num;
char cgroup_current[16]; /* cgroup.NAME sets current cgroup for a set of services */
struct conf_change {
TAILQ_ENTRY(conf_change) link;
char *name;
@@ -423,6 +428,67 @@ error:
logit(LOG_WARNING, "rlimit: parse error");
}
/* reset defaults */
static void init_cgroups(void)
{
struct cgroup init[3] = {
{ .name = "init", .cfg = "cpu.weight:100" },
{ .name = "user", .cfg = "cpu.weight:100" },
{ .name = "system", .cfg = "cpu.weight:9800" },
};
size_t i;
for (i = 0; i < NELEMS(init); i++)
memcpy(&cgroups[i], &init[i], sizeof(struct cgroup));
cgroups_num = i;
}
struct cgroup *conf_cgfind(char *name)
{
size_t i;
for (i = 0; i < cgroups_num; i++) {
if (strcmp(cgroups[i].name, name))
continue;
return &cgroups[i];
}
return NULL;
}
/* cgroup NAME ctrl.prop:value,ctrl.prop:value ... */
static void conf_parse_cgroup(char *line)
{
struct cgroup *cg;
char *ptr, *name;
name = strtok(line, " \t");
if (!name)
return;
cg = conf_cgfind(name);
if (!cg) {
if (strstr(name, "..") || strchr(name, '/'))
return; /* illegal */
if (cgroups_num + 1 == NELEMS(cgroups))
return; /* exhausted */
cg = &cgroups[cgroups_num++];
strlcpy(cg->name, name, sizeof(cg->name));
}
cg->cfg[0] = 0;
while ((ptr = strtok(NULL, " \t"))) {
if (cg->cfg[0])
strlcat(cg->cfg, ",", sizeof(cg->cfg));
strlcat(cg->cfg, ptr, sizeof(cg->cfg));
}
_d("cg->cfg: %s", cg->cfg);
}
static void parse_static(char *line, int is_rcsd)
{
char cmd[CMD_SIZE];
@@ -552,6 +618,18 @@ static void parse_dynamic(char *line, struct rlimit rlimit[], char *file)
return;
}
/* Read contrl group limits */
if (MATCH_CMD(line, "cgroup ", x)) {
conf_parse_cgroup(x);
return;
}
/* Set current cgroup for the following services/run/tasks */
if (MATCH_CMD(line, "cgroup.", x)) {
strlcpy(cgroup_current, x, sizeof(cgroup_current));
return;
}
/* Regular or serial TTYs to run getty */
if (MATCH_CMD(line, "tty ", x)) {
tty_register(strip_line(x), rlimit, file);
@@ -580,9 +658,11 @@ static int parse_conf(char *file, int is_rcsd)
if (!fp)
return 1;
/* Prepare default limits for each service in /etc/finit.d/ */
if (is_rcsd)
/* Prepare default limits and group for each service in /etc/finit.d/ */
if (is_rcsd) {
memcpy(rlimit, global_rlimit, sizeof(rlimit));
memset(cgroup_current, 0, sizeof(cgroup_current));
}
_d("*** Parsing %s", file);
while (!feof(fp)) {
@@ -630,6 +710,9 @@ int conf_reload(void)
*/
memcpy(global_rlimit, initial_rlimit, sizeof(global_rlimit));
/* Initialize default cgroups */
init_cgroups();
if (rescue) {
int rc;
char line[80] = "tty [12345] @console noclear nologin";
@@ -699,6 +782,9 @@ int conf_reload(void)
/* Mark any reverse deps as chenaged. */
service_update_rdeps();
/* Set up top-level cgroups */
cgroup_config(cgroups_num, cgroups);
done:
/* Drop record of all .conf changes */
drop_changes();
@@ -888,7 +974,7 @@ int conf_monitor(void)
}
/*
* Prepare .conf parser and load all .conf files
* Prepare .conf parser and load /etc/finit.conf for global settings
*/
int conf_init(uev_ctx_t *ctx)
{
@@ -911,6 +997,12 @@ int conf_init(uev_ctx_t *ctx)
/* Initialize global rlimits, e.g. for built-in services */
memcpy(global_rlimit, initial_rlimit, sizeof(global_rlimit));
/* Initialize default cgroups */
init_cgroups();
/* Read global rlimits and global cgroup setup from /etc/finit.conf */
parse_conf(FINIT_CONF, 0);
/* prepare /etc watcher */
fd = iwatch_init(&iw_conf);
if (fd < 0)
+6
View File
@@ -24,16 +24,22 @@
#ifndef FINIT_CONF_H_
#define FINIT_CONF_H_
#include "cgroup.h"
#include "svc.h"
extern int logfile_size_max;
extern int logfile_count_max;
extern struct rlimit global_rlimit[];
extern struct cgroup cgroups[];
extern size_t cgroup_num;
extern char cgroup_current[];
int str2rlim(char *str);
char *rlim2str(int rlim);
struct cgroup *conf_cgfind(char *name);
int conf_init (uev_ctx_t *ctx);
void conf_reload (void);
int conf_any_change (void);
+36 -3
View File
@@ -570,7 +570,7 @@ static int service_start(svc_t *svc)
_d("Starting %s %s", svc->cmd, buf);
}
cgroup_service(group_name(svc, grnam, sizeof(grnam)), pid);
cgroup_service(group_name(svc, grnam, sizeof(grnam)), pid, &svc->cgroup);
logit(LOG_CONSOLE | LOG_NOTICE, "Starting %s[%d]", svc_ident(svc, NULL, 0), pid);
@@ -857,6 +857,30 @@ static void parse_env(svc_t *svc, char *env)
strlcpy(svc->env, env, sizeof(svc->env));
}
static void parse_cgroup(svc_t *svc, char *cgroup)
{
char *ptr = cgroup;
if (!cgroup)
return;
if (cgroup[0] == '.') {
ptr = strchr(cgroup, ':');
if (ptr)
*ptr++ = 0;
strlcpy(svc->cgroup.name, &cgroup[1], sizeof(svc->cgroup.name));
if (!ptr)
return;
}
if (strlen(ptr) >= sizeof(svc->cgroup)) {
_e("%s: cgroup settings too long (>%d chars)", svc->cmd, sizeof(svc->cgroup));
return;
}
strlcpy(svc->cgroup.cfg, ptr, sizeof(svc->cgroup.cfg));
}
static void parse_sighalt(svc_t *svc, char *arg)
{
int signo;
@@ -1026,7 +1050,7 @@ int service_register(int type, char *cfg, struct rlimit rlimit[], char *file)
char *cmd, *desc, *runlevels = NULL, *cond = NULL;
char *username = NULL, *log = NULL, *pid = NULL;
char *name = NULL, *halt = NULL, *delay = NULL;
char *id = NULL, *env = NULL;
char *id = NULL, *env = NULL, *cgroup = NULL;
int levels = 0;
int manual = 0;
char *line;
@@ -1090,6 +1114,10 @@ int service_register(int type, char *cfg, struct rlimit rlimit[], char *file)
delay = &cmd[5];
else if (!strncasecmp(cmd, "env:", 4))
env = &cmd[4];
else if (!strncasecmp(cmd, "cgroup:", 7))
cgroup = &cmd[7]; /* only settings */
else if (!strncasecmp(cmd, "cgroup.", 7))
cgroup = &cmd[6]; /* with group */
else
break;
@@ -1148,7 +1176,7 @@ int service_register(int type, char *cfg, struct rlimit rlimit[], char *file)
parse_cmdline_args(svc, cmd);
svc->runlevels = levels;
_d("Service %s runlevel 0x%2x", svc->cmd, svc->runlevels);
_d("Service %s runlevel 0x%02x", svc->cmd, svc->runlevels);
conf_parse_cond(svc, cond);
@@ -1169,6 +1197,11 @@ int service_register(int type, char *cfg, struct rlimit rlimit[], char *file)
/* Set configured limits */
memcpy(svc->rlimit, rlimit, sizeof(svc->rlimit));
/* Seed with currently active group, may be empty */
strlcpy(svc->cgroup.name, cgroup_current, sizeof(svc->cgroup.name));
if (cgroup)
parse_cgroup(svc, cgroup);
/* New, recently modified or unchanged ... used on reload. */
if ((file && conf_changed(file)) || conf_changed(svc_getenv(svc)))
svc_mark_dirty(svc);
+2
View File
@@ -32,6 +32,7 @@
#include <lite/queue.h> /* BSD sys/queue.h API */
#include <uev/uev.h>
#include "cgroup.h"
#include "helpers.h"
typedef int svc_cmd_t;
@@ -94,6 +95,7 @@ typedef struct svc {
/* Limits and scoping */
struct rlimit rlimit[RLIMIT_NLIMITS];
struct cgroup cgroup;
/* Service details */
int sighalt; /* Signal to stop prorcess, default: SIGTERM */
+4 -3
View File
@@ -88,9 +88,10 @@ int fnwrite(char *value, char *fmt, ...)
return -1;
/* echo(1) always adds a newline */
fputs(value, fp);
fputs("\n", fp);
fclose(fp);
if (fputs(value, fp) == EOF ||
fputs("\n", fp) == EOF ||
fclose(fp))
return -1;
return 0;
}