Basic support for cpuset cgroup config and process assignment

Signed-off-by: Joachim Nilsson <troglobit@gmail.com>
This commit is contained in:
Joachim Nilsson
2019-05-14 10:18:45 +02:00
parent 0848b9b07a
commit 44137806e4
10 changed files with 221 additions and 17 deletions
+48
View File
@@ -98,6 +98,54 @@ Syntax
* `network <PATH>`
Script or program to bring up networking, with optional arguments
* `cgroup <NAME> cpuset:SPEC cpu:SPEC mem:SPEC`
Create a control group for resource limiting services using cgroups.
Unlike `rlimit` this setting defines a common group to be selected for
any set of run/task/services using the `group` directive (below).
A cpuset for a cgroup can be the traditional list of CPU cores to
assign to the group, or a range that Finit will use to dynamically
allocate cores based on the amount available. The range values can be
either in absolute form or percent.
```shell
cgroup first cpuset:1
```
To create a cgroup with one exclusive CPU:
```shell
cgroup single cpuset:[1,1]
```
To create cgroup that uses at most two shared CPUs:
```shell
cgroup uptotwo cpuset:[,2]
```
To create cgroup with at least two shared CPUs:
```shell
cgroup dual cpuset:[2,]
```
To create cgroup with at least one CPU and at most half of all
available CPUs:
```shell
cgroup greedy cpuset:[1,50%]
```
* `group <NAME>`
All subsequent `service`, `task`, or `run` directives are executed in the
given control group `NAME`, which must be defined prior to this line. If
`NAME` does not yet exist the default group `default` is used.
When separate Finit `.conf` files are used for services the default group
is reset for each `.conf` file read.
* `rlimit [hard|soft] RESOURCE <LIMIT|unlimited>` Set the hard or soft
limit for a resource, or both if that argument is omitted. `RESOURCE`
is the lower-case `RLIMIT_` string constants from `setrlimit(2)`,
+77 -1
View File
@@ -30,7 +30,15 @@
#include "log.h"
#include "util.h"
static struct {
int active;
char cgroup[16];
char path[32];
} cg_avail[200];
static int cg_init = 0;
static int cg_num = 0;
/*
* Called by Finit at early boot to mount initial cgroups
@@ -64,8 +72,19 @@ void cgroup_init(void)
snprintf(rc, sizeof(rc), "/sys/fs/cgroup/%s", cgroup);
mkdir(rc, 0755);
if (mount("cgroup", rc, "cgroup", opts, cgroup))
if (mount("cgroup", rc, "cgroup", opts, cgroup)) {
_d("Failed mounting %s cgroup on %s", cgroup, rc);
continue;
}
cg_avail[cg_num].active = 1;
strlcpy(cg_avail[cg_num].cgroup, cgroup, sizeof(cg_avail[cg_num].cgroup));
strlcpy(cg_avail[cg_num].path, rc, sizeof(cg_avail[cg_num].path));
cg_num++;
/* Create default group in all available cgroups */
strlcat(rc, "/default", sizeof(rc));
mkdir(rc, 0755);
}
/* Default cgroups for process monitoring */
@@ -93,6 +112,49 @@ fail:
fclose(fp);
}
int cgroup_add(char *name, void (*cb)(char *, void *), void *arg)
{
int rc = 0;
int i;
for (i = 0; i < cg_num; i++) {
char path[80];
snprintf(path, sizeof(path), "%s/%s", cg_avail[i].path, name);
if (mkdir(path, 0755) && EEXIST != errno) {
rc++;
continue;
}
/* XXX: Add to cache of added groups, for cgroup_find() */
if (cb)
cb(path, arg);
}
return 0;
}
/* XXX: Temporary hackish implementation */
int cgroup_find(char *name, char *group, size_t len)
{
char path[80];
if (cg_num <= 0)
return 1;
snprintf(path, sizeof(path), "%s/%s", cg_avail[0].path, name);
if (access(path, F_OK))
return 1;
strlcpy(group, name, len);
return 0;
}
/* int cgroup_cpuset_range(char *group, int min, int max) */
/* { */
/* } */
static int move_pid(char *group, char *name, int pid)
{
char path[256];
@@ -135,6 +197,20 @@ int cgroup_service(char *cmd, int pid)
return move_pid("finit/system", nm, pid);
}
int cgroup_assign(char *name, int pid)
{
char path[120];
int i, rc = 0;
for (i = 0; i < cg_num; i++) {
snprintf(path, sizeof(path), "%s/%s/cgroup.procs",
cg_avail[i].path, name);
rc += echo(path, 0, "%d", pid);
}
return rc;
}
/**
* Local Variables:
* indent-tabs-mode: t
+4
View File
@@ -26,7 +26,11 @@
void cgroup_init (void);
int cgroup_add (char *name, void (*cb)(char *, void *), void *arg);
int cgroup_find (char *name, char *group, size_t len);
int cgroup_user (char *name);
int cgroup_service (char *cmd, int pid);
int cgroup_assign (char *name, int pid);
#endif /* FINIT_CGROUP_H_ */
+62 -7
View File
@@ -33,6 +33,7 @@
#include <glob.h>
#include "finit.h"
#include "cgroup.h"
#include "cond.h"
#include "service.h"
#include "tty.h"
@@ -53,6 +54,7 @@ struct conf_change {
char *name;
};
static char cgroup[32];
static uev_t w1, w2, w3, w4;
static TAILQ_HEAD(head, conf_change) conf_change_list = TAILQ_HEAD_INITIALIZER(conf_change_list);
@@ -342,6 +344,45 @@ error:
logit(LOG_WARNING, "rlimit: parse error");
}
static void cpuset_cb(char *grp, void *arg)
{
int cpuset = *(int *)arg;
char path[128];
if (!strstr(grp, "cpuset"))
return;
/* Both .cpus and .mems are mandatory settings */
snprintf(path, sizeof(path), "%s/cpuset.cpus", grp);
echo(path, 0, "%d", cpuset);
snprintf(path, sizeof(path), "%s/cpuset.mems", grp);
echo(path, 0, "0");
/* XXX: HACK for now, fix proper CPU allocation */
snprintf(path, sizeof(path), "%s/cpuset.cpu_exclusive", grp);
echo(path, 0, "1");
}
static void parse_cgroup(char *line)
{
char *group, *token;
int cpuset = -1;
group = strtok(line, " ");
if (!group)
return;
while ((token = strtok(NULL, " "))) {
if (!strncmp(token, "cpuset:", 7)) {
token += 7;
cpuset = atoi(token); /* XXX: fixme */
}
}
if (cpuset >= 0)
cgroup_add(group, cpuset_cb, &cpuset);
}
static void parse_static(char *line)
{
char *x;
@@ -430,6 +471,11 @@ static void parse_static(char *line)
cfglevel = 2; /* Fallback */
return;
}
if (MATCH_CMD(line, "cgroup ", x)) {
parse_cgroup(strip_line(x));
return;
}
}
static void parse_dynamic(char *line, struct rlimit rlimit[], char *file)
@@ -448,26 +494,26 @@ static void parse_dynamic(char *line, struct rlimit rlimit[], char *file)
/* Monitored daemon, will be respawned on exit */
if (MATCH_CMD(line, "service ", x)) {
service_register(SVC_TYPE_SERVICE, x, rlimit, file);
service_register(SVC_TYPE_SERVICE, x, rlimit, cgroup, file);
return;
}
/* One-shot task, will not be respawned */
if (MATCH_CMD(line, "task ", x)) {
service_register(SVC_TYPE_TASK, x, rlimit, file);
service_register(SVC_TYPE_TASK, x, rlimit, cgroup, file);
return;
}
/* Like task but waits for completion, useful w/ [S] */
if (MATCH_CMD(line, "run ", x)) {
service_register(SVC_TYPE_RUN, x, rlimit, file);
service_register(SVC_TYPE_RUN, x, rlimit, cgroup, file);
return;
}
/* Classic inetd service */
if (MATCH_CMD(line, "inetd ", x)) {
#ifdef INETD_ENABLED
service_register(SVC_TYPE_INETD, x, rlimit, file);
service_register(SVC_TYPE_INETD, x, rlimit, cgroup, file);
#else
_e("Finit built with inetd support disabled, cannot register service inetd %s!", x);
#endif
@@ -480,9 +526,15 @@ static void parse_dynamic(char *line, struct rlimit rlimit[], char *file)
return;
}
/* All subsequent run/task/service/tty/etc. directives to use this cgroup */
if (MATCH_CMD(line, "group ", x)) {
cgroup_find(x, cgroup, sizeof(cgroup));
return;
}
/* Regular or serial TTYs to run getty */
if (MATCH_CMD(line, "tty ", x)) {
tty_register(strip_line(x), rlimit, file);
tty_register(strip_line(x), rlimit, cgroup, file);
return;
}
}
@@ -500,8 +552,8 @@ static void tabstospaces(char *line)
static int parse_conf_dynamic(char *file)
{
FILE *fp;
struct rlimit rlimit[RLIMIT_NLIMITS];
FILE *fp;
fp = fopen(file, "r");
if (!fp) {
@@ -512,6 +564,9 @@ static int parse_conf_dynamic(char *file)
/* Prepare default limits for each service */
memcpy(rlimit, global_rlimit, sizeof(rlimit));
/* Prepare default cgroup for each service */
strlcpy(cgroup, "default", sizeof(cgroup));
_d("Parsing %s <<<<<<", file);
while (!feof(fp)) {
char line[LINE_SIZE] = "";
@@ -592,7 +647,7 @@ int conf_reload(void)
/* If rescue.conf is missing, fall back to a root shell */
rc = parse_conf(RESCUE_CONF);
if (rc)
tty_register(line, global_rlimit, NULL);
tty_register(line, global_rlimit, cgroup, NULL);
print(rc, "Entering rescue mode");
goto done;
+4 -4
View File
@@ -450,19 +450,19 @@ int main(int argc, char* argv[])
/* Register udevd as a monitored service */
snprintf(cmd, sizeof(cmd), "[S12345789] pid:udevd %s -- Device event managing daemon", path);
if (service_register(SVC_TYPE_SERVICE, cmd, global_rlimit, NULL)) {
if (service_register(SVC_TYPE_SERVICE, cmd, global_rlimit, "default", NULL)) {
_pe("Failed registering %s", path);
udev = 0;
} else {
snprintf(cmd, sizeof(cmd), ":1 [S] <svc%s> "
"udevadm trigger -c add -t devices "
"-- Requesting device events", path);
service_register(SVC_TYPE_RUN, cmd, global_rlimit, NULL);
service_register(SVC_TYPE_RUN, cmd, global_rlimit, "default", NULL);
snprintf(cmd, sizeof(cmd), ":2 [S] <svc%s> "
"udevadm trigger -c add -t subsystems "
"-- Requesting subsystem events", path);
service_register(SVC_TYPE_RUN, cmd, global_rlimit, NULL);
service_register(SVC_TYPE_RUN, cmd, global_rlimit, "default", NULL);
}
free(path);
} else {
@@ -483,7 +483,7 @@ int main(int argc, char* argv[])
* Start bundled watchdogd as soon as possible, if enabled
*/
if (which(FINIT_LIBPATH_ "/watchdogd"))
service_register(SVC_TYPE_SERVICE, FINIT_LIBPATH_ "/watchdogd", global_rlimit, NULL);
service_register(SVC_TYPE_SERVICE, FINIT_LIBPATH_ "/watchdogd", global_rlimit, "default", NULL);
/*
* Mount filesystems
+10 -2
View File
@@ -186,7 +186,11 @@ static int service_start(svc_t *svc)
sigprocmask(SIG_BLOCK, &nmask, &omask);
pid = fork();
cgroup_service(svc->cmd, pid);
if (pid > 0) {
cgroup_service(svc->cmd, pid);
cgroup_assign(svc->cgroup, pid);
}
if (pid == 0) {
int status;
@@ -626,6 +630,7 @@ static void parse_cmdline_args(svc_t *svc, char *cmd)
* @type: %SVC_TYPE_SERVICE(0), %SVC_TYPE_TASK(1), %SVC_TYPE_RUN(2)
* @cfg: Configuration, complete command, with -- for description text
* @rlimit: Limits for this service/task/run/inetd, may be global limits
* @cgroup: Control group, may be default
* @file: The file name service was loaded from
*
* This function is used to register commands to be run on different
@@ -673,7 +678,7 @@ static void parse_cmdline_args(svc_t *svc, char *cmd)
* Returns:
* POSIX OK(0) on success, or non-zero errno exit status on failure.
*/
int service_register(int type, char *cfg, struct rlimit rlimit[], char *file)
int service_register(int type, char *cfg, struct rlimit rlimit[], char *cgroup, char *file)
{
char id_str[MAX_ID_LEN];
#ifdef INETD_ENABLED
@@ -912,6 +917,9 @@ recreate:
/* Set configured limits */
memcpy(svc->rlimit, rlimit, sizeof(svc->rlimit));
/* Register desired control group */
strlcpy(svc->cgroup, cgroup, sizeof(svc->cgroup));
/* New, recently modified or unchanged ... used on reload. */
if (file && conf_changed(file))
svc_mark_dirty(svc);
+1 -1
View File
@@ -28,7 +28,7 @@
#include "svc.h"
void service_runlevel (int newlevel);
int service_register (int type, char *line, struct rlimit rlimit[], char *file);
int service_register (int type, char *line, struct rlimit rlimit[], char *cgroup, char *file);
void service_unregister (svc_t *svc);
void service_runtask_clean (void);
+3
View File
@@ -93,6 +93,9 @@ typedef struct svc {
/* Limits and scoping */
struct rlimit rlimit[RLIMIT_NLIMITS];
/* Control group */
char cgroup[32];
/* Service details */
pid_t pid;
char pidfile[MAX_ARG_LEN];
+8 -1
View File
@@ -32,6 +32,7 @@
#include "config.h" /* Generated by configure script */
#include "finit.h"
#include "cgroup.h"
#include "conf.h"
#include "helpers.h"
#include "tty.h"
@@ -100,6 +101,7 @@ void tty_sweep(void)
* tty_register - Register a getty on a device
* @line: Configuration, text after initial "tty"
* @rlimit: Limits for this service/task/run/inetd, may be global limits
* @cgroup: Control group, may be default
* @file: The file name TTY was loaded from
*
* A Finit tty line can use the internal getty implementation or an
@@ -118,7 +120,7 @@ void tty_sweep(void)
* Different getty implementations prefer the TTY device argument in
* different order, so take care to investigate this first.
*/
int tty_register(char *line, struct rlimit rlimit[], char *file)
int tty_register(char *line, struct rlimit rlimit[], char *cgroup, char *file)
{
struct tty *entry;
size_t i, num = 0;
@@ -274,6 +276,9 @@ again:
/* Register configured limits */
memcpy(entry->rlimit, rlimit, sizeof(entry->rlimit));
/* Register desired control group */
strlcpy(entry->cgroup, cgroup, sizeof(entry->cgroup));
if (file && conf_changed(file))
entry->dirty = 1; /* Modified, restart */
else
@@ -410,6 +415,8 @@ void tty_start(struct tty *tty)
tty->pid = run_getty(dev, tty->baud, tty->term, tty->noclear, tty->nowait, tty->rlimit);
else
tty->pid = run_getty2(dev, tty->cmd, tty->args, tty->noclear, tty->nowait, tty->rlimit);
cgroup_assign(tty->cgroup, tty->pid);
}
void tty_stop(struct tty *tty)
+4 -1
View File
@@ -51,6 +51,9 @@ struct tty {
/* Limits and scoping */
struct rlimit rlimit[RLIMIT_NLIMITS];
/* Control group */
char cgroup[32];
/* Set if modified => reloaded, or -1 when marked for removal */
int dirty;
};
@@ -58,7 +61,7 @@ struct tty {
void tty_mark (void);
void tty_sweep (void);
int tty_register (char *line, struct rlimit rlimit[], char *file);
int tty_register (char *line, struct rlimit rlimit[], char *cgroup, char *file);
int tty_unregister (struct tty *tty);
struct tty *tty_find (char *dev);