Follow-up to 10b72f1: periodically age service crash counter

Traditionally we've reset the service restart counter on two occasions:

  1) when a service reaches its max crash count (default 10), and
  2) when a service somehow "stabilizes"

In the second case the service is restored to normal running state and
we "forget" its bad previous behavior.  Meaning we cannot catch daemons
that act in an unstable manner outside the "rage quit" scenario.

This patch allows for slowing aging (decrementing) the restart counter
once every five minutes.  Meaning we still catch rage quitters but now
are also able catch other types of misbehavior.

Signed-off-by: Joachim Wiberg <troglobit@gmail.com>
This commit is contained in:
Joachim Wiberg
2022-01-13 23:13:26 +01:00
parent ee867e7f3b
commit 449bc7d12b
5 changed files with 94 additions and 1 deletions
+21
View File
@@ -41,6 +41,7 @@
#include "finit.h"
#include "cond.h"
#include "iwatch.h"
#include "private.h"
#include "service.h"
#include "tty.h"
#include "helpers.h"
@@ -572,6 +573,26 @@ static void parse_static(char *line, int is_rcsd)
cfglevel = 2; /* Fallback */
return;
}
/*
* Periodic check and instability index leveler, seconds
*/
if (MATCH_CMD(line, "service-interval ", x)) {
char *token = strip_line(x);
const char *err = NULL;
int val;
/* 0 min to 1 day, should check at least daily */
val = strtonum(token, 0, 1440, &err);
if (!err) {
int disabled = !service_interval;
service_interval = val * 1000; /* to milliseconds */
if (disabled)
service_init();
}
return;
}
}
static void parse_dynamic(char *line, struct rlimit rlimit[], char *file)
+5
View File
@@ -705,6 +705,11 @@ int main(int argc, char *argv[])
_d("Starting bootstrap finalize timer ...");
schedule_work(&bootstrap_work);
/*
* Background service tasks
*/
service_init();
/*
* Enter main loop to monitor /dev/initctl and services
*/
+3
View File
@@ -28,6 +28,9 @@
#include "svc.h"
#include "plugin.h"
#define SERVICE_INTERVAL_DEFAULT 300000 /* 5 mins */
extern int service_interval;
int api_init (uev_ctx_t *ctx);
int api_exit (void);
+63 -1
View File
@@ -56,6 +56,7 @@
static struct wq work = {
.cb = service_worker,
};
int service_interval = SERVICE_INTERVAL_DEFAULT;
static void svc_set_state(svc_t *svc, svc_state_t new);
@@ -1628,10 +1629,12 @@ static void service_retry(svc_t *svc)
if (svc->state != SVC_HALTED_STATE ||
svc->block != SVC_BLOCK_RESTARTING) {
*restart_cnt = 0;
logit(LOG_CONSOLE | LOG_NOTICE, "Successfully restarted crashing service %s.",
svc_ident(svc, NULL, 0));
return;
}
/* Peak instability index */
if (*restart_cnt >= svc->restart_max) {
logit(LOG_CONSOLE | LOG_WARNING, "Service %s keeps crashing, not restarting.",
svc_ident(svc, NULL, 0));
@@ -1968,6 +1971,65 @@ int service_completed(void)
return 1;
}
/*
* Every five¹ minutes we sweep over all services, skipping crashed or
* otherwise no longer running ones. Decrement non-zero crash counters
* to allow services that have started after an initial crash to slowly
* prove themselves again as stable services. Previously this counter
* was reset as soon as such services had stopped crashing at least once
* per second. This new scheme allows us to catch those that rage-quit
* immediately when we try to start them, but now also those that are
* only slightly buggy -- when they reach their restart_max, they too
* are marked 'crashed'.
*
* This does not affect the restart_tot counter, which you can see in
* the output from 'initctl status foo', along with this instability
* "index" in parethesis: total (cnt/max)
*/
static void service_interval_cb(uev_t *w, void *arg, int events)
{
svc_t *svc, *iter = NULL;
(void)arg;
if (UEV_ERROR == events) {
uev_timer_start(w);
return;
}
for (svc = svc_iterator(&iter, 1); svc; svc = svc_iterator(&iter, 0)) {
if (svc_is_daemon(svc) || svc_is_sysv(svc)) {
char *restart_cnt = (char *)&svc->restart_cnt;
if (!svc_is_running(svc))
continue;
if (*restart_cnt > 0) {
logit(LOG_CONSOLE | LOG_DEBUG, "Aging %s instability index (%d/%d)",
svc_ident(svc, NULL, 0), svc->restart_cnt, svc->restart_max);
(*restart_cnt)--;
}
}
}
service_init();
}
/*
* The service_interval may change (conf) between invocations, so we
* periodically reset the one-shot timer instead of using a periodic.
*/
void service_init(void)
{
static int initialized = 0;
static uev_t watcher;
if (!initialized) {
uev_timer_init(ctx, &watcher, service_interval_cb, NULL, service_interval, 0);
initialized = 1;
} else
uev_timer_set(&watcher, service_interval, 0);
}
/**
* Local Variables:
* indent-tabs-mode: t
+2
View File
@@ -41,6 +41,8 @@ void service_worker (void *unused);
int service_completed (void);
void service_init (void);
#endif /* FINIT_SERVICE_H_ */
/**