mirror of
https://github.com/troglobit/finit.git
synced 2026-10-02 05:52:48 +07:00
Follow-up to 10b72f1: periodically age service crash counter
Traditionally we've reset the service restart counter on two occasions: 1) when a service reaches its max crash count (default 10), and 2) when a service somehow "stabilizes" In the second case the service is restored to normal running state and we "forget" its bad previous behavior. Meaning we cannot catch daemons that act in an unstable manner outside the "rage quit" scenario. This patch allows for slowing aging (decrementing) the restart counter once every five minutes. Meaning we still catch rage quitters but now are also able catch other types of misbehavior. Signed-off-by: Joachim Wiberg <troglobit@gmail.com>
This commit is contained in:
+21
@@ -41,6 +41,7 @@
|
||||
#include "finit.h"
|
||||
#include "cond.h"
|
||||
#include "iwatch.h"
|
||||
#include "private.h"
|
||||
#include "service.h"
|
||||
#include "tty.h"
|
||||
#include "helpers.h"
|
||||
@@ -572,6 +573,26 @@ static void parse_static(char *line, int is_rcsd)
|
||||
cfglevel = 2; /* Fallback */
|
||||
return;
|
||||
}
|
||||
|
||||
/*
|
||||
* Periodic check and instability index leveler, seconds
|
||||
*/
|
||||
if (MATCH_CMD(line, "service-interval ", x)) {
|
||||
char *token = strip_line(x);
|
||||
const char *err = NULL;
|
||||
int val;
|
||||
|
||||
/* 0 min to 1 day, should check at least daily */
|
||||
val = strtonum(token, 0, 1440, &err);
|
||||
if (!err) {
|
||||
int disabled = !service_interval;
|
||||
|
||||
service_interval = val * 1000; /* to milliseconds */
|
||||
if (disabled)
|
||||
service_init();
|
||||
}
|
||||
return;
|
||||
}
|
||||
}
|
||||
|
||||
static void parse_dynamic(char *line, struct rlimit rlimit[], char *file)
|
||||
|
||||
@@ -705,6 +705,11 @@ int main(int argc, char *argv[])
|
||||
_d("Starting bootstrap finalize timer ...");
|
||||
schedule_work(&bootstrap_work);
|
||||
|
||||
/*
|
||||
* Background service tasks
|
||||
*/
|
||||
service_init();
|
||||
|
||||
/*
|
||||
* Enter main loop to monitor /dev/initctl and services
|
||||
*/
|
||||
|
||||
@@ -28,6 +28,9 @@
|
||||
#include "svc.h"
|
||||
#include "plugin.h"
|
||||
|
||||
#define SERVICE_INTERVAL_DEFAULT 300000 /* 5 mins */
|
||||
extern int service_interval;
|
||||
|
||||
int api_init (uev_ctx_t *ctx);
|
||||
int api_exit (void);
|
||||
|
||||
|
||||
+63
-1
@@ -56,6 +56,7 @@
|
||||
static struct wq work = {
|
||||
.cb = service_worker,
|
||||
};
|
||||
int service_interval = SERVICE_INTERVAL_DEFAULT;
|
||||
|
||||
static void svc_set_state(svc_t *svc, svc_state_t new);
|
||||
|
||||
@@ -1628,10 +1629,12 @@ static void service_retry(svc_t *svc)
|
||||
|
||||
if (svc->state != SVC_HALTED_STATE ||
|
||||
svc->block != SVC_BLOCK_RESTARTING) {
|
||||
*restart_cnt = 0;
|
||||
logit(LOG_CONSOLE | LOG_NOTICE, "Successfully restarted crashing service %s.",
|
||||
svc_ident(svc, NULL, 0));
|
||||
return;
|
||||
}
|
||||
|
||||
/* Peak instability index */
|
||||
if (*restart_cnt >= svc->restart_max) {
|
||||
logit(LOG_CONSOLE | LOG_WARNING, "Service %s keeps crashing, not restarting.",
|
||||
svc_ident(svc, NULL, 0));
|
||||
@@ -1968,6 +1971,65 @@ int service_completed(void)
|
||||
return 1;
|
||||
}
|
||||
|
||||
/*
|
||||
* Every five¹ minutes we sweep over all services, skipping crashed or
|
||||
* otherwise no longer running ones. Decrement non-zero crash counters
|
||||
* to allow services that have started after an initial crash to slowly
|
||||
* prove themselves again as stable services. Previously this counter
|
||||
* was reset as soon as such services had stopped crashing at least once
|
||||
* per second. This new scheme allows us to catch those that rage-quit
|
||||
* immediately when we try to start them, but now also those that are
|
||||
* only slightly buggy -- when they reach their restart_max, they too
|
||||
* are marked 'crashed'.
|
||||
*
|
||||
* This does not affect the restart_tot counter, which you can see in
|
||||
* the output from 'initctl status foo', along with this instability
|
||||
* "index" in parethesis: total (cnt/max)
|
||||
*/
|
||||
static void service_interval_cb(uev_t *w, void *arg, int events)
|
||||
{
|
||||
svc_t *svc, *iter = NULL;
|
||||
|
||||
(void)arg;
|
||||
if (UEV_ERROR == events) {
|
||||
uev_timer_start(w);
|
||||
return;
|
||||
}
|
||||
|
||||
for (svc = svc_iterator(&iter, 1); svc; svc = svc_iterator(&iter, 0)) {
|
||||
if (svc_is_daemon(svc) || svc_is_sysv(svc)) {
|
||||
char *restart_cnt = (char *)&svc->restart_cnt;
|
||||
|
||||
if (!svc_is_running(svc))
|
||||
continue;
|
||||
|
||||
if (*restart_cnt > 0) {
|
||||
logit(LOG_CONSOLE | LOG_DEBUG, "Aging %s instability index (%d/%d)",
|
||||
svc_ident(svc, NULL, 0), svc->restart_cnt, svc->restart_max);
|
||||
(*restart_cnt)--;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
service_init();
|
||||
}
|
||||
|
||||
/*
|
||||
* The service_interval may change (conf) between invocations, so we
|
||||
* periodically reset the one-shot timer instead of using a periodic.
|
||||
*/
|
||||
void service_init(void)
|
||||
{
|
||||
static int initialized = 0;
|
||||
static uev_t watcher;
|
||||
|
||||
if (!initialized) {
|
||||
uev_timer_init(ctx, &watcher, service_interval_cb, NULL, service_interval, 0);
|
||||
initialized = 1;
|
||||
} else
|
||||
uev_timer_set(&watcher, service_interval, 0);
|
||||
}
|
||||
|
||||
/**
|
||||
* Local Variables:
|
||||
* indent-tabs-mode: t
|
||||
|
||||
@@ -41,6 +41,8 @@ void service_worker (void *unused);
|
||||
|
||||
int service_completed (void);
|
||||
|
||||
void service_init (void);
|
||||
|
||||
#endif /* FINIT_SERVICE_H_ */
|
||||
|
||||
/**
|
||||
|
||||
Reference in New Issue
Block a user