From f6f9e7492c5a423fe9b2ab622495502e421f6775 Mon Sep 17 00:00:00 2001 From: Adam Harrison Date: Tue, 20 Nov 2018 17:53:35 +0000 Subject: [PATCH] Allow selected pods to prevent reboots --- README.md | 52 ++++++++++++++++++++++++++++++++++++----------- cmd/kured/main.go | 35 +++++++++++++++++++++++++++++-- kured-ds.yaml | 3 +++ 3 files changed, 76 insertions(+), 14 deletions(-) diff --git a/README.md b/README.md index e60a359..d177e1e 100644 --- a/README.md +++ b/README.md @@ -7,6 +7,7 @@ * [Configuration](#configuration) * [Reboot Sentinel File & Period](#reboot-sentinel-file-&-period) * [Blocking Reboots via Alerts](#blocking-reboots-via-alerts) + * [Blocking Reboots via Pods](#blocking-reboots-via-pods) * [Prometheus Metrics](#prometheus-metrics) * [Slack Notifications](#slack-notifications) * [Overriding Lock Configuration](#overriding-lock-configuration) @@ -27,7 +28,7 @@ indicated by the package management system of the underlying OS. * Watches for the presence of a reboot sentinel e.g. `/var/run/reboot-required` * Utilises a lock in the API server to ensure only one node reboots at a time -* Optionally defers reboots in the presence of active Prometheus alerts +* Optionally defers reboots in the presence of active Prometheus alerts or selected pods * Cordons & drains worker nodes before reboot, uncordoning them after ## Kubernetes & OS Compatibility @@ -67,15 +68,17 @@ The following arguments can be passed to kured via the daemonset pod template: ``` Flags: - --alert-filter-regexp value alert names to ignore when checking for active alerts - --ds-name string namespace containing daemonset on which to place lock (default "kube-system") - --ds-namespace string name of daemonset on which to place lock (default "kured") - --lock-annotation string annotation in which to record locking node (default "weave.works/kured-node-lock") - --period duration reboot check period (default 1h0m0s) - --prometheus-url string Prometheus instance to probe for active alerts - --reboot-sentinel string path to file whose existence signals need to reboot (default "/var/run/reboot-required") - --slack-hook-url string slack hook URL for reboot notfications - --slack-username string slack username for reboot notfications (default "kured") + --alert-filter-regexp regexp.Regexp alert names to ignore when checking for active alerts + --blocking-pod-selector stringArray label selector identifying pods whose presence should prevent reboots + --ds-name string name of daemonset on which to place lock (default "kured") + --ds-namespace string namespace containing daemonset on which to place lock (default "kube-system") + -h, --help help for kured + --lock-annotation string annotation in which to record locking node (default "weave.works/kured-node-lock") + --period duration reboot check period (default 1h0m0s) + --prometheus-url string Prometheus instance to probe for active alerts + --reboot-sentinel string path to file whose existence signals need to reboot (default "/var/run/reboot-required") + --slack-hook-url string slack hook URL for reboot notfications + --slack-username string slack username for reboot notfications (default "kured") ``` ### Reboot Sentinel File & Period @@ -103,8 +106,33 @@ will block reboots, however you can ignore specific alerts: --alert-filter-regexp=^(RebootRequired|AnotherBenignAlert|...$ ``` -An important application of this filter will become apparent in the -next section. +See the section on Prometheus metrics for an important application of this +filter. + +### Blocking Reboots via Pods + +You can also block reboots of an _individual node_ when specific pods +are scheduled on it: + +``` +--blocking-pod-selector=runtime=long,cost=expensive +``` + +Since label selector strings use commas to express logical 'and', you can +specify this parameter multiple times for 'or': + +``` +--blocking-pod-selector=runtime=long,cost=expensive +--blocking-pod-selector=name=temperamental +``` + +In this case, the presence of either an (appropriately labelled) expensive long +running job or a known temperamental pod on a node will stop it rebooting. + +> Try not to abuse this mechanism - it's better to strive for +> restartability where possible. If you do use it, make sure you set +> up a RebootRequired alert as described in the next section so that +> you can intervene manually if reboots are blocked for too long. ### Prometheus Metrics diff --git a/cmd/kured/main.go b/cmd/kured/main.go index e5b3d1b..cffa09d 100644 --- a/cmd/kured/main.go +++ b/cmd/kured/main.go @@ -1,6 +1,7 @@ package main import ( + "fmt" "math/rand" "net/http" "os" @@ -35,6 +36,7 @@ var ( rebootSentinel string slackHookURL string slackUsername string + podSelectors []string // Metrics rebootRequiredGauge = prometheus.NewGaugeVec(prometheus.GaugeOpts{ @@ -74,6 +76,9 @@ func main() { rootCmd.PersistentFlags().StringVar(&slackUsername, "slack-username", "kured", "slack username for reboot notfications") + rootCmd.PersistentFlags().StringArrayVar(&podSelectors, "blocking-pod-selector", nil, + "label selector identifying pods whose presence should prevent reboots") + if err := rootCmd.Execute(); err != nil { log.Fatal(err) } @@ -126,7 +131,7 @@ func rebootRequired() bool { } } -func rebootBlocked() bool { +func rebootBlocked(client *kubernetes.Clientset, nodeID string) bool { if prometheusURL != "" { alertNames, err := alerts.PrometheusActiveAlerts(prometheusURL, alertFilter) if err != nil { @@ -142,6 +147,31 @@ func rebootBlocked() bool { return true } } + + fieldSelector := fmt.Sprintf("spec.nodeName=%s", nodeID) + for _, labelSelector := range podSelectors { + podList, err := client.CoreV1().Pods("").List(metav1.ListOptions{ + LabelSelector: labelSelector, + FieldSelector: fieldSelector, + Limit: 10}) + if err != nil { + log.Warnf("Reboot blocked: pod query error: %v", err) + return true + } + + if len(podList.Items) > 0 { + podNames := make([]string, 0, len(podList.Items)) + for _, pod := range podList.Items { + podNames = append(podNames, pod.Name) + } + if len(podList.Continue) > 0 { + podNames = append(podNames, "...") + } + log.Warnf("Reboot blocked: matching pods: %v", podNames) + return true + } + } + return false } @@ -259,7 +289,7 @@ func rebootAsRequired(nodeID string) { source := rand.NewSource(time.Now().UnixNano()) tick := delaytick.New(source, period) for _ = range tick { - if rebootRequired() && !rebootBlocked() { + if rebootRequired() && !rebootBlocked(client, nodeID) { node, err := client.CoreV1().Nodes().Get(nodeID, metav1.GetOptions{}) if err != nil { log.Fatal(err) @@ -291,6 +321,7 @@ func root(cmd *cobra.Command, args []string) { log.Infof("Node ID: %s", nodeID) log.Infof("Lock Annotation: %s/%s:%s", dsNamespace, dsName, lockAnnotation) log.Infof("Reboot Sentinel: %s every %v", rebootSentinel, period) + log.Infof("Blocking Pod Selectors: %v", podSelectors) go rebootAsRequired(nodeID) go maintainRebootRequiredMetric(nodeID) diff --git a/kured-ds.yaml b/kured-ds.yaml index 13d8cee..72978dc 100644 --- a/kured-ds.yaml +++ b/kured-ds.yaml @@ -46,6 +46,9 @@ spec: command: - /usr/bin/kured # - --alert-filter-regexp=^RebootRequired$ +# - --blocking-pod-selector=runtime=long,cost=expensive +# - --blocking-pod-selector=name=temperamental +# - --blocking-pod-selector=... # - --ds-name=kured # - --ds-namespace=kube-system # - --lock-annotation=weave.works/kured-node-lock