diff --git a/README.md b/README.md index 01022ad..bdde5a3 100644 --- a/README.md +++ b/README.md @@ -83,9 +83,13 @@ The following arguments can be passed to kured via the daemonset pod template: Flags: --alert-filter-regexp regexp.Regexp alert names to ignore when checking for active alerts --blocking-pod-selector stringArray label selector identifying pods whose presence should prevent reboots + --drain-grace-period int time in seconds given to each pod to terminate gracefully, if negative, the default value specified in the pod will be used (default: -1) + --skip-wait-for-delete-timeout int when seconds is greater than zero, skip waiting for the pods whose deletion timestamp is older than N seconds while draining a node (default: 0) --ds-name string name of daemonset on which to place lock (default "kured") --ds-namespace string namespace containing daemonset on which to place lock (default "kube-system") --end-time string schedule reboot only before this time of day (default "23:59:59") + --force-reboot bool force a reboot even if the drain is still running (default: false) + --drain-timeout duration timeout after which the drain is aborted (default: 0, infinite time) -h, --help help for kured --lock-annotation string annotation in which to record locking node (default "weave.works/kured-node-lock") --lock-ttl duration expire lock annotation after this duration (default: 0, disabled) diff --git a/cmd/kured/main.go b/cmd/kured/main.go index 3810145..45b71f8 100644 --- a/cmd/kured/main.go +++ b/cmd/kured/main.go @@ -38,24 +38,28 @@ var ( version = "unreleased" // Command line flags - period time.Duration - dsNamespace string - dsName string - lockAnnotation string - lockTTL time.Duration - prometheusURL string - preferNoScheduleTaintName string - alertFilter *regexp.Regexp - rebootSentinelFile string - rebootSentinelCommand string - notifyURL string - slackHookURL string - slackUsername string - slackChannel string - messageTemplateDrain string - messageTemplateReboot string - podSelectors []string - rebootCommand string + forceReboot bool + drainTimeout time.Duration + period time.Duration + drainGracePeriod int + skipWaitForDeleteTimeoutSeconds int + dsNamespace string + dsName string + lockAnnotation string + lockTTL time.Duration + prometheusURL string + preferNoScheduleTaintName string + alertFilter *regexp.Regexp + rebootSentinelFile string + rebootSentinelCommand string + notifyURL string + slackHookURL string + slackUsername string + slackChannel string + messageTemplateDrain string + messageTemplateReboot string + podSelectors []string + rebootCommand string rebootDays []string rebootStart string @@ -91,6 +95,14 @@ func main() { PreRun: flagCheck, Run: root} + rootCmd.PersistentFlags().BoolVar(&forceReboot, "force-reboot", false, + "force a reboot even if the drain is still running (default: false)") + rootCmd.PersistentFlags().IntVar(&drainGracePeriod, "drain-grace-period", -1, + "time in seconds given to each pod to terminate gracefully, if negative, the default value specified in the pod will be used (default: -1)") + rootCmd.PersistentFlags().IntVar(&skipWaitForDeleteTimeoutSeconds, "skip-wait-for-delete-timeout", 0, + "when seconds is greater than zero, skip waiting for the pods whose deletion timestamp is older than N seconds while draining a node (default: 0)") + rootCmd.PersistentFlags().DurationVar(&drainTimeout, "drain-timeout", 0, + "timeout after which the drain is aborted (default: 0, infinite time)") rootCmd.PersistentFlags().DurationVar(&period, "period", time.Minute*60, "sentinel check period") rootCmd.PersistentFlags().StringVar(&dsNamespace, "ds-namespace", "kube-system", @@ -332,20 +344,32 @@ func drain(client *kubernetes.Clientset, node *v1.Node) { } drainer := &kubectldrain.Helper{ - Client: client, - GracePeriodSeconds: -1, - Force: true, - DeleteEmptyDirData: true, - IgnoreAllDaemonSets: true, - ErrOut: os.Stderr, - Out: os.Stdout, + Client: client, + Ctx: context.Background(), + GracePeriodSeconds: drainGracePeriod, + SkipWaitForDeleteTimeoutSeconds: skipWaitForDeleteTimeoutSeconds, + Force: true, + DeleteEmptyDirData: true, + IgnoreAllDaemonSets: true, + ErrOut: os.Stderr, + Out: os.Stdout, + Timeout: drainTimeout, } + if err := kubectldrain.RunCordonOrUncordon(drainer, node, true); err != nil { - log.Fatalf("Error cordonning %s: %v", nodename, err) + if !forceReboot { + log.Fatalf("Error cordonning %s: %v", nodename, err) + } + log.Errorf("Error cordonning %s: %v, continuing with reboot anyway", nodename, err) + return } if err := kubectldrain.RunNodeDrain(drainer, nodename); err != nil { - log.Fatalf("Error draining %s: %v", nodename, err) + if !forceReboot { + log.Fatalf("Error draining %s: %v", nodename, err) + } + log.Errorf("Error draining %s: %v, continuing with reboot anyway", nodename, err) + return } }