From 29b4af1ab73d79469dfc32ffd85d49dcbe91dd07 Mon Sep 17 00:00:00 2001 From: Jean-Philippe Evrard Date: Tue, 1 Oct 2024 21:56:16 +0200 Subject: [PATCH 1/3] Automatically point to correct repository Without this, the CI would automatically point DH_ORG to kubereboot/kured on ghcr, instead of pointing to the owner of the repo. This makes the CI smoother. Signed-off-by: Jean-Philippe Evrard --- .github/workflows/on-pr.yaml | 14 +++++++------- .github/workflows/periodics-daily.yaml | 2 +- Makefile | 2 +- 3 files changed, 9 insertions(+), 9 deletions(-) diff --git a/.github/workflows/on-pr.yaml b/.github/workflows/on-pr.yaml index 46ed0be..a56a436 100644 --- a/.github/workflows/on-pr.yaml +++ b/.github/workflows/on-pr.yaml @@ -86,7 +86,7 @@ jobs: run: echo "sha_short=$(git rev-parse --short HEAD)" >> $GITHUB_OUTPUT id: tags - name: Build image - run: VERSION="${{ steps.tags.outputs.sha_short }}" make image + run: VERSION="${{ steps.tags.outputs.sha_short }}" DH_ORG="${{ github.repository_owner }}" make image - name: Run Trivy vulnerability scanner uses: aquasecurity/trivy-action@6e7b7d1fd3e4fef0c5fa8cce1229c54b2c9bd0d8 with: @@ -132,8 +132,8 @@ jobs: id: tags - name: Build artifacts run: | - VERSION="${{ steps.tags.outputs.sha_short }}" make image - VERSION="${{ steps.tags.outputs.sha_short }}" make manifest + VERSION="${{ steps.tags.outputs.sha_short }}" DH_ORG="${{ github.repository_owner }}" make image + VERSION="${{ steps.tags.outputs.sha_short }}" DH_ORG="${{ github.repository_owner }}" make manifest - name: Workaround "Failed to attach 1 to compat systemd cgroup /actions_job/..." on gh actions run: | @@ -217,8 +217,8 @@ jobs: id: tags - name: Build artifacts run: | - VERSION="${{ steps.tags.outputs.sha_short }}" make image - VERSION="${{ steps.tags.outputs.sha_short }}" make manifest + VERSION="${{ steps.tags.outputs.sha_short }}" DH_ORG="${{ github.repository_owner }}" make image + VERSION="${{ steps.tags.outputs.sha_short }}" DH_ORG="${{ github.repository_owner }}" make manifest - name: Workaround "Failed to attach 1 to compat systemd cgroup /actions_job/..." on gh actions run: | @@ -303,8 +303,8 @@ jobs: id: tags - name: Build artifacts run: | - VERSION="${{ steps.tags.outputs.sha_short }}" make image - VERSION="${{ steps.tags.outputs.sha_short }}" make manifest + VERSION="${{ steps.tags.outputs.sha_short }}" DH_ORG="${{ github.repository_owner }}" make image + VERSION="${{ steps.tags.outputs.sha_short }}" DH_ORG="${{ github.repository_owner }}" make manifest - name: Workaround "Failed to attach 1 to compat systemd cgroup /actions_job/..." on gh actions run: | diff --git a/.github/workflows/periodics-daily.yaml b/.github/workflows/periodics-daily.yaml index 40ba746..a7fa8c8 100644 --- a/.github/workflows/periodics-daily.yaml +++ b/.github/workflows/periodics-daily.yaml @@ -68,7 +68,7 @@ jobs: run: echo "sha_short=$(git rev-parse --short HEAD)" >> $GITHUB_OUTPUT id: tags - name: Build artifacts - run: VERSION="${{ steps.tags.outputs.sha_short }}" make image + run: VERSION="${{ steps.tags.outputs.sha_short }}" DH_ORG="${{ github.repository_owner }}" make image - name: Run Trivy vulnerability scanner uses: aquasecurity/trivy-action@6e7b7d1fd3e4fef0c5fa8cce1229c54b2c9bd0d8 with: diff --git a/Makefile b/Makefile index f4240f3..378032d 100644 --- a/Makefile +++ b/Makefile @@ -3,7 +3,7 @@ TEMPDIR=./.tmp GORELEASER_CMD=$(TEMPDIR)/goreleaser -DH_ORG=kubereboot +DH_ORG ?= kubereboot VERSION=$(shell git rev-parse --short HEAD) SUDO=$(shell docker info >/dev/null 2>&1 || echo "sudo -E") From 5536bf7e30c27d0c15f7b7cd0ff1a37c6354ef33 Mon Sep 17 00:00:00 2001 From: Jean-Philippe Evrard Date: Tue, 1 Oct 2024 22:15:44 +0200 Subject: [PATCH 2/3] Add CVE in ignore list We can't move to use 1.22 yet, so we'll ignore this one. Signed-off-by: Jean-Philippe Evrard --- .trivyignore | 3 +++ 1 file changed, 3 insertions(+) create mode 100644 .trivyignore diff --git a/.trivyignore b/.trivyignore new file mode 100644 index 0000000..114be2d --- /dev/null +++ b/.trivyignore @@ -0,0 +1,3 @@ +# https://pkg.go.dev/vuln/GO-2024-3106 +# Will be automatically fixed when we'll use golang 1.22.7 +CVE-2024-34156 From a02ae67559b438ca03617a52718e781185db1661 Mon Sep 17 00:00:00 2001 From: Jean-Philippe Evrard Date: Wed, 2 Oct 2024 23:26:33 +0200 Subject: [PATCH 3/3] Accelerate CI jobs Without this, some CI jobs are flaky or slow due to the following issues: - Triggering a reboot cause an unrecoverable boot loop. This fixes it by restarting the containers that are incorrectly exited. - API server is down while operations happen. This fixes it by ensuring at least one API server is up. In this case, we don't add a reboot marker on the unique api server. - The amount of nodes in a test environment is larger than necessary. This fixes it by ensuring two nodes are required to reboot. This is enough for concurrency, and for the e2e testing. - The wait time between operations is high, and can cause a heartbeat to be missed in the check script. This fixes it by checking more often, at the expense of more logging. This is compensated by increasing the amount of tries. Signed-off-by: Jean-Philippe Evrard --- .github/kind-cluster-1.28.yaml | 4 ---- .github/kind-cluster-1.29.yaml | 4 ---- .github/kind-cluster-1.30.yaml | 4 ---- .github/workflows/on-pr.yaml | 16 ++++++++-------- tests/kind/create-reboot-sentinels.sh | 5 +++-- tests/kind/follow-coordinated-reboot.sh | 17 +++++++++++++---- 6 files changed, 24 insertions(+), 26 deletions(-) diff --git a/.github/kind-cluster-1.28.yaml b/.github/kind-cluster-1.28.yaml index 11a9b99..9eda6de 100644 --- a/.github/kind-cluster-1.28.yaml +++ b/.github/kind-cluster-1.28.yaml @@ -1,10 +1,6 @@ kind: Cluster apiVersion: kind.x-k8s.io/v1alpha4 nodes: -- role: control-plane - image: "kindest/node:v1.28.9" -- role: control-plane - image: "kindest/node:v1.28.9" - role: control-plane image: "kindest/node:v1.28.9" - role: worker diff --git a/.github/kind-cluster-1.29.yaml b/.github/kind-cluster-1.29.yaml index 492374a..69eb6b7 100644 --- a/.github/kind-cluster-1.29.yaml +++ b/.github/kind-cluster-1.29.yaml @@ -1,10 +1,6 @@ kind: Cluster apiVersion: kind.x-k8s.io/v1alpha4 nodes: -- role: control-plane - image: "kindest/node:v1.29.4" -- role: control-plane - image: "kindest/node:v1.29.4" - role: control-plane image: "kindest/node:v1.29.4" - role: worker diff --git a/.github/kind-cluster-1.30.yaml b/.github/kind-cluster-1.30.yaml index d926e2a..eaf9d9c 100644 --- a/.github/kind-cluster-1.30.yaml +++ b/.github/kind-cluster-1.30.yaml @@ -1,10 +1,6 @@ kind: Cluster apiVersion: kind.x-k8s.io/v1alpha4 nodes: -- role: control-plane - image: "kindest/node:v1.30.2" -- role: control-plane - image: "kindest/node:v1.30.2" - role: control-plane image: "kindest/node:v1.30.2" - role: worker diff --git a/.github/workflows/on-pr.yaml b/.github/workflows/on-pr.yaml index a56a436..7b97533 100644 --- a/.github/workflows/on-pr.yaml +++ b/.github/workflows/on-pr.yaml @@ -145,7 +145,7 @@ jobs: EOF # Default name for helm/kind-action kind clusters is "chart-testing" - - name: Create kind cluster with 5 nodes + - name: Create kind cluster with 3 nodes uses: helm/kind-action@v1.10.0 with: config: .github/kind-cluster-${{ matrix.kubernetes }}.yaml @@ -169,7 +169,7 @@ jobs: max_attempts: 10 retry_wait_seconds: 60 # DESIRED CURRENT READY UP-TO-DATE AVAILABLE should all be = to cluster_size - command: "kubectl get ds -n kube-system kured | grep -E 'kured.*5.*5.*5.*5.*5'" + command: "kubectl get ds -n kube-system kured | grep -E 'kured.*3.*3.*3.*3.*3'" - name: Create reboot sentinel files run: | @@ -230,7 +230,7 @@ jobs: EOF # Default name for helm/kind-action kind clusters is "chart-testing" - - name: Create kind cluster with 5 nodes + - name: Create kind cluster with 3 nodes uses: helm/kind-action@v1.10.0 with: config: .github/kind-cluster-${{ matrix.kubernetes }}.yaml @@ -241,7 +241,7 @@ jobs: - name: Do not wait for an hour before detecting the rebootSentinel run: | - sed -i 's/#\(.*\)--period=1h/\1--period=30s/g' kured-ds-signal.yaml + sed -i 's/#\(.*\)--period=1h/\1--period=15s/g' kured-ds-signal.yaml - name: Install kured with kubectl run: | @@ -254,7 +254,7 @@ jobs: max_attempts: 10 retry_wait_seconds: 60 # DESIRED CURRENT READY UP-TO-DATE AVAILABLE should all be = to cluster_size - command: "kubectl get ds -n kube-system kured | grep -E 'kured.*5.*5.*5.*5.*5'" + command: "kubectl get ds -n kube-system kured | grep -E 'kured.*3.*3.*3.*3.*3'" - name: Create reboot sentinel files run: | @@ -316,7 +316,7 @@ jobs: EOF # Default name for helm/kind-action kind clusters is "chart-testing" - - name: Create kind cluster with 5 nodes + - name: Create kind cluster with 3 nodes uses: helm/kind-action@v1.10.0 with: config: .github/kind-cluster-${{ matrix.kubernetes }}.yaml @@ -327,7 +327,7 @@ jobs: - name: Do not wait for an hour before detecting the rebootSentinel run: | - sed -i 's/#\(.*\)--period=1h/\1--period=30s/g' kured-ds.yaml + sed -i 's/#\(.*\)--period=1h/\1--period=15s/g' kured-ds.yaml sed -i 's/#\(.*\)--concurrency=1/\1--concurrency=2/g' kured-ds.yaml - name: Install kured with kubectl @@ -341,7 +341,7 @@ jobs: max_attempts: 10 retry_wait_seconds: 60 # DESIRED CURRENT READY UP-TO-DATE AVAILABLE should all be = to cluster_size - command: "kubectl get ds -n kube-system kured | grep -E 'kured.*5.*5.*5.*5.*5'" + command: "kubectl get ds -n kube-system kured | grep -E 'kured.*3.*3.*3.*3.*3'" - name: Create reboot sentinel files run: | diff --git a/tests/kind/create-reboot-sentinels.sh b/tests/kind/create-reboot-sentinels.sh index e95dc3b..51bd127 100755 --- a/tests/kind/create-reboot-sentinels.sh +++ b/tests/kind/create-reboot-sentinels.sh @@ -4,9 +4,10 @@ KUBECTL_CMD="${KUBECTL_CMD:-kubectl}" SENTINEL_FILE="${SENTINEL_FILE:-/var/run/reboot-required}" -echo "Creating reboot sentinel on all nodes" +echo "Creating reboot sentinel on worker nodes" -for nodename in $("$KUBECTL_CMD" get nodes -o name); do +# To speed up the system, let's not kill the control plane. +for nodename in $("$KUBECTL_CMD" get nodes -o name | grep -v control-plane); do docker exec "${nodename/node\//}" hostname docker exec "${nodename/node\//}" touch "${SENTINEL_FILE}" done diff --git a/tests/kind/follow-coordinated-reboot.sh b/tests/kind/follow-coordinated-reboot.sh index e5885fd..4559ec1 100755 --- a/tests/kind/follow-coordinated-reboot.sh +++ b/tests/kind/follow-coordinated-reboot.sh @@ -1,6 +1,6 @@ #!/usr/bin/env bash -NODECOUNT=${NODECOUNT:-5} +NODECOUNT=${NODECOUNT:-2} KUBECTL_CMD="${KUBECTL_CMD:-kubectl}" DEBUG="${DEBUG:-false}" CONTAINER_NAME_FORMAT=${CONTAINER_NAME_FORMAT:-"chart-testing-*"} @@ -35,10 +35,12 @@ trap gather_logs_and_cleanup EXIT declare -A was_unschedulable declare -A has_recovered -max_attempts="60" -sleep_time=60 +max_attempts="200" +sleep_time=5 attempt_num=1 +# Get docker info of each of those kind containers. If one has crashed, restart it. + set +o errexit echo "There are $NODECOUNT nodes in the cluster" until [ ${#was_unschedulable[@]} == "$NODECOUNT" ] && [ ${#has_recovered[@]} == "$NODECOUNT" ] @@ -52,13 +54,14 @@ do # cat "$tmp_dir"/node_output #fi - "$KUBECTL_CMD" get nodes -o custom-columns=NAME:.metadata.name,SCHEDULABLE:.spec.unschedulable --no-headers > "$tmp_dir"/node_output + "$KUBECTL_CMD" get nodes -o custom-columns=NAME:.metadata.name,SCHEDULABLE:.spec.unschedulable --no-headers | grep -v control-plane > "$tmp_dir"/node_output if [[ "$DEBUG" == "true" ]]; then # This is useful to see if a node gets stuck after drain, and doesn't # come back up. echo "Result of command $KUBECTL_CMD get nodes ... showing unschedulable nodes:" cat "$tmp_dir"/node_output fi + while read -r node; do unschedulable=$(echo "$node" | grep true | cut -f 1 -d ' ') if [ -n "$unschedulable" ] && [ -z ${was_unschedulable["$unschedulable"]+x} ] ; then @@ -70,6 +73,12 @@ do echo "$schedulable has recovered!" has_recovered["$schedulable"]=1 fi + + # If the container has crashed, restart it. + node_name=$(echo "$node" | cut -f 1 -d ' ') + stopped_container_id=$(docker container ls --filter=name="$node_name" --filter=status=exited -q) + if [ -n "$stopped_container_id" ]; then echo "Node $stopped_container_id needs restart"; docker start "$stopped_container_id"; echo "Container started."; fi + done < "$tmp_dir"/node_output if [[ "${#has_recovered[@]}" == "$NODECOUNT" ]]; then