diff --git a/Makefile b/Makefile index 15771dcc..2374df72 100644 --- a/Makefile +++ b/Makefile @@ -30,6 +30,12 @@ run-smi: -slack-url=https://hooks.slack.com/services/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 \ -slack-channel="devops-alerts" +run-gloo: + go run cmd/flagger/* -kubeconfig=$$HOME/.kube/config -log-level=info -mesh-provider=gloo -namespace=gloo \ + -metrics-server=https://prometheus.istio.weavedx.com \ + -slack-url=https://hooks.slack.com/services/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 \ + -slack-channel="devops-alerts" + build: docker build -t weaveworks/flagger:$(TAG) . -f Dockerfile diff --git a/artifacts/flagger/account.yaml b/artifacts/flagger/account.yaml index d31e7568..b7cf0e28 100644 --- a/artifacts/flagger/account.yaml +++ b/artifacts/flagger/account.yaml @@ -64,6 +64,21 @@ rules: resources: - trafficsplits verbs: ["*"] + - apiGroups: + - gloo.solo.io + resources: + - settings + - upstreams + - upstreamgroups + - proxies + - virtualservices + verbs: ["*"] + - apiGroups: + - gateway.solo.io + resources: + - virtualservices + - gateways + verbs: ["*"] - nonResourceURLs: - /version verbs: diff --git a/artifacts/gloo/canary.yaml b/artifacts/gloo/canary.yaml new file mode 100644 index 00000000..3be05a78 --- /dev/null +++ b/artifacts/gloo/canary.yaml @@ -0,0 +1,36 @@ +apiVersion: flagger.app/v1alpha3 +kind: Canary +metadata: + name: podinfo + namespace: test +spec: + targetRef: + apiVersion: apps/v1 + kind: Deployment + name: podinfo + progressDeadlineSeconds: 60 + autoscalerRef: + apiVersion: autoscaling/v2beta1 + kind: HorizontalPodAutoscaler + name: podinfo + service: + port: 9898 + canaryAnalysis: + interval: 10s + threshold: 10 + maxWeight: 50 + stepWeight: 5 + metrics: + - name: request-success-rate + threshold: 99 + interval: 1m + - name: request-duration + threshold: 500 + interval: 30s + webhooks: + - name: load-test + url: http://flagger-loadtester.test/ + timeout: 5s + metadata: + type: cmd + cmd: "hey -z 1m -q 10 -c 2 http://gloo.example.com/" diff --git a/artifacts/gloo/deployment.yaml b/artifacts/gloo/deployment.yaml new file mode 100644 index 00000000..57ed8a41 --- /dev/null +++ b/artifacts/gloo/deployment.yaml @@ -0,0 +1,67 @@ +apiVersion: apps/v1 +kind: Deployment +metadata: + name: podinfo + namespace: test + labels: + app: podinfo +spec: + minReadySeconds: 5 + revisionHistoryLimit: 5 + progressDeadlineSeconds: 60 + strategy: + rollingUpdate: + maxUnavailable: 0 + type: RollingUpdate + selector: + matchLabels: + app: podinfo + template: + metadata: + annotations: + prometheus.io/scrape: "true" + labels: + app: podinfo + spec: + containers: + - name: podinfod + image: quay.io/stefanprodan/podinfo:1.4.0 + imagePullPolicy: IfNotPresent + ports: + - containerPort: 9898 + name: http + protocol: TCP + command: + - ./podinfo + - --port=9898 + - --level=info + - --random-delay=false + - --random-error=false + env: + - name: PODINFO_UI_COLOR + value: blue + livenessProbe: + exec: + command: + - podcli + - check + - http + - localhost:9898/healthz + initialDelaySeconds: 5 + timeoutSeconds: 5 + readinessProbe: + exec: + command: + - podcli + - check + - http + - localhost:9898/readyz + initialDelaySeconds: 5 + timeoutSeconds: 5 + resources: + limits: + cpu: 2000m + memory: 512Mi + requests: + cpu: 100m + memory: 64Mi diff --git a/artifacts/gloo/hpa.yaml b/artifacts/gloo/hpa.yaml new file mode 100644 index 00000000..48ec76e8 --- /dev/null +++ b/artifacts/gloo/hpa.yaml @@ -0,0 +1,19 @@ +apiVersion: autoscaling/v2beta1 +kind: HorizontalPodAutoscaler +metadata: + name: podinfo + namespace: test +spec: + scaleTargetRef: + apiVersion: apps/v1 + kind: Deployment + name: podinfo + minReplicas: 1 + maxReplicas: 4 + metrics: + - type: Resource + resource: + name: cpu + # scale up if usage is above + # 99% of the requested CPU (100m) + targetAverageUtilization: 99 diff --git a/artifacts/gloo/virtual-service.yaml b/artifacts/gloo/virtual-service.yaml new file mode 100644 index 00000000..53d010dd --- /dev/null +++ b/artifacts/gloo/virtual-service.yaml @@ -0,0 +1,17 @@ +apiVersion: gateway.solo.io/v1 +kind: VirtualService +metadata: + name: podinfo + namespace: test +spec: + virtualHost: + domains: + - '*' + name: podinfo.default + routes: + - matcher: + prefix: / + routeAction: + upstreamGroup: + name: podinfo + namespace: gloo diff --git a/charts/flagger/templates/rbac.yaml b/charts/flagger/templates/rbac.yaml index 782e1df1..cb5c52ef 100644 --- a/charts/flagger/templates/rbac.yaml +++ b/charts/flagger/templates/rbac.yaml @@ -60,6 +60,21 @@ rules: resources: - trafficsplits verbs: ["*"] + - apiGroups: + - gloo.solo.io + resources: + - settings + - upstreams + - upstreamgroups + - proxies + - virtualservices + verbs: ["*"] + - apiGroups: + - gateway.solo.io + resources: + - virtualservices + - gateways + verbs: ["*"] - nonResourceURLs: - /version verbs: diff --git a/docs/diagrams/flagger-gloo-overview.png b/docs/diagrams/flagger-gloo-overview.png new file mode 100644 index 00000000..393428e9 Binary files /dev/null and b/docs/diagrams/flagger-gloo-overview.png differ diff --git a/docs/gitbook/install/flagger-install-on-eks-appmesh.md b/docs/gitbook/install/flagger-install-on-eks-appmesh.md index 761ba436..7e4edde1 100644 --- a/docs/gitbook/install/flagger-install-on-eks-appmesh.md +++ b/docs/gitbook/install/flagger-install-on-eks-appmesh.md @@ -163,7 +163,7 @@ Deploy Grafana in the _**appmesh-system**_ namespace: ```bash helm upgrade -i flagger-grafana flagger/grafana \ --namespace=appmesh-system \ ---set url=http://prometheus.appmesh-system:9090 +--set url=http://flagger-prometheus.appmesh-system:9090 ``` You can access Grafana using port forwarding: diff --git a/docs/gitbook/usage/gloo-progressive-delivery.md b/docs/gitbook/usage/gloo-progressive-delivery.md new file mode 100644 index 00000000..bff29cee --- /dev/null +++ b/docs/gitbook/usage/gloo-progressive-delivery.md @@ -0,0 +1,366 @@ +# NGNIX Ingress Controller Canary Deployments + +This guide shows you how to use the [Gloo](https://gloo.solo.io/) ingress controller and Flagger to automate canary deployments. + +![Flagger Gloo Ingress Controller](https://raw.githubusercontent.com/weaveworks/flagger/master/docs/diagrams/flagger-gloo-overview.png) + +### Prerequisites + +Flagger requires a Kubernetes cluster **v1.11** or newer and Gloo ingress **0.13.29** or newer. + +Install Gloo with Helm: + +```bash +helm repo add gloo https://storage.googleapis.com/solo-public-helm + +helm upgrade -i gloo gloo/gloo \ +--namespace gloo-system +``` + +Install Flagger and the Prometheus add-on in the same namespace as Gloo: + +```bash +helm repo add flagger https://flagger.app + +helm upgrade -i flagger flagger/flagger \ +--namespace gloo-system \ +--set prometheus.install=true \ +--set meshProvider=gloo +``` + +Optionally you can enable Slack notifications: + +```bash +helm upgrade -i flagger flagger/flagger \ +--reuse-values \ +--namespace gloo-system \ +--set slack.url=https://hooks.slack.com/services/YOUR/SLACK/WEBHOOK \ +--set slack.channel=general \ +--set slack.user=flagger +``` + +### Bootstrap + +Flagger takes a Kubernetes deployment and optionally a horizontal pod autoscaler (HPA), +then creates a series of objects (Kubernetes deployments, ClusterIP services and Gloo upstream groups). +These objects expose the application outside the cluster and drive the canary analysis and promotion. + +Create a test namespace: + +```bash +kubectl create ns test +``` + +Create a deployment and a horizontal pod autoscaler: + +```bash +kubectl apply -f ${REPO}/artifacts/gloo/deployment.yaml +kubectl apply -f ${REPO}/artifacts/gloo/hpa.yaml +``` + +Deploy the load testing service to generate traffic during the canary analysis: + +```bash +helm upgrade -i flagger-loadtester flagger/loadtester \ +--namespace=test +``` + +Create an virtual service definition that references an upstream group that will be generated by Flagger +(replace `app.example.com` with your own domain): + +```yaml +apiVersion: gateway.solo.io/v1 +kind: VirtualService +metadata: + name: podinfo + namespace: test +spec: + virtualHost: + domains: + - 'app.example.com' + name: podinfo.test + routes: + - matcher: + prefix: / + routeAction: + upstreamGroup: + name: podinfo + namespace: test +``` + +Save the above resource as podinfo-virtualservice.yaml and then apply it: + +```bash +kubectl apply -f ./podinfo-virtualservice.yaml +``` + +Create a canary custom resource (replace `app.example.com` with your own domain): + +```yaml +apiVersion: flagger.app/v1alpha3 +kind: Canary +metadata: + name: podinfo + namespace: test +spec: + # deployment reference + targetRef: + apiVersion: apps/v1 + kind: Deployment + name: podinfo + # HPA reference (optional) + autoscalerRef: + apiVersion: autoscaling/v2beta1 + kind: HorizontalPodAutoscaler + name: podinfo + # the maximum time in seconds for the canary deployment + # to make progress before it is rollback (default 600s) + progressDeadlineSeconds: 60 + service: + # container port + port: 9898 + canaryAnalysis: + # schedule interval (default 60s) + interval: 10s + # max number of failed metric checks before rollback + threshold: 5 + # max traffic percentage routed to canary + # percentage (0-100) + maxWeight: 50 + # canary increment step + # percentage (0-100) + stepWeight: 5 + # Gloo Prometheus checks + metrics: + - name: request-success-rate + # minimum req success rate (non 5xx responses) + # percentage (0-100) + threshold: 99 + interval: 1m + - name: request-duration + # maximum req duration P99 + # milliseconds + threshold: 500 + interval: 30s + # load testing (optional) + webhooks: + - name: load-test + url: http://flagger-loadtester.test/ + timeout: 5s + metadata: + type: cmd + cmd: "hey -z 1m -q 10 -c 2 http://app.example.com/" +``` + +Save the above resource as podinfo-canary.yaml and then apply it: + +```bash +kubectl apply -f ./podinfo-canary.yaml +``` + +After a couple of seconds Flagger will create the canary objects: + +```bash +# applied +deployment.apps/podinfo +horizontalpodautoscaler.autoscaling/podinfo +virtualservices.gateway.solo.io/podinfo +canary.flagger.app/podinfo + +# generated +deployment.apps/podinfo-primary +horizontalpodautoscaler.autoscaling/podinfo-primary +service/podinfo +service/podinfo-canary +service/podinfo-primary +upstreamgroups.gloo.solo.io/podinfo +``` + +When the bootstrap finishes Flagger will set the canary status to initialized: + +```bash +kubectl -n test get canary podinfo + +NAME STATUS WEIGHT LASTTRANSITIONTIME +podinfo Initialized 0 2019-05-17T08:09:51Z +``` + +### Automated canary promotion + +Flagger implements a control loop that gradually shifts traffic to the canary while measuring key performance indicators +like HTTP requests success rate, requests average duration and pod health. +Based on analysis of the KPIs a canary is promoted or aborted, and the analysis result is published to Slack. + +![Flagger Canary Stages](https://raw.githubusercontent.com/weaveworks/flagger/master/docs/diagrams/flagger-canary-steps.png) + +Trigger a canary deployment by updating the container image: + +```bash +kubectl -n test set image deployment/podinfo \ +podinfod=quay.io/stefanprodan/podinfo:1.4.1 +``` + +Flagger detects that the deployment revision changed and starts a new rollout: + +```text +kubectl -n test describe canary/podinfo + +Status: + Canary Weight: 0 + Failed Checks: 0 + Phase: Succeeded +Events: + Type Reason Age From Message + ---- ------ ---- ---- ------- + Normal Synced 3m flagger New revision detected podinfo.test + Normal Synced 3m flagger Scaling up podinfo.test + Warning Synced 3m flagger Waiting for podinfo.test rollout to finish: 0 of 1 updated replicas are available + Normal Synced 3m flagger Advance podinfo.test canary weight 5 + Normal Synced 3m flagger Advance podinfo.test canary weight 10 + Normal Synced 3m flagger Advance podinfo.test canary weight 15 + Normal Synced 2m flagger Advance podinfo.test canary weight 20 + Normal Synced 2m flagger Advance podinfo.test canary weight 25 + Normal Synced 1m flagger Advance podinfo.test canary weight 30 + Normal Synced 1m flagger Advance podinfo.test canary weight 35 + Normal Synced 55s flagger Advance podinfo.test canary weight 40 + Normal Synced 45s flagger Advance podinfo.test canary weight 45 + Normal Synced 35s flagger Advance podinfo.test canary weight 50 + Normal Synced 25s flagger Copying podinfo.test template spec to podinfo-primary.test + Warning Synced 15s flagger Waiting for podinfo-primary.test rollout to finish: 1 of 2 updated replicas are available + Normal Synced 5s flagger Promotion completed! Scaling down podinfo.test +``` + +**Note** that if you apply new changes to the deployment during the canary analysis, Flagger will restart the analysis. + +You can monitor all canaries with: + +```bash +watch kubectl get canaries --all-namespaces + +NAMESPACE NAME STATUS WEIGHT LASTTRANSITIONTIME +test podinfo Progressing 15 2019-05-17T14:05:07Z +prod frontend Succeeded 0 2019-05-17T16:15:07Z +prod backend Failed 0 2019-05-17T17:05:07Z +``` + +### Automated rollback + +During the canary analysis you can generate HTTP 500 errors and high latency to test if Flagger pauses and rolls back the faulted version. + +Trigger another canary deployment: + +```bash +kubectl -n test set image deployment/podinfo \ +podinfod=quay.io/stefanprodan/podinfo:1.4.2 +``` + +Generate HTTP 500 errors: + +```bash +watch curl http://app.example.com/status/500 +``` + +Generate high latency: + +```bash +watch curl http://app.example.com/delay/2 +``` + +When the number of failed checks reaches the canary analysis threshold, the traffic is routed back to the primary, +the canary is scaled to zero and the rollout is marked as failed. + +```text +kubectl -n test describe canary/podinfo + +Status: + Canary Weight: 0 + Failed Checks: 10 + Phase: Failed +Events: + Type Reason Age From Message + ---- ------ ---- ---- ------- + Normal Synced 3m flagger Starting canary deployment for podinfo.test + Normal Synced 3m flagger Advance podinfo.test canary weight 5 + Normal Synced 3m flagger Advance podinfo.test canary weight 10 + Normal Synced 3m flagger Advance podinfo.test canary weight 15 + Normal Synced 3m flagger Halt podinfo.test advancement success rate 69.17% < 99% + Normal Synced 2m flagger Halt podinfo.test advancement success rate 61.39% < 99% + Normal Synced 2m flagger Halt podinfo.test advancement success rate 55.06% < 99% + Normal Synced 2m flagger Halt podinfo.test advancement success rate 47.00% < 99% + Normal Synced 2m flagger (combined from similar events): Halt podinfo.test advancement success rate 38.08% < 99% + Warning Synced 1m flagger Rolling back podinfo.test failed checks threshold reached 10 + Warning Synced 1m flagger Canary failed! Scaling down podinfo.test +``` + +### Custom metrics + +The canary analysis can be extended with Prometheus queries. + +The demo app is instrumented with Prometheus so you can create a custom check that will use the HTTP request duration +histogram to validate the canary. + +Edit the canary analysis and add the following metric: + +```yaml + canaryAnalysis: + metrics: + - name: "404s percentage" + threshold: 5 + query: | + 100 - sum( + rate( + http_request_duration_seconds_count{ + kubernetes_namespace="test", + kubernetes_pod_name=~"podinfo-[0-9a-zA-Z]+(-[0-9a-zA-Z]+)" + status!="404" + }[1m] + ) + ) + / + sum( + rate( + http_request_duration_seconds_count{ + kubernetes_namespace="test", + kubernetes_pod_name=~"podinfo-[0-9a-zA-Z]+(-[0-9a-zA-Z]+)" + }[1m] + ) + ) * 100 +``` + +The above configuration validates the canary by checking if the HTTP 404 req/sec percentage is below 5 +percent of the total traffic. If the 404s rate reaches the 5% threshold, then the canary fails. + +Trigger a canary deployment by updating the container image: + +```bash +kubectl -n test set image deployment/podinfo \ +podinfod=quay.io/stefanprodan/podinfo:1.4.3 +``` + +Generate 404s: + +```bash +watch curl http://app.example.com/status/400 +``` + +Watch Flagger logs: + +``` +kubectl -n gloo-system logs deployment/flagger -f | jq .msg + +Starting canary deployment for podinfo.test +Advance podinfo.test canary weight 5 +Advance podinfo.test canary weight 10 +Advance podinfo.test canary weight 15 +Halt podinfo.test advancement 404s percentage 6.20 > 5 +Halt podinfo.test advancement 404s percentage 6.45 > 5 +Halt podinfo.test advancement 404s percentage 7.60 > 5 +Halt podinfo.test advancement 404s percentage 8.69 > 5 +Halt podinfo.test advancement 404s percentage 9.70 > 5 +Rolling back podinfo.test failed checks threshold reached 5 +Canary failed! Scaling down podinfo.test +``` + +If you have Slack configured, Flagger will send a notification with the reason why the canary failed. + + diff --git a/pkg/router/gloo.go b/pkg/router/gloo.go index 62303e20..7cf735c4 100644 --- a/pkg/router/gloo.go +++ b/pkg/router/gloo.go @@ -152,8 +152,10 @@ func (gr *GlooRouter) writeUpstreamGroupRuleForCanary(canary *flaggerv1.Canary, targetName := canary.Spec.TargetRef.Name if oldUg, err := gr.ugClient.Read(ug.Metadata.Namespace, ug.Metadata.Name, solokitclients.ReadOpts{}); err != nil { - // ignore not exist errors.. - if !solokiterror.IsNotExist(err) { + if solokiterror.IsNotExist(err) { + gr.logger.With("canary", fmt.Sprintf("%s.%s", canary.Name, canary.Namespace)). + Infof("UpstreamGroup %s created", ug.Metadata.Name) + } else { return fmt.Errorf("RoutingRule %s.%s read failed: %v", targetName, canary.Namespace, err) } } else { @@ -181,6 +183,5 @@ func (gr *GlooRouter) writeUpstreamGroupRuleForCanary(canary *flaggerv1.Canary, if err != nil { return fmt.Errorf("UpstreamGroup %s.%s update failed: %v", targetName, canary.Namespace, err) } - gr.logger.With("canary", fmt.Sprintf("%s.%s", canary.Name, canary.Namespace)).Infof("UpstreamGroup %s updated", ug.Metadata.Name) return nil }