test: add local rollback lab
This commit is contained in:
parent
a622739053
commit
0dd6f003f3
4 changed files with 349 additions and 0 deletions
33
scripts/k3d-nereus.yaml
Normal file
33
scripts/k3d-nereus.yaml
Normal file
|
|
@ -0,0 +1,33 @@
|
|||
apiVersion: k3d.io/v1alpha5
|
||||
kind: Simple
|
||||
metadata:
|
||||
name: nereus
|
||||
|
||||
# Three nodes so this is a real multi-node k3s, not a single-node toy.
|
||||
# Mirrors the two Fedora VMs plus room for a rollout to spread across nodes.
|
||||
servers: 1
|
||||
agents: 2
|
||||
|
||||
image: docker.io/rancher/k3s:v1.33.4-k3s1
|
||||
|
||||
ports:
|
||||
# Traefik on 8080/8443 so it never fights the Mini PC stack or a local dev server.
|
||||
- port: 8080:80
|
||||
nodeFilters: [loadbalancer]
|
||||
- port: 8443:443
|
||||
nodeFilters: [loadbalancer]
|
||||
|
||||
options:
|
||||
k3d:
|
||||
wait: true
|
||||
timeout: 180s
|
||||
k3s:
|
||||
extraArgs:
|
||||
# kube-prometheus-stack scrapes these; k3s binds them to localhost by default.
|
||||
- arg: --kube-controller-manager-arg=bind-address=0.0.0.0
|
||||
nodeFilters: [server:*]
|
||||
- arg: --kube-scheduler-arg=bind-address=0.0.0.0
|
||||
nodeFilters: [server:*]
|
||||
kubeconfig:
|
||||
updateDefaultKubeconfig: true
|
||||
switchCurrentContext: true
|
||||
71
scripts/k3d/analysis-harness/harness.yaml
Normal file
71
scripts/k3d/analysis-harness/harness.yaml
Normal file
|
|
@ -0,0 +1,71 @@
|
|||
# Mechanism test for the automated rollback, run before the real API exists.
|
||||
#
|
||||
# It proves the exact chain the project depends on: Argo Rollouts pauses a
|
||||
# blue-green promotion, runs an AnalysisRun, that run queries Prometheus over
|
||||
# the network, evaluates the result against a threshold, and aborts the rollout
|
||||
# on failure. The only thing faked is the metric itself.
|
||||
#
|
||||
# Throwaway. deploy/rollouts/ holds the real thing.
|
||||
---
|
||||
apiVersion: v1
|
||||
kind: Service
|
||||
metadata: {name: probe-active, namespace: nereus}
|
||||
spec:
|
||||
selector: {app: probe}
|
||||
ports: [{port: 80, targetPort: 8080}]
|
||||
---
|
||||
apiVersion: v1
|
||||
kind: Service
|
||||
metadata: {name: probe-preview, namespace: nereus}
|
||||
spec:
|
||||
selector: {app: probe}
|
||||
ports: [{port: 80, targetPort: 8080}]
|
||||
---
|
||||
apiVersion: argoproj.io/v1alpha1
|
||||
kind: AnalysisTemplate
|
||||
metadata: {name: error-rate, namespace: nereus}
|
||||
spec:
|
||||
metrics:
|
||||
- name: error-rate
|
||||
interval: 10s
|
||||
count: 3
|
||||
# Same shape as the real query will use: fail when the error ratio is
|
||||
# above the threshold. failureLimit 0 means one bad sample aborts.
|
||||
successCondition: "result[0] < 0.05"
|
||||
failureLimit: 0
|
||||
provider:
|
||||
prometheus:
|
||||
address: http://kube-prometheus-stack-prometheus.observability.svc.cluster.local:9090
|
||||
query: "{{args.query}}"
|
||||
args:
|
||||
- name: query
|
||||
---
|
||||
apiVersion: argoproj.io/v1alpha1
|
||||
kind: Rollout
|
||||
metadata: {name: probe, namespace: nereus}
|
||||
spec:
|
||||
replicas: 1
|
||||
revisionHistoryLimit: 2
|
||||
selector:
|
||||
matchLabels: {app: probe}
|
||||
template:
|
||||
metadata:
|
||||
labels: {app: probe}
|
||||
spec:
|
||||
containers:
|
||||
- name: web
|
||||
image: nginxinc/nginx-unprivileged:alpine
|
||||
ports: [{containerPort: 8080}]
|
||||
resources:
|
||||
requests: {cpu: 10m, memory: 16Mi}
|
||||
strategy:
|
||||
blueGreen:
|
||||
activeService: probe-active
|
||||
previewService: probe-preview
|
||||
autoPromotionEnabled: true
|
||||
prePromotionAnalysis:
|
||||
templates:
|
||||
- templateName: error-rate
|
||||
args:
|
||||
- name: query
|
||||
value: "vector(0.0)"
|
||||
65
scripts/k3d/kube-prometheus-stack.values.yaml
Normal file
65
scripts/k3d/kube-prometheus-stack.values.yaml
Normal file
|
|
@ -0,0 +1,65 @@
|
|||
# Values for the k3d (level 2) cluster only. Kept lean so the whole stack fits
|
||||
# alongside the app on a laptop-sized machine.
|
||||
#
|
||||
# No credentials here. Grafana's admin password is generated by the chart into a
|
||||
# secret; read it with:
|
||||
# kubectl -n observability get secret kube-prometheus-stack-grafana \
|
||||
# -o jsonpath='{.data.admin-password}' | base64 -d
|
||||
|
||||
# k3s does not expose these the way a kubeadm cluster does. Left enabled they
|
||||
# produce permanently-firing "target down" alerts that bury the real ones.
|
||||
kubeEtcd:
|
||||
enabled: false
|
||||
kubeProxy:
|
||||
enabled: false
|
||||
|
||||
kubeControllerManager:
|
||||
service:
|
||||
port: 10257
|
||||
targetPort: 10257
|
||||
serviceMonitor:
|
||||
https: true
|
||||
insecureSkipVerify: true
|
||||
kubeScheduler:
|
||||
service:
|
||||
port: 10259
|
||||
targetPort: 10259
|
||||
serviceMonitor:
|
||||
https: true
|
||||
insecureSkipVerify: true
|
||||
|
||||
prometheus:
|
||||
prometheusSpec:
|
||||
retention: 6h
|
||||
resources:
|
||||
requests: {cpu: 100m, memory: 512Mi}
|
||||
limits: {memory: 1500Mi}
|
||||
# Pick up ServiceMonitors from every namespace, not just the chart's own
|
||||
# release. The app lives in `nereus` and must be scraped from there.
|
||||
serviceMonitorSelectorNilUsesHelmValues: false
|
||||
podMonitorSelectorNilUsesHelmValues: false
|
||||
ruleSelectorNilUsesHelmValues: false
|
||||
storageSpec:
|
||||
volumeClaimTemplate:
|
||||
spec:
|
||||
accessModes: [ReadWriteOnce]
|
||||
resources:
|
||||
requests:
|
||||
storage: 5Gi
|
||||
|
||||
grafana:
|
||||
defaultDashboardsTimezone: browser
|
||||
resources:
|
||||
requests: {cpu: 50m, memory: 128Mi}
|
||||
limits: {memory: 512Mi}
|
||||
|
||||
alertmanager:
|
||||
alertmanagerSpec:
|
||||
resources:
|
||||
requests: {cpu: 25m, memory: 64Mi}
|
||||
limits: {memory: 256Mi}
|
||||
|
||||
prometheusOperator:
|
||||
resources:
|
||||
requests: {cpu: 50m, memory: 128Mi}
|
||||
limits: {memory: 512Mi}
|
||||
180
scripts/k3d/lab.sh
Executable file
180
scripts/k3d/lab.sh
Executable file
|
|
@ -0,0 +1,180 @@
|
|||
#!/usr/bin/env bash
|
||||
|
||||
set -euo pipefail
|
||||
|
||||
readonly SCRIPT_DIR="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)"
|
||||
readonly REPO_ROOT="$(cd "${SCRIPT_DIR}/../.." && pwd)"
|
||||
readonly CLUSTER_NAME="nereus"
|
||||
readonly CONTEXT="k3d-${CLUSTER_NAME}"
|
||||
readonly ARGO_CHART_VERSION="2.41.1"
|
||||
readonly PROMETHEUS_CHART_VERSION="88.5.2"
|
||||
|
||||
require_commands() {
|
||||
local command_name
|
||||
for command_name in docker k3d helm kubectl; do
|
||||
command -v "${command_name}" >/dev/null || {
|
||||
echo "missing required command: ${command_name}" >&2
|
||||
exit 1
|
||||
}
|
||||
done
|
||||
docker info >/dev/null
|
||||
}
|
||||
|
||||
use_context() {
|
||||
kubectl config use-context "${CONTEXT}" >/dev/null
|
||||
}
|
||||
|
||||
cluster_exists() {
|
||||
k3d cluster get "${CLUSTER_NAME}" >/dev/null 2>&1
|
||||
}
|
||||
|
||||
up() {
|
||||
require_commands
|
||||
if ! cluster_exists; then
|
||||
k3d cluster create --config "${REPO_ROOT}/scripts/k3d-nereus.yaml"
|
||||
else
|
||||
k3d cluster start "${CLUSTER_NAME}"
|
||||
fi
|
||||
use_context
|
||||
|
||||
helm repo add argo https://argoproj.github.io/argo-helm --force-update
|
||||
helm repo add prometheus-community https://prometheus-community.github.io/helm-charts --force-update
|
||||
helm repo update
|
||||
|
||||
helm upgrade --install argo-rollouts argo/argo-rollouts \
|
||||
--namespace argo-rollouts \
|
||||
--create-namespace \
|
||||
--version "${ARGO_CHART_VERSION}" \
|
||||
--wait \
|
||||
--timeout 5m
|
||||
|
||||
helm upgrade --install kube-prometheus-stack prometheus-community/kube-prometheus-stack \
|
||||
--namespace observability \
|
||||
--create-namespace \
|
||||
--version "${PROMETHEUS_CHART_VERSION}" \
|
||||
--values "${SCRIPT_DIR}/kube-prometheus-stack.values.yaml" \
|
||||
--wait \
|
||||
--timeout 10m
|
||||
|
||||
kubectl create namespace nereus --dry-run=client -o yaml | kubectl apply -f -
|
||||
kubectl apply -f "${SCRIPT_DIR}/analysis-harness/harness.yaml"
|
||||
check
|
||||
}
|
||||
|
||||
check() {
|
||||
require_commands
|
||||
cluster_exists || {
|
||||
echo "cluster ${CLUSTER_NAME} does not exist; run $0 up" >&2
|
||||
exit 1
|
||||
}
|
||||
use_context
|
||||
|
||||
kubectl wait node --all --for=condition=Ready --timeout=3m
|
||||
kubectl wait deployment --all --namespace argo-rollouts --for=condition=Available --timeout=3m
|
||||
kubectl wait deployment --all --namespace observability --for=condition=Available --timeout=5m
|
||||
kubectl get rollout probe --namespace nereus >/dev/null
|
||||
kubectl get analysistemplate error-rate --namespace nereus >/dev/null
|
||||
echo "k3d rollback lab is ready"
|
||||
}
|
||||
|
||||
analysis_name() {
|
||||
kubectl get rollout probe --namespace nereus \
|
||||
-o jsonpath='{.status.blueGreen.prePromotionAnalysisRunStatus.name}'
|
||||
}
|
||||
|
||||
wait_for_analysis() {
|
||||
local previous_name="$1"
|
||||
local expected_phase="$2"
|
||||
local analysis_run=""
|
||||
local phase=""
|
||||
local attempt
|
||||
|
||||
for attempt in {1..90}; do
|
||||
analysis_run="$(analysis_name)"
|
||||
if [[ -n "${analysis_run}" && "${analysis_run}" != "${previous_name}" ]]; then
|
||||
phase="$(kubectl get analysisrun "${analysis_run}" --namespace nereus -o jsonpath='{.status.phase}')"
|
||||
if [[ "${phase}" == "${expected_phase}" ]]; then
|
||||
echo "${analysis_run} reached ${expected_phase}"
|
||||
return 0
|
||||
fi
|
||||
if [[ "${phase}" == "Error" || "${phase}" == "Inconclusive" ]]; then
|
||||
echo "${analysis_run} ended unexpectedly with ${phase}" >&2
|
||||
return 1
|
||||
fi
|
||||
fi
|
||||
sleep 2
|
||||
done
|
||||
|
||||
echo "analysis did not reach ${expected_phase} within 180 seconds" >&2
|
||||
return 1
|
||||
}
|
||||
|
||||
set_proof_revision() {
|
||||
local query="$1"
|
||||
local revision="$2"
|
||||
kubectl patch rollout probe --namespace nereus --type=merge --patch \
|
||||
"{\"spec\":{\"strategy\":{\"blueGreen\":{\"prePromotionAnalysis\":{\"args\":[{\"name\":\"query\",\"value\":\"${query}\"}]}}},\"template\":{\"metadata\":{\"annotations\":{\"nereus.fiwlabs.dev/proof\":\"${revision}\"}}}}}"
|
||||
}
|
||||
|
||||
prove() {
|
||||
check
|
||||
|
||||
local proof_id
|
||||
local previous_analysis
|
||||
local healthy_analysis
|
||||
local healthy_revision=""
|
||||
local active_revision=""
|
||||
local stable_revision=""
|
||||
local attempt
|
||||
proof_id="$(date -u +%Y%m%d%H%M%S)"
|
||||
|
||||
previous_analysis="$(analysis_name)"
|
||||
set_proof_revision "vector(0.0)" "healthy-${proof_id}"
|
||||
wait_for_analysis "${previous_analysis}" Successful
|
||||
healthy_analysis="$(analysis_name)"
|
||||
|
||||
for attempt in {1..30}; do
|
||||
healthy_revision="$(kubectl get rollout probe --namespace nereus -o jsonpath='{.status.stableRS}')"
|
||||
active_revision="$(kubectl get service probe-active --namespace nereus -o jsonpath='{.spec.selector.rollouts-pod-template-hash}')"
|
||||
if [[ -n "${healthy_revision}" && "${active_revision}" == "${healthy_revision}" ]]; then
|
||||
break
|
||||
fi
|
||||
sleep 2
|
||||
done
|
||||
[[ -n "${healthy_revision}" && "${active_revision}" == "${healthy_revision}" ]] || {
|
||||
echo "healthy revision was not promoted" >&2
|
||||
return 1
|
||||
}
|
||||
|
||||
set_proof_revision "vector(1.0)" "failing-${proof_id}"
|
||||
wait_for_analysis "${healthy_analysis}" Failed
|
||||
stable_revision="$(kubectl get rollout probe --namespace nereus -o jsonpath='{.status.stableRS}')"
|
||||
active_revision="$(kubectl get service probe-active --namespace nereus -o jsonpath='{.spec.selector.rollouts-pod-template-hash}')"
|
||||
[[ "${stable_revision}" == "${healthy_revision}" && "${active_revision}" == "${healthy_revision}" ]] || {
|
||||
echo "failed revision replaced the active healthy revision" >&2
|
||||
return 1
|
||||
}
|
||||
|
||||
set_proof_revision "vector(0.0)" "healthy-${proof_id}"
|
||||
echo "rollback proof passed; active revision stayed ${healthy_revision}"
|
||||
}
|
||||
|
||||
destroy() {
|
||||
require_commands
|
||||
if cluster_exists; then
|
||||
k3d cluster delete "${CLUSTER_NAME}"
|
||||
else
|
||||
echo "cluster ${CLUSTER_NAME} does not exist"
|
||||
fi
|
||||
}
|
||||
|
||||
case "${1:-}" in
|
||||
up) up ;;
|
||||
check) check ;;
|
||||
prove) prove ;;
|
||||
destroy) destroy ;;
|
||||
*)
|
||||
echo "usage: $0 {up|check|prove|destroy}" >&2
|
||||
exit 2
|
||||
;;
|
||||
esac
|
||||
Loading…
Add table
Reference in a new issue