diff --git a/scripts/k3d-nereus.yaml b/scripts/k3d-nereus.yaml new file mode 100644 index 0000000..a27be6f --- /dev/null +++ b/scripts/k3d-nereus.yaml @@ -0,0 +1,33 @@ +apiVersion: k3d.io/v1alpha5 +kind: Simple +metadata: + name: nereus + +# Three nodes so this is a real multi-node k3s, not a single-node toy. +# Mirrors the two Fedora VMs plus room for a rollout to spread across nodes. +servers: 1 +agents: 2 + +image: docker.io/rancher/k3s:v1.33.4-k3s1 + +ports: + # Traefik on 8080/8443 so it never fights the Mini PC stack or a local dev server. + - port: 8080:80 + nodeFilters: [loadbalancer] + - port: 8443:443 + nodeFilters: [loadbalancer] + +options: + k3d: + wait: true + timeout: 180s + k3s: + extraArgs: + # kube-prometheus-stack scrapes these; k3s binds them to localhost by default. + - arg: --kube-controller-manager-arg=bind-address=0.0.0.0 + nodeFilters: [server:*] + - arg: --kube-scheduler-arg=bind-address=0.0.0.0 + nodeFilters: [server:*] + kubeconfig: + updateDefaultKubeconfig: true + switchCurrentContext: true diff --git a/scripts/k3d/analysis-harness/harness.yaml b/scripts/k3d/analysis-harness/harness.yaml new file mode 100644 index 0000000..f4ba46b --- /dev/null +++ b/scripts/k3d/analysis-harness/harness.yaml @@ -0,0 +1,71 @@ +# Mechanism test for the automated rollback, run before the real API exists. +# +# It proves the exact chain the project depends on: Argo Rollouts pauses a +# blue-green promotion, runs an AnalysisRun, that run queries Prometheus over +# the network, evaluates the result against a threshold, and aborts the rollout +# on failure. The only thing faked is the metric itself. +# +# Throwaway. deploy/rollouts/ holds the real thing. +--- +apiVersion: v1 +kind: Service +metadata: {name: probe-active, namespace: nereus} +spec: + selector: {app: probe} + ports: [{port: 80, targetPort: 8080}] +--- +apiVersion: v1 +kind: Service +metadata: {name: probe-preview, namespace: nereus} +spec: + selector: {app: probe} + ports: [{port: 80, targetPort: 8080}] +--- +apiVersion: argoproj.io/v1alpha1 +kind: AnalysisTemplate +metadata: {name: error-rate, namespace: nereus} +spec: + metrics: + - name: error-rate + interval: 10s + count: 3 + # Same shape as the real query will use: fail when the error ratio is + # above the threshold. failureLimit 0 means one bad sample aborts. + successCondition: "result[0] < 0.05" + failureLimit: 0 + provider: + prometheus: + address: http://kube-prometheus-stack-prometheus.observability.svc.cluster.local:9090 + query: "{{args.query}}" + args: + - name: query +--- +apiVersion: argoproj.io/v1alpha1 +kind: Rollout +metadata: {name: probe, namespace: nereus} +spec: + replicas: 1 + revisionHistoryLimit: 2 + selector: + matchLabels: {app: probe} + template: + metadata: + labels: {app: probe} + spec: + containers: + - name: web + image: nginxinc/nginx-unprivileged:alpine + ports: [{containerPort: 8080}] + resources: + requests: {cpu: 10m, memory: 16Mi} + strategy: + blueGreen: + activeService: probe-active + previewService: probe-preview + autoPromotionEnabled: true + prePromotionAnalysis: + templates: + - templateName: error-rate + args: + - name: query + value: "vector(0.0)" diff --git a/scripts/k3d/kube-prometheus-stack.values.yaml b/scripts/k3d/kube-prometheus-stack.values.yaml new file mode 100644 index 0000000..3669000 --- /dev/null +++ b/scripts/k3d/kube-prometheus-stack.values.yaml @@ -0,0 +1,65 @@ +# Values for the k3d (level 2) cluster only. Kept lean so the whole stack fits +# alongside the app on a laptop-sized machine. +# +# No credentials here. Grafana's admin password is generated by the chart into a +# secret; read it with: +# kubectl -n observability get secret kube-prometheus-stack-grafana \ +# -o jsonpath='{.data.admin-password}' | base64 -d + +# k3s does not expose these the way a kubeadm cluster does. Left enabled they +# produce permanently-firing "target down" alerts that bury the real ones. +kubeEtcd: + enabled: false +kubeProxy: + enabled: false + +kubeControllerManager: + service: + port: 10257 + targetPort: 10257 + serviceMonitor: + https: true + insecureSkipVerify: true +kubeScheduler: + service: + port: 10259 + targetPort: 10259 + serviceMonitor: + https: true + insecureSkipVerify: true + +prometheus: + prometheusSpec: + retention: 6h + resources: + requests: {cpu: 100m, memory: 512Mi} + limits: {memory: 1500Mi} + # Pick up ServiceMonitors from every namespace, not just the chart's own + # release. The app lives in `nereus` and must be scraped from there. + serviceMonitorSelectorNilUsesHelmValues: false + podMonitorSelectorNilUsesHelmValues: false + ruleSelectorNilUsesHelmValues: false + storageSpec: + volumeClaimTemplate: + spec: + accessModes: [ReadWriteOnce] + resources: + requests: + storage: 5Gi + +grafana: + defaultDashboardsTimezone: browser + resources: + requests: {cpu: 50m, memory: 128Mi} + limits: {memory: 512Mi} + +alertmanager: + alertmanagerSpec: + resources: + requests: {cpu: 25m, memory: 64Mi} + limits: {memory: 256Mi} + +prometheusOperator: + resources: + requests: {cpu: 50m, memory: 128Mi} + limits: {memory: 512Mi} diff --git a/scripts/k3d/lab.sh b/scripts/k3d/lab.sh new file mode 100755 index 0000000..bceadf5 --- /dev/null +++ b/scripts/k3d/lab.sh @@ -0,0 +1,180 @@ +#!/usr/bin/env bash + +set -euo pipefail + +readonly SCRIPT_DIR="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)" +readonly REPO_ROOT="$(cd "${SCRIPT_DIR}/../.." && pwd)" +readonly CLUSTER_NAME="nereus" +readonly CONTEXT="k3d-${CLUSTER_NAME}" +readonly ARGO_CHART_VERSION="2.41.1" +readonly PROMETHEUS_CHART_VERSION="88.5.2" + +require_commands() { + local command_name + for command_name in docker k3d helm kubectl; do + command -v "${command_name}" >/dev/null || { + echo "missing required command: ${command_name}" >&2 + exit 1 + } + done + docker info >/dev/null +} + +use_context() { + kubectl config use-context "${CONTEXT}" >/dev/null +} + +cluster_exists() { + k3d cluster get "${CLUSTER_NAME}" >/dev/null 2>&1 +} + +up() { + require_commands + if ! cluster_exists; then + k3d cluster create --config "${REPO_ROOT}/scripts/k3d-nereus.yaml" + else + k3d cluster start "${CLUSTER_NAME}" + fi + use_context + + helm repo add argo https://argoproj.github.io/argo-helm --force-update + helm repo add prometheus-community https://prometheus-community.github.io/helm-charts --force-update + helm repo update + + helm upgrade --install argo-rollouts argo/argo-rollouts \ + --namespace argo-rollouts \ + --create-namespace \ + --version "${ARGO_CHART_VERSION}" \ + --wait \ + --timeout 5m + + helm upgrade --install kube-prometheus-stack prometheus-community/kube-prometheus-stack \ + --namespace observability \ + --create-namespace \ + --version "${PROMETHEUS_CHART_VERSION}" \ + --values "${SCRIPT_DIR}/kube-prometheus-stack.values.yaml" \ + --wait \ + --timeout 10m + + kubectl create namespace nereus --dry-run=client -o yaml | kubectl apply -f - + kubectl apply -f "${SCRIPT_DIR}/analysis-harness/harness.yaml" + check +} + +check() { + require_commands + cluster_exists || { + echo "cluster ${CLUSTER_NAME} does not exist; run $0 up" >&2 + exit 1 + } + use_context + + kubectl wait node --all --for=condition=Ready --timeout=3m + kubectl wait deployment --all --namespace argo-rollouts --for=condition=Available --timeout=3m + kubectl wait deployment --all --namespace observability --for=condition=Available --timeout=5m + kubectl get rollout probe --namespace nereus >/dev/null + kubectl get analysistemplate error-rate --namespace nereus >/dev/null + echo "k3d rollback lab is ready" +} + +analysis_name() { + kubectl get rollout probe --namespace nereus \ + -o jsonpath='{.status.blueGreen.prePromotionAnalysisRunStatus.name}' +} + +wait_for_analysis() { + local previous_name="$1" + local expected_phase="$2" + local analysis_run="" + local phase="" + local attempt + + for attempt in {1..90}; do + analysis_run="$(analysis_name)" + if [[ -n "${analysis_run}" && "${analysis_run}" != "${previous_name}" ]]; then + phase="$(kubectl get analysisrun "${analysis_run}" --namespace nereus -o jsonpath='{.status.phase}')" + if [[ "${phase}" == "${expected_phase}" ]]; then + echo "${analysis_run} reached ${expected_phase}" + return 0 + fi + if [[ "${phase}" == "Error" || "${phase}" == "Inconclusive" ]]; then + echo "${analysis_run} ended unexpectedly with ${phase}" >&2 + return 1 + fi + fi + sleep 2 + done + + echo "analysis did not reach ${expected_phase} within 180 seconds" >&2 + return 1 +} + +set_proof_revision() { + local query="$1" + local revision="$2" + kubectl patch rollout probe --namespace nereus --type=merge --patch \ + "{\"spec\":{\"strategy\":{\"blueGreen\":{\"prePromotionAnalysis\":{\"args\":[{\"name\":\"query\",\"value\":\"${query}\"}]}}},\"template\":{\"metadata\":{\"annotations\":{\"nereus.fiwlabs.dev/proof\":\"${revision}\"}}}}}" +} + +prove() { + check + + local proof_id + local previous_analysis + local healthy_analysis + local healthy_revision="" + local active_revision="" + local stable_revision="" + local attempt + proof_id="$(date -u +%Y%m%d%H%M%S)" + + previous_analysis="$(analysis_name)" + set_proof_revision "vector(0.0)" "healthy-${proof_id}" + wait_for_analysis "${previous_analysis}" Successful + healthy_analysis="$(analysis_name)" + + for attempt in {1..30}; do + healthy_revision="$(kubectl get rollout probe --namespace nereus -o jsonpath='{.status.stableRS}')" + active_revision="$(kubectl get service probe-active --namespace nereus -o jsonpath='{.spec.selector.rollouts-pod-template-hash}')" + if [[ -n "${healthy_revision}" && "${active_revision}" == "${healthy_revision}" ]]; then + break + fi + sleep 2 + done + [[ -n "${healthy_revision}" && "${active_revision}" == "${healthy_revision}" ]] || { + echo "healthy revision was not promoted" >&2 + return 1 + } + + set_proof_revision "vector(1.0)" "failing-${proof_id}" + wait_for_analysis "${healthy_analysis}" Failed + stable_revision="$(kubectl get rollout probe --namespace nereus -o jsonpath='{.status.stableRS}')" + active_revision="$(kubectl get service probe-active --namespace nereus -o jsonpath='{.spec.selector.rollouts-pod-template-hash}')" + [[ "${stable_revision}" == "${healthy_revision}" && "${active_revision}" == "${healthy_revision}" ]] || { + echo "failed revision replaced the active healthy revision" >&2 + return 1 + } + + set_proof_revision "vector(0.0)" "healthy-${proof_id}" + echo "rollback proof passed; active revision stayed ${healthy_revision}" +} + +destroy() { + require_commands + if cluster_exists; then + k3d cluster delete "${CLUSTER_NAME}" + else + echo "cluster ${CLUSTER_NAME} does not exist" + fi +} + +case "${1:-}" in + up) up ;; + check) check ;; + prove) prove ;; + destroy) destroy ;; + *) + echo "usage: $0 {up|check|prove|destroy}" >&2 + exit 2 + ;; +esac