diff --git a/.github/workflows/lint.yaml b/.github/workflows/lint.yaml index 144a5cd..de64355 100644 --- a/.github/workflows/lint.yaml +++ b/.github/workflows/lint.yaml @@ -36,4 +36,7 @@ jobs: sudo chmod +x /usr/bin/yq - name: Lint Helm chart and rules - run: make lint \ No newline at end of file + run: make lint + + - name: Validate Prometheus config correctness + run: make test/prometheus diff --git a/.github/workflows/nightly-build.yaml b/.github/workflows/nightly-build.yaml index e64f92c..0d59494 100644 --- a/.github/workflows/nightly-build.yaml +++ b/.github/workflows/nightly-build.yaml @@ -28,9 +28,9 @@ jobs: sudo wget https://github.com/mikefarah/yq/releases/download/v4.42.1/yq_linux_amd64 -O /usr/bin/yq &&\ sudo chmod +x /usr/bin/yq - - name: make build + - name: Validate chart run: | - make build > output.log 2>&1 + { make build && make test/prometheus; } > output.log 2>&1 continue-on-error: false - name: Upload script output @@ -45,4 +45,4 @@ jobs: with: title: nightly build failure content-filepath: output.log - assignees: dannykopping \ No newline at end of file + assignees: dannykopping diff --git a/.github/workflows/release.yaml b/.github/workflows/release.yaml index 23d45a6..c2dccaf 100644 --- a/.github/workflows/release.yaml +++ b/.github/workflows/release.yaml @@ -47,7 +47,12 @@ jobs: - name: Install helm uses: azure/setup-helm@v4 with: - version: v3.9.2 + version: v3.17.1 + + - name: Install yq + run: | + sudo wget https://github.com/mikefarah/yq/releases/download/v4.42.1/yq_linux_amd64 -O /usr/bin/yq &&\ + sudo chmod +x /usr/bin/yq - name: Publish Helm Chart if: ${{ !inputs.dry_run }} diff --git a/Makefile b/Makefile index 3a1fe74..9b5b83a 100644 --- a/Makefile +++ b/Makefile @@ -25,6 +25,10 @@ build: ./scripts/compile.sh .PHONY: build +test/prometheus: build + ./scripts/test-prometheus.sh +.PHONY: test/prometheus + lint/rules: lint/helm/prometheus-rules .PHONY: lint/rules @@ -55,4 +59,4 @@ publish-%: readme: chart/version-sync go install github.com/norwoodj/helm-docs/cmd/helm-docs@latest helm-docs --output-file ../README.md \ - --values-file=values.yaml --chart-search-root=coder-observability --template-files=../README.gotmpl \ No newline at end of file + --values-file=values.yaml --chart-search-root=coder-observability --template-files=../README.gotmpl diff --git a/README.md b/README.md index dbf5d5f..c6198cf 100644 --- a/README.md +++ b/README.md @@ -22,7 +22,7 @@ Logs will be scraped from all pods in the Kubernetes cluster. ```bash helm repo add coder-observability https://helm.coder.com/observability -helm upgrade --install coder-observability coder-observability/coder-observability --version 0.7.3 --namespace coder-observability --create-namespace +helm upgrade --install coder-observability coder-observability/coder-observability --version 0.7.4 --namespace coder-observability --create-namespace ``` ## Requirements @@ -587,7 +587,7 @@ values which are defined [here](https://github.com/grafana/helm-charts/tree/main | prometheus.server.extraConfigmapMounts[0].optional | bool | `true` | | | prometheus.server.extraConfigmapMounts[0].readonly | bool | `true` | | | prometheus.server.extraFlags[0] | string | `"web.enable-lifecycle"` | | -| prometheus.server.extraFlags[1] | string | `"enable-feature=remote-write-receiver"` | | +| prometheus.server.extraFlags[1] | string | `"web.enable-remote-write-receiver"` | | | prometheus.server.fullnameOverride | string | `"prometheus"` | | | prometheus.server.global.evaluation_interval | string | `"30s"` | | | prometheus.server.persistentVolume.enabled | bool | `true` | | @@ -598,7 +598,6 @@ values which are defined [here](https://github.com/grafana/helm-charts/tree/main | prometheus.server.service.type | string | `"ClusterIP"` | | | prometheus.server.statefulSet.enabled | bool | `true` | | | prometheus.serverFiles."prometheus.yml".rule_files[0] | string | `"/etc/config/alerts/*.yaml"` | | -| prometheus.serverFiles."prometheus.yml".scrape_configs | list | `[]` | | | prometheus.testFramework.enabled | bool | `false` | | | pyroscope.alloy.enabled | bool | `false` | | | pyroscope.enabled | bool | `false` | | diff --git a/coder-observability/Chart.yaml b/coder-observability/Chart.yaml index 64552f7..1df2afe 100644 --- a/coder-observability/Chart.yaml +++ b/coder-observability/Chart.yaml @@ -2,7 +2,7 @@ apiVersion: v2 name: coder-observability description: Gain insights into your Coder deployment type: application -version: 0.7.3 +version: 0.7.4 dependencies: - name: pyroscope condition: pyroscope.enabled diff --git a/coder-observability/values.yaml b/coder-observability/values.yaml index 873eea9..deefce5 100644 --- a/coder-observability/values.yaml +++ b/coder-observability/values.yaml @@ -515,7 +515,7 @@ prometheus: type: ClusterIP extraFlags: - web.enable-lifecycle - - enable-feature=remote-write-receiver + - web.enable-remote-write-receiver extraConfigmapMounts: - name: alerts mountPath: /etc/config/alerts @@ -525,8 +525,6 @@ prometheus: serverFiles: prometheus.yml: - # disables scraping of metrics by the Prometheus helm chart since this is managed by the collector - scrape_configs: [] # use custom rule files to be able to render templates (can't do that in values.yaml, unless that value is evaluated by a tpl call) rule_files: - /etc/config/alerts/*.yaml diff --git a/compiled/resources.yaml b/compiled/resources.yaml index 7c6db5d..ad508b4 100644 --- a/compiled/resources.yaml +++ b/compiled/resources.yaml @@ -751,7 +751,6 @@ data: scrape_configs: rule_files: - /etc/config/alerts/*.yaml - scrape_configs: [] alerting: alertmanagers: - kubernetes_sd_configs: @@ -13458,7 +13457,7 @@ spec: - --web.console.libraries=/etc/prometheus/console_libraries - --web.console.templates=/etc/prometheus/consoles - --web.enable-lifecycle - - --enable-feature=remote-write-receiver + - --web.enable-remote-write-receiver - --log.level=debug ports: - containerPort: 9090 diff --git a/scripts/publish.sh b/scripts/publish.sh index b51878a..278f0e0 100755 --- a/scripts/publish.sh +++ b/scripts/publish.sh @@ -4,10 +4,11 @@ set -euox pipefail version=$("$(dirname "${BASH_SOURCE[0]}")/version.sh") mkdir -p build/helm helm package coder-observability --version=${version} --dependency-update --destination build/helm +"$(dirname "${BASH_SOURCE[0]}")/test-prometheus.sh" "build/helm/coder-observability-${version}.tgz" gsutil cp gs://helm.coder.com/observability/index.yaml build/helm/index.yaml helm repo index build/helm --url https://helm.coder.com/observability --merge build/helm/index.yaml gsutil -h "Cache-Control:no-cache,max-age=0" cp build/helm/index.yaml gs://helm.coder.com/observability/ gsutil -h "Cache-Control:no-cache,max-age=0" cp build/helm/coder-observability-${version}.tgz gs://helm.coder.com/observability/ gsutil -h "Cache-Control:no-cache,max-age=0" cp artifacthub-repo.yaml gs://helm.coder.com/observability/ -echo $version \ No newline at end of file +echo $version diff --git a/scripts/test-prometheus.sh b/scripts/test-prometheus.sh new file mode 100755 index 0000000..8e76ede --- /dev/null +++ b/scripts/test-prometheus.sh @@ -0,0 +1,50 @@ +#!/usr/bin/env bash +set -euo pipefail + +temp_dir="$(mktemp -d)" +container="" +cleanup() { + if [[ -n "$container" ]]; then + docker logs "$container" + docker rm -f "$container" >/dev/null + fi + rm -rf "$temp_dir" +} +trap cleanup EXIT + +# Compile fresh manifests so the config, image, and arguments all come from the +# same chart. Accept a packaged chart too, for the pre-publish check. +helm template coder-observability "${1:-coder-observability}" \ + --namespace coder-observability > "$temp_dir/resources.yaml" +yq -e 'select(.kind == "StatefulSet" and .metadata.name == "prometheus") | + .spec.template.spec.containers[] | select(.name == "prometheus-server")' \ + "$temp_dir/resources.yaml" > "$temp_dir/server.yaml" +image="$(yq -er '.image' "$temp_dir/server.yaml")" +mapfile -t args < <(yq -r '.args[]' "$temp_dir/server.yaml") + +mkdir -p "$temp_dir/config/alerts" "$temp_dir/serviceaccount" +# Extract the embedded config as text, without parsing/re-serializing its YAML. +yq -er 'select(.kind == "ConfigMap" and .metadata.name == "prometheus") | + .data."prometheus.yml"' "$temp_dir/resources.yaml" > "$temp_dir/config/prometheus.yml" +yq -e 'select(.kind == "ConfigMap" and .metadata.name == "coder-metrics-alerts") | + .data' "$temp_dir/resources.yaml" > "$temp_dir/alerts.yaml" +for key in $(yq -r 'keys | .[]' "$temp_dir/alerts.yaml"); do + KEY="$key" yq -r '.[strenv(KEY)]' "$temp_dir/alerts.yaml" > "$temp_dir/config/alerts/$key" +done + +# The default config uses in-cluster Alertmanager discovery. Supply local +# service-account files and point discovery at loopback, not a real cluster. +printf 'smoke-test-only\n' > "$temp_dir/serviceaccount/token" +cp /etc/ssl/certs/ca-certificates.crt "$temp_dir/serviceaccount/ca.crt" +container="$(docker run -d \ + -v "$temp_dir/config:/etc/config:ro" \ + -v "$temp_dir/serviceaccount:/var/run/secrets/kubernetes.io/serviceaccount:ro" \ + --tmpfs /data:rw,mode=1777 \ + -p 127.0.0.1::9090 \ + -e KUBERNETES_SERVICE_HOST=127.0.0.1 -e KUBERNETES_SERVICE_PORT=9 \ + "$image" "${args[@]}")" + +# Readiness succeeds only after Prometheus has loaded its config and rules. +address="$(docker port "$container" 9090/tcp)" +curl --fail --silent --show-error --retry 30 --retry-all-errors \ + --retry-delay 1 --retry-max-time 60 --max-time 2 "http://$address/-/ready"