From 21671c9bd7d208190356284d1a0a345b03cd81df Mon Sep 17 00:00:00 2001 From: "coder[bot]" <137810501+coder[bot]@users.noreply.github.com> Date: Tue, 6 Oct 2026 12:56:44 +0000 Subject: [PATCH 1/8] fix: remove duplicate scrape_configs from prometheus.yml --- README.md | 1 - coder-observability/values.yaml | 2 -- compiled/resources.yaml | 1 - 3 files changed, 4 deletions(-) diff --git a/README.md b/README.md index dbf5d5f..38c81b7 100644 --- a/README.md +++ b/README.md @@ -598,7 +598,6 @@ values which are defined [here](https://github.com/grafana/helm-charts/tree/main | prometheus.server.service.type | string | `"ClusterIP"` | | | prometheus.server.statefulSet.enabled | bool | `true` | | | prometheus.serverFiles."prometheus.yml".rule_files[0] | string | `"/etc/config/alerts/*.yaml"` | | -| prometheus.serverFiles."prometheus.yml".scrape_configs | list | `[]` | | | prometheus.testFramework.enabled | bool | `false` | | | pyroscope.alloy.enabled | bool | `false` | | | pyroscope.enabled | bool | `false` | | diff --git a/coder-observability/values.yaml b/coder-observability/values.yaml index 873eea9..4ba49de 100644 --- a/coder-observability/values.yaml +++ b/coder-observability/values.yaml @@ -525,8 +525,6 @@ prometheus: serverFiles: prometheus.yml: - # disables scraping of metrics by the Prometheus helm chart since this is managed by the collector - scrape_configs: [] # use custom rule files to be able to render templates (can't do that in values.yaml, unless that value is evaluated by a tpl call) rule_files: - /etc/config/alerts/*.yaml diff --git a/compiled/resources.yaml b/compiled/resources.yaml index 7c6db5d..d4d238d 100644 --- a/compiled/resources.yaml +++ b/compiled/resources.yaml @@ -751,7 +751,6 @@ data: scrape_configs: rule_files: - /etc/config/alerts/*.yaml - scrape_configs: [] alerting: alertmanagers: - kubernetes_sd_configs: From b556d88242671fd8ed60624b928744e303722028 Mon Sep 17 00:00:00 2001 From: "coder[bot]" <137810501+coder[bot]@users.noreply.github.com> Date: Tue, 6 Oct 2026 13:05:02 +0000 Subject: [PATCH 2/8] chore: sync chart version to v0.7.4 --- README.md | 4 +++- coder-observability/Chart.yaml | 2 +- 2 files changed, 4 insertions(+), 2 deletions(-) diff --git a/README.md b/README.md index 38c81b7..c0ef245 100644 --- a/README.md +++ b/README.md @@ -22,7 +22,7 @@ Logs will be scraped from all pods in the Kubernetes cluster. ```bash helm repo add coder-observability https://helm.coder.com/observability -helm upgrade --install coder-observability coder-observability/coder-observability --version 0.7.3 --namespace coder-observability --create-namespace +helm upgrade --install coder-observability coder-observability/coder-observability --version 0.7.4 --namespace coder-observability --create-namespace ``` ## Requirements @@ -622,3 +622,5 @@ values which are defined [here](https://github.com/grafana/helm-charts/tree/main | tempo.tempo.reportingEnabled | bool | `false` | | | tempo.tempo.retention | string | `"336h"` | | +---------------------------------------------- +Autogenerated from chart metadata using [helm-docs v1.14.2](https://github.com/norwoodj/helm-docs/releases/v1.14.2) diff --git a/coder-observability/Chart.yaml b/coder-observability/Chart.yaml index 64552f7..1df2afe 100644 --- a/coder-observability/Chart.yaml +++ b/coder-observability/Chart.yaml @@ -2,7 +2,7 @@ apiVersion: v2 name: coder-observability description: Gain insights into your Coder deployment type: application -version: 0.7.3 +version: 0.7.4 dependencies: - name: pyroscope condition: pyroscope.enabled From a9ecd8ec2779bffc9828f3342dfbfbf1548e2e10 Mon Sep 17 00:00:00 2001 From: "coder[bot]" <137810501+coder[bot]@users.noreply.github.com> Date: Tue, 6 Oct 2026 13:05:12 +0000 Subject: [PATCH 3/8] docs: drop helm-docs version footer --- README.md | 2 -- 1 file changed, 2 deletions(-) diff --git a/README.md b/README.md index c0ef245..61f562c 100644 --- a/README.md +++ b/README.md @@ -622,5 +622,3 @@ values which are defined [here](https://github.com/grafana/helm-charts/tree/main | tempo.tempo.reportingEnabled | bool | `false` | | | tempo.tempo.retention | string | `"336h"` | | ----------------------------------------------- -Autogenerated from chart metadata using [helm-docs v1.14.2](https://github.com/norwoodj/helm-docs/releases/v1.14.2) From 01e8edc98638499587e47b6de5c46d8af0b90f7f Mon Sep 17 00:00:00 2001 From: "coder[bot]" <137810501+coder[bot]@users.noreply.github.com> Date: Tue, 6 Oct 2026 13:07:03 +0000 Subject: [PATCH 4/8] chore: bump chart version to 0.7.5 --- README.md | 2 +- coder-observability/Chart.yaml | 2 +- 2 files changed, 2 insertions(+), 2 deletions(-) diff --git a/README.md b/README.md index 61f562c..a602cec 100644 --- a/README.md +++ b/README.md @@ -22,7 +22,7 @@ Logs will be scraped from all pods in the Kubernetes cluster. ```bash helm repo add coder-observability https://helm.coder.com/observability -helm upgrade --install coder-observability coder-observability/coder-observability --version 0.7.4 --namespace coder-observability --create-namespace +helm upgrade --install coder-observability coder-observability/coder-observability --version 0.7.5 --namespace coder-observability --create-namespace ``` ## Requirements diff --git a/coder-observability/Chart.yaml b/coder-observability/Chart.yaml index 1df2afe..dad0cd3 100644 --- a/coder-observability/Chart.yaml +++ b/coder-observability/Chart.yaml @@ -2,7 +2,7 @@ apiVersion: v2 name: coder-observability description: Gain insights into your Coder deployment type: application -version: 0.7.4 +version: 0.7.5 dependencies: - name: pyroscope condition: pyroscope.enabled From 143acbd465e55434977fe18ad78995ad2e8e97b8 Mon Sep 17 00:00:00 2001 From: "coder[bot]" <137810501+coder[bot]@users.noreply.github.com> Date: Tue, 6 Oct 2026 13:11:01 +0000 Subject: [PATCH 5/8] Revert "chore: bump chart version to 0.7.5" This reverts commit 01e8edc98638499587e47b6de5c46d8af0b90f7f. --- README.md | 2 +- coder-observability/Chart.yaml | 2 +- 2 files changed, 2 insertions(+), 2 deletions(-) diff --git a/README.md b/README.md index a602cec..61f562c 100644 --- a/README.md +++ b/README.md @@ -22,7 +22,7 @@ Logs will be scraped from all pods in the Kubernetes cluster. ```bash helm repo add coder-observability https://helm.coder.com/observability -helm upgrade --install coder-observability coder-observability/coder-observability --version 0.7.5 --namespace coder-observability --create-namespace +helm upgrade --install coder-observability coder-observability/coder-observability --version 0.7.4 --namespace coder-observability --create-namespace ``` ## Requirements diff --git a/coder-observability/Chart.yaml b/coder-observability/Chart.yaml index dad0cd3..1df2afe 100644 --- a/coder-observability/Chart.yaml +++ b/coder-observability/Chart.yaml @@ -2,7 +2,7 @@ apiVersion: v2 name: coder-observability description: Gain insights into your Coder deployment type: application -version: 0.7.5 +version: 0.7.4 dependencies: - name: pyroscope condition: pyroscope.enabled From 6438c44b6e380832ff64e0805fd87cd5571a8f25 Mon Sep 17 00:00:00 2001 From: Danny Kopping Date: Wed, 7 Oct 2026 10:57:08 +0200 Subject: [PATCH 6/8] fix: enable Prometheus remote write and test runtime before release --- .github/workflows/nightly-build.yaml | 6 +- .github/workflows/release.yaml | 7 +- Makefile | 9 +- README.md | 2 +- TESTING_KIND.md | 27 +++++- coder-observability/values.yaml | 2 +- compiled/resources.yaml | 2 +- scripts/publish.sh | 3 +- scripts/test-prometheus.py | 132 +++++++++++++++++++++++++++ 9 files changed, 179 insertions(+), 11 deletions(-) create mode 100644 scripts/test-prometheus.py diff --git a/.github/workflows/nightly-build.yaml b/.github/workflows/nightly-build.yaml index e64f92c..f79d6a8 100644 --- a/.github/workflows/nightly-build.yaml +++ b/.github/workflows/nightly-build.yaml @@ -28,9 +28,9 @@ jobs: sudo wget https://github.com/mikefarah/yq/releases/download/v4.42.1/yq_linux_amd64 -O /usr/bin/yq &&\ sudo chmod +x /usr/bin/yq - - name: make build + - name: Build and test Prometheus run: | - make build > output.log 2>&1 + { make build && make test/prometheus; } > output.log 2>&1 continue-on-error: false - name: Upload script output @@ -45,4 +45,4 @@ jobs: with: title: nightly build failure content-filepath: output.log - assignees: dannykopping \ No newline at end of file + assignees: dannykopping diff --git a/.github/workflows/release.yaml b/.github/workflows/release.yaml index 23d45a6..c2dccaf 100644 --- a/.github/workflows/release.yaml +++ b/.github/workflows/release.yaml @@ -47,7 +47,12 @@ jobs: - name: Install helm uses: azure/setup-helm@v4 with: - version: v3.9.2 + version: v3.17.1 + + - name: Install yq + run: | + sudo wget https://github.com/mikefarah/yq/releases/download/v4.42.1/yq_linux_amd64 -O /usr/bin/yq &&\ + sudo chmod +x /usr/bin/yq - name: Publish Helm Chart if: ${{ !inputs.dry_run }} diff --git a/Makefile b/Makefile index 3a1fe74..e47e124 100644 --- a/Makefile +++ b/Makefile @@ -10,7 +10,7 @@ SHELL := bash all: lint .PHONY: all -lint: build lint/helm lint/rules readme +lint: build lint/helm test/prometheus lint/rules readme ./scripts/check-unstaged.sh .PHONY: lint @@ -25,6 +25,11 @@ build: ./scripts/compile.sh .PHONY: build +# Requires chart dependencies (make build), Docker, Helm, yq, and Python 3. +test/prometheus: + python3 ./scripts/test-prometheus.py +.PHONY: test/prometheus + lint/rules: lint/helm/prometheus-rules .PHONY: lint/rules @@ -55,4 +60,4 @@ publish-%: readme: chart/version-sync go install github.com/norwoodj/helm-docs/cmd/helm-docs@latest helm-docs --output-file ../README.md \ - --values-file=values.yaml --chart-search-root=coder-observability --template-files=../README.gotmpl \ No newline at end of file + --values-file=values.yaml --chart-search-root=coder-observability --template-files=../README.gotmpl diff --git a/README.md b/README.md index 61f562c..c6198cf 100644 --- a/README.md +++ b/README.md @@ -587,7 +587,7 @@ values which are defined [here](https://github.com/grafana/helm-charts/tree/main | prometheus.server.extraConfigmapMounts[0].optional | bool | `true` | | | prometheus.server.extraConfigmapMounts[0].readonly | bool | `true` | | | prometheus.server.extraFlags[0] | string | `"web.enable-lifecycle"` | | -| prometheus.server.extraFlags[1] | string | `"enable-feature=remote-write-receiver"` | | +| prometheus.server.extraFlags[1] | string | `"web.enable-remote-write-receiver"` | | | prometheus.server.fullnameOverride | string | `"prometheus"` | | | prometheus.server.global.evaluation_interval | string | `"30s"` | | | prometheus.server.persistentVolume.enabled | bool | `true` | | diff --git a/TESTING_KIND.md b/TESTING_KIND.md index e1c9786..9ac9b77 100644 --- a/TESTING_KIND.md +++ b/TESTING_KIND.md @@ -1,5 +1,30 @@ # Using kind to test observability +## Prometheus smoke test + +Before running a full cluster test, run the same Prometheus smoke test used by CI: + +```bash +helm dependency build coder-observability +make test/prometheus +``` + +This requires Docker, Helm, yq v4, and Python 3. It renders the chart, runs +`promtool check config` (including alert rules) from the rendered Prometheus image, +then starts that image with the rendered arguments. It checks readiness, zero +scrape targets (the collector handles scraping), and a remote-written metric that +can be queried back. Kubernetes discovery uses local service-account fixtures; +this does not test discovery or the collector end to end. + +The release script runs the same check against the packaged `.tgz` before uploading +it. To test an existing package locally: + +```bash +python3 scripts/test-prometheus.py path/to/coder-observability-VERSION.tgz +``` + +## Full cluster test +
If using nix @@ -128,4 +153,4 @@ To view the monitoring services, you can port forward their UI's. - Grafana: `kubectl -n coder-observability port-forward svc/grafana 3000:80` - Prometheus: `kubectl -n coder-observability port-forward svc/prometheus 3001:80` - Pyroscope: `kubectl port-forward svc/pyroscope 3002:4040` -- Grafana Agent: `kubectl port-forward svc/grafana-agent 3003:80` \ No newline at end of file +- Grafana Agent: `kubectl port-forward svc/grafana-agent 3003:80` diff --git a/coder-observability/values.yaml b/coder-observability/values.yaml index 4ba49de..deefce5 100644 --- a/coder-observability/values.yaml +++ b/coder-observability/values.yaml @@ -515,7 +515,7 @@ prometheus: type: ClusterIP extraFlags: - web.enable-lifecycle - - enable-feature=remote-write-receiver + - web.enable-remote-write-receiver extraConfigmapMounts: - name: alerts mountPath: /etc/config/alerts diff --git a/compiled/resources.yaml b/compiled/resources.yaml index d4d238d..ad508b4 100644 --- a/compiled/resources.yaml +++ b/compiled/resources.yaml @@ -13457,7 +13457,7 @@ spec: - --web.console.libraries=/etc/prometheus/console_libraries - --web.console.templates=/etc/prometheus/consoles - --web.enable-lifecycle - - --enable-feature=remote-write-receiver + - --web.enable-remote-write-receiver - --log.level=debug ports: - containerPort: 9090 diff --git a/scripts/publish.sh b/scripts/publish.sh index b51878a..ca59fbf 100755 --- a/scripts/publish.sh +++ b/scripts/publish.sh @@ -4,10 +4,11 @@ set -euox pipefail version=$("$(dirname "${BASH_SOURCE[0]}")/version.sh") mkdir -p build/helm helm package coder-observability --version=${version} --dependency-update --destination build/helm +python3 "$(dirname "${BASH_SOURCE[0]}")/test-prometheus.py" "build/helm/coder-observability-${version}.tgz" gsutil cp gs://helm.coder.com/observability/index.yaml build/helm/index.yaml helm repo index build/helm --url https://helm.coder.com/observability --merge build/helm/index.yaml gsutil -h "Cache-Control:no-cache,max-age=0" cp build/helm/index.yaml gs://helm.coder.com/observability/ gsutil -h "Cache-Control:no-cache,max-age=0" cp build/helm/coder-observability-${version}.tgz gs://helm.coder.com/observability/ gsutil -h "Cache-Control:no-cache,max-age=0" cp artifacthub-repo.yaml gs://helm.coder.com/observability/ -echo $version \ No newline at end of file +echo $version diff --git a/scripts/test-prometheus.py b/scripts/test-prometheus.py new file mode 100644 index 0000000..b8115be --- /dev/null +++ b/scripts/test-prometheus.py @@ -0,0 +1,132 @@ +#!/usr/bin/env python3 +"""Smoke-test the default chart, or a packaged chart, with its rendered image.""" + +import json +import pathlib +import shutil +import ssl +import subprocess +import sys +import tempfile +import time +import urllib.error +import urllib.request + + +def run(*args, **kwargs): + return subprocess.run(args, check=True, text=True, **kwargs) + + +def output(*args, **kwargs): + try: + return run(*args, capture_output=True, **kwargs).stdout.strip() + except subprocess.CalledProcessError as error: + print(error.stdout + error.stderr, file=sys.stderr) + raise + + +def check(chart): + with tempfile.TemporaryDirectory(prefix="prometheus-smoke-") as tmp: + root = pathlib.Path(tmp) + manifest = output( + "helm", "template", "coder-observability", chart, + "--namespace", "coder-observability", + ) + # Only parse the outer manifests. Re-serializing prometheus.yml here + # could silently remove the duplicate YAML keys we need promtool to catch. + resources = json.loads(output( + "yq", "eval-all", "-o=json", "[.]", "-", input=manifest, + )) + configmaps = { + resource["metadata"]["name"]: resource["data"] + for resource in resources if resource and resource["kind"] == "ConfigMap" + } + pod = next( + resource["spec"]["template"]["spec"] for resource in resources + if resource and resource["kind"] == "StatefulSet" + and resource["metadata"]["name"] == "prometheus" + ) + server = next(c for c in pod["containers"] if c["name"] == "prometheus-server") + image = server["image"] + print(f"Testing {chart} with {image}", flush=True) + + config = root / "config" + alerts = config / "alerts" + config.mkdir() + alerts.mkdir() + for name, content in configmaps["prometheus"].items(): + # The chart's extra alerts mount shadows this legacy ConfigMap key. + if name == "alerts": + continue + (config / name).write_text(content) + for name, content in configmaps["coder-metrics-alerts"].items(): + (alerts / name).write_text(content) + + # Keep the rendered Kubernetes discovery config intact. Supply local + # service-account fixtures, but never contact a real Kubernetes cluster. + serviceaccount = root / "serviceaccount" + serviceaccount.mkdir() + (serviceaccount / "token").write_text("smoke-test-only\n") + (serviceaccount / "namespace").write_text("coder-observability\n") + shutil.copyfile(ssl.get_default_verify_paths().cafile, serviceaccount / "ca.crt") + mounts = [ + "-v", f"{config}:/etc/config:ro", + "-v", f"{serviceaccount}:/var/run/secrets/kubernetes.io/serviceaccount:ro", + ] + # Use the same image that will run in the cluster, including its promtool. + # Validate the real alert rules too, not just the config's YAML syntax. + run("docker", "run", "--rm", *mounts, "--entrypoint", "/bin/promtool", + image, "check", "config", "/etc/config/prometheus.yml") + + container = output( + "docker", "run", "-d", *mounts, "--tmpfs", "/data:rw,mode=1777", + "-p", "127.0.0.1::9090", + "-e", "KUBERNETES_SERVICE_HOST=127.0.0.1", + "-e", "KUBERNETES_SERVICE_PORT=9", + image, *server["args"], + ) + try: + address = output("docker", "port", container, "9090/tcp") + base_url = f"http://{address}" + deadline = time.monotonic() + 30 + while True: + try: + with urllib.request.urlopen(f"{base_url}/-/ready", timeout=1) as response: + if response.status == 200: + break + except (urllib.error.URLError, TimeoutError, ConnectionError): + pass + if output("docker", "inspect", "-f", "{{.State.Running}}", container) != "true": + raise RuntimeError("Prometheus exited before becoming ready") + if time.monotonic() >= deadline: + raise RuntimeError("Prometheus did not become ready within 30 seconds") + time.sleep(0.5) + + with urllib.request.urlopen(f"{base_url}/api/v1/targets", timeout=5) as response: + targets = json.load(response)["data"] + if targets["activeTargets"] or targets["droppedTargets"]: + raise RuntimeError(f"Expected scraping to be left to the collector: {targets}") + + # Exercise ingestion and queryability, rather than checking specific + # flags or just testing whether the HTTP endpoint exists. + run( + "docker", "exec", "-i", container, "/bin/promtool", "push", "metrics", + "--timeout=5s", "http://127.0.0.1:9090", + input="# TYPE observability_smoke_test gauge\nobservability_smoke_test 42\n", + ) + with urllib.request.urlopen( + f"{base_url}/api/v1/query?query=observability_smoke_test", timeout=5, + ) as response: + result = json.load(response)["data"]["result"] + if len(result) != 1 or float(result[0]["value"][1]) != 42: + raise RuntimeError(f"Remote-written metric was not queryable: {result}") + print("PASS: config and rules valid; server ready; no scrape targets; metric written and queried") + except Exception: + subprocess.run(["docker", "logs", container], check=False) + raise + finally: + run("docker", "rm", "-f", container, stdout=subprocess.DEVNULL) + + +if __name__ == "__main__": + check(sys.argv[1] if len(sys.argv) > 1 else "coder-observability") From da5a0b44a30ebce9f1c09802c261396a76cbb2fd Mon Sep 17 00:00:00 2001 From: Danny Kopping Date: Wed, 7 Oct 2026 11:51:04 +0200 Subject: [PATCH 7/8] ci: validate Prometheus config with a readiness check --- .github/workflows/lint.yaml | 5 +- .github/workflows/nightly-build.yaml | 2 +- Makefile | 7 +- TESTING_KIND.md | 13 ++- scripts/publish.sh | 2 +- scripts/test-prometheus.py | 132 --------------------------- scripts/test-prometheus.sh | 50 ++++++++++ 7 files changed, 65 insertions(+), 146 deletions(-) delete mode 100644 scripts/test-prometheus.py create mode 100755 scripts/test-prometheus.sh diff --git a/.github/workflows/lint.yaml b/.github/workflows/lint.yaml index 144a5cd..de64355 100644 --- a/.github/workflows/lint.yaml +++ b/.github/workflows/lint.yaml @@ -36,4 +36,7 @@ jobs: sudo chmod +x /usr/bin/yq - name: Lint Helm chart and rules - run: make lint \ No newline at end of file + run: make lint + + - name: Validate Prometheus config correctness + run: make test/prometheus diff --git a/.github/workflows/nightly-build.yaml b/.github/workflows/nightly-build.yaml index f79d6a8..0d59494 100644 --- a/.github/workflows/nightly-build.yaml +++ b/.github/workflows/nightly-build.yaml @@ -28,7 +28,7 @@ jobs: sudo wget https://github.com/mikefarah/yq/releases/download/v4.42.1/yq_linux_amd64 -O /usr/bin/yq &&\ sudo chmod +x /usr/bin/yq - - name: Build and test Prometheus + - name: Validate chart run: | { make build && make test/prometheus; } > output.log 2>&1 continue-on-error: false diff --git a/Makefile b/Makefile index e47e124..9b5b83a 100644 --- a/Makefile +++ b/Makefile @@ -10,7 +10,7 @@ SHELL := bash all: lint .PHONY: all -lint: build lint/helm test/prometheus lint/rules readme +lint: build lint/helm lint/rules readme ./scripts/check-unstaged.sh .PHONY: lint @@ -25,9 +25,8 @@ build: ./scripts/compile.sh .PHONY: build -# Requires chart dependencies (make build), Docker, Helm, yq, and Python 3. -test/prometheus: - python3 ./scripts/test-prometheus.py +test/prometheus: build + ./scripts/test-prometheus.sh .PHONY: test/prometheus lint/rules: lint/helm/prometheus-rules diff --git a/TESTING_KIND.md b/TESTING_KIND.md index 9ac9b77..5dfb0b1 100644 --- a/TESTING_KIND.md +++ b/TESTING_KIND.md @@ -9,18 +9,17 @@ helm dependency build coder-observability make test/prometheus ``` -This requires Docker, Helm, yq v4, and Python 3. It renders the chart, runs -`promtool check config` (including alert rules) from the rendered Prometheus image, -then starts that image with the rendered arguments. It checks readiness, zero -scrape targets (the collector handles scraping), and a remote-written metric that -can be queried back. Kubernetes discovery uses local service-account fixtures; -this does not test discovery or the collector end to end. +This requires Docker, Helm, yq v4, and curl. It compiles the chart's configuration +and alert rules, starts the rendered Prometheus image with its rendered arguments, +and waits for `/-/ready` to return success. Logs are printed and the container is +removed when the check finishes. Kubernetes discovery uses local service-account +fixtures; this check does not test discovery or metrics ingestion. The release script runs the same check against the packaged `.tgz` before uploading it. To test an existing package locally: ```bash -python3 scripts/test-prometheus.py path/to/coder-observability-VERSION.tgz +./scripts/test-prometheus.sh path/to/coder-observability-VERSION.tgz ``` ## Full cluster test diff --git a/scripts/publish.sh b/scripts/publish.sh index ca59fbf..278f0e0 100755 --- a/scripts/publish.sh +++ b/scripts/publish.sh @@ -4,7 +4,7 @@ set -euox pipefail version=$("$(dirname "${BASH_SOURCE[0]}")/version.sh") mkdir -p build/helm helm package coder-observability --version=${version} --dependency-update --destination build/helm -python3 "$(dirname "${BASH_SOURCE[0]}")/test-prometheus.py" "build/helm/coder-observability-${version}.tgz" +"$(dirname "${BASH_SOURCE[0]}")/test-prometheus.sh" "build/helm/coder-observability-${version}.tgz" gsutil cp gs://helm.coder.com/observability/index.yaml build/helm/index.yaml helm repo index build/helm --url https://helm.coder.com/observability --merge build/helm/index.yaml gsutil -h "Cache-Control:no-cache,max-age=0" cp build/helm/index.yaml gs://helm.coder.com/observability/ diff --git a/scripts/test-prometheus.py b/scripts/test-prometheus.py deleted file mode 100644 index b8115be..0000000 --- a/scripts/test-prometheus.py +++ /dev/null @@ -1,132 +0,0 @@ -#!/usr/bin/env python3 -"""Smoke-test the default chart, or a packaged chart, with its rendered image.""" - -import json -import pathlib -import shutil -import ssl -import subprocess -import sys -import tempfile -import time -import urllib.error -import urllib.request - - -def run(*args, **kwargs): - return subprocess.run(args, check=True, text=True, **kwargs) - - -def output(*args, **kwargs): - try: - return run(*args, capture_output=True, **kwargs).stdout.strip() - except subprocess.CalledProcessError as error: - print(error.stdout + error.stderr, file=sys.stderr) - raise - - -def check(chart): - with tempfile.TemporaryDirectory(prefix="prometheus-smoke-") as tmp: - root = pathlib.Path(tmp) - manifest = output( - "helm", "template", "coder-observability", chart, - "--namespace", "coder-observability", - ) - # Only parse the outer manifests. Re-serializing prometheus.yml here - # could silently remove the duplicate YAML keys we need promtool to catch. - resources = json.loads(output( - "yq", "eval-all", "-o=json", "[.]", "-", input=manifest, - )) - configmaps = { - resource["metadata"]["name"]: resource["data"] - for resource in resources if resource and resource["kind"] == "ConfigMap" - } - pod = next( - resource["spec"]["template"]["spec"] for resource in resources - if resource and resource["kind"] == "StatefulSet" - and resource["metadata"]["name"] == "prometheus" - ) - server = next(c for c in pod["containers"] if c["name"] == "prometheus-server") - image = server["image"] - print(f"Testing {chart} with {image}", flush=True) - - config = root / "config" - alerts = config / "alerts" - config.mkdir() - alerts.mkdir() - for name, content in configmaps["prometheus"].items(): - # The chart's extra alerts mount shadows this legacy ConfigMap key. - if name == "alerts": - continue - (config / name).write_text(content) - for name, content in configmaps["coder-metrics-alerts"].items(): - (alerts / name).write_text(content) - - # Keep the rendered Kubernetes discovery config intact. Supply local - # service-account fixtures, but never contact a real Kubernetes cluster. - serviceaccount = root / "serviceaccount" - serviceaccount.mkdir() - (serviceaccount / "token").write_text("smoke-test-only\n") - (serviceaccount / "namespace").write_text("coder-observability\n") - shutil.copyfile(ssl.get_default_verify_paths().cafile, serviceaccount / "ca.crt") - mounts = [ - "-v", f"{config}:/etc/config:ro", - "-v", f"{serviceaccount}:/var/run/secrets/kubernetes.io/serviceaccount:ro", - ] - # Use the same image that will run in the cluster, including its promtool. - # Validate the real alert rules too, not just the config's YAML syntax. - run("docker", "run", "--rm", *mounts, "--entrypoint", "/bin/promtool", - image, "check", "config", "/etc/config/prometheus.yml") - - container = output( - "docker", "run", "-d", *mounts, "--tmpfs", "/data:rw,mode=1777", - "-p", "127.0.0.1::9090", - "-e", "KUBERNETES_SERVICE_HOST=127.0.0.1", - "-e", "KUBERNETES_SERVICE_PORT=9", - image, *server["args"], - ) - try: - address = output("docker", "port", container, "9090/tcp") - base_url = f"http://{address}" - deadline = time.monotonic() + 30 - while True: - try: - with urllib.request.urlopen(f"{base_url}/-/ready", timeout=1) as response: - if response.status == 200: - break - except (urllib.error.URLError, TimeoutError, ConnectionError): - pass - if output("docker", "inspect", "-f", "{{.State.Running}}", container) != "true": - raise RuntimeError("Prometheus exited before becoming ready") - if time.monotonic() >= deadline: - raise RuntimeError("Prometheus did not become ready within 30 seconds") - time.sleep(0.5) - - with urllib.request.urlopen(f"{base_url}/api/v1/targets", timeout=5) as response: - targets = json.load(response)["data"] - if targets["activeTargets"] or targets["droppedTargets"]: - raise RuntimeError(f"Expected scraping to be left to the collector: {targets}") - - # Exercise ingestion and queryability, rather than checking specific - # flags or just testing whether the HTTP endpoint exists. - run( - "docker", "exec", "-i", container, "/bin/promtool", "push", "metrics", - "--timeout=5s", "http://127.0.0.1:9090", - input="# TYPE observability_smoke_test gauge\nobservability_smoke_test 42\n", - ) - with urllib.request.urlopen( - f"{base_url}/api/v1/query?query=observability_smoke_test", timeout=5, - ) as response: - result = json.load(response)["data"]["result"] - if len(result) != 1 or float(result[0]["value"][1]) != 42: - raise RuntimeError(f"Remote-written metric was not queryable: {result}") - print("PASS: config and rules valid; server ready; no scrape targets; metric written and queried") - except Exception: - subprocess.run(["docker", "logs", container], check=False) - raise - finally: - run("docker", "rm", "-f", container, stdout=subprocess.DEVNULL) - - -if __name__ == "__main__": - check(sys.argv[1] if len(sys.argv) > 1 else "coder-observability") diff --git a/scripts/test-prometheus.sh b/scripts/test-prometheus.sh new file mode 100755 index 0000000..8e76ede --- /dev/null +++ b/scripts/test-prometheus.sh @@ -0,0 +1,50 @@ +#!/usr/bin/env bash +set -euo pipefail + +temp_dir="$(mktemp -d)" +container="" +cleanup() { + if [[ -n "$container" ]]; then + docker logs "$container" + docker rm -f "$container" >/dev/null + fi + rm -rf "$temp_dir" +} +trap cleanup EXIT + +# Compile fresh manifests so the config, image, and arguments all come from the +# same chart. Accept a packaged chart too, for the pre-publish check. +helm template coder-observability "${1:-coder-observability}" \ + --namespace coder-observability > "$temp_dir/resources.yaml" +yq -e 'select(.kind == "StatefulSet" and .metadata.name == "prometheus") | + .spec.template.spec.containers[] | select(.name == "prometheus-server")' \ + "$temp_dir/resources.yaml" > "$temp_dir/server.yaml" +image="$(yq -er '.image' "$temp_dir/server.yaml")" +mapfile -t args < <(yq -r '.args[]' "$temp_dir/server.yaml") + +mkdir -p "$temp_dir/config/alerts" "$temp_dir/serviceaccount" +# Extract the embedded config as text, without parsing/re-serializing its YAML. +yq -er 'select(.kind == "ConfigMap" and .metadata.name == "prometheus") | + .data."prometheus.yml"' "$temp_dir/resources.yaml" > "$temp_dir/config/prometheus.yml" +yq -e 'select(.kind == "ConfigMap" and .metadata.name == "coder-metrics-alerts") | + .data' "$temp_dir/resources.yaml" > "$temp_dir/alerts.yaml" +for key in $(yq -r 'keys | .[]' "$temp_dir/alerts.yaml"); do + KEY="$key" yq -r '.[strenv(KEY)]' "$temp_dir/alerts.yaml" > "$temp_dir/config/alerts/$key" +done + +# The default config uses in-cluster Alertmanager discovery. Supply local +# service-account files and point discovery at loopback, not a real cluster. +printf 'smoke-test-only\n' > "$temp_dir/serviceaccount/token" +cp /etc/ssl/certs/ca-certificates.crt "$temp_dir/serviceaccount/ca.crt" +container="$(docker run -d \ + -v "$temp_dir/config:/etc/config:ro" \ + -v "$temp_dir/serviceaccount:/var/run/secrets/kubernetes.io/serviceaccount:ro" \ + --tmpfs /data:rw,mode=1777 \ + -p 127.0.0.1::9090 \ + -e KUBERNETES_SERVICE_HOST=127.0.0.1 -e KUBERNETES_SERVICE_PORT=9 \ + "$image" "${args[@]}")" + +# Readiness succeeds only after Prometheus has loaded its config and rules. +address="$(docker port "$container" 9090/tcp)" +curl --fail --silent --show-error --retry 30 --retry-all-errors \ + --retry-delay 1 --retry-max-time 60 --max-time 2 "http://$address/-/ready" From 8a125ab6d77ff22499e871cb6ca264483a68086e Mon Sep 17 00:00:00 2001 From: Danny Kopping Date: Wed, 7 Oct 2026 11:54:53 +0200 Subject: [PATCH 8/8] docs: revert testing guide changes --- TESTING_KIND.md | 26 +------------------------- 1 file changed, 1 insertion(+), 25 deletions(-) diff --git a/TESTING_KIND.md b/TESTING_KIND.md index 5dfb0b1..e1c9786 100644 --- a/TESTING_KIND.md +++ b/TESTING_KIND.md @@ -1,29 +1,5 @@ # Using kind to test observability -## Prometheus smoke test - -Before running a full cluster test, run the same Prometheus smoke test used by CI: - -```bash -helm dependency build coder-observability -make test/prometheus -``` - -This requires Docker, Helm, yq v4, and curl. It compiles the chart's configuration -and alert rules, starts the rendered Prometheus image with its rendered arguments, -and waits for `/-/ready` to return success. Logs are printed and the container is -removed when the check finishes. Kubernetes discovery uses local service-account -fixtures; this check does not test discovery or metrics ingestion. - -The release script runs the same check against the packaged `.tgz` before uploading -it. To test an existing package locally: - -```bash -./scripts/test-prometheus.sh path/to/coder-observability-VERSION.tgz -``` - -## Full cluster test -
If using nix @@ -152,4 +128,4 @@ To view the monitoring services, you can port forward their UI's. - Grafana: `kubectl -n coder-observability port-forward svc/grafana 3000:80` - Prometheus: `kubectl -n coder-observability port-forward svc/prometheus 3001:80` - Pyroscope: `kubectl port-forward svc/pyroscope 3002:4040` -- Grafana Agent: `kubectl port-forward svc/grafana-agent 3003:80` +- Grafana Agent: `kubectl port-forward svc/grafana-agent 3003:80` \ No newline at end of file