diff --git a/.github/workflows/upstream-sync.yml b/.github/workflows/upstream-sync.yml
deleted file mode 100644
index 92e5c1f548..0000000000
--- a/.github/workflows/upstream-sync.yml
+++ /dev/null
@@ -1,38 +0,0 @@
----
-name: Upstream Sync
-'on':
- schedule:
- - cron: "15 8 * * 1"
- workflow_dispatch:
-permissions:
- contents: write
- pull-requests: write
-jobs:
- synchronise-2023-1:
- if: github.repository == 'stackhpc/stackhpc-kayobe-config'
- name: Synchronise 2023.1
- uses: stackhpc/.github/.github/workflows/upstream-sync.yml@main
- with:
- release_series: 2023.1
- upstream: openstack/kayobe-config
- synchronise-2024-1:
- if: github.repository == 'stackhpc/stackhpc-kayobe-config'
- name: Synchronise 2024.1
- uses: stackhpc/.github/.github/workflows/upstream-sync.yml@main
- with:
- release_series: 2024.1
- upstream: openstack/kayobe-config
- synchronise-2025-1:
- if: github.repository == 'stackhpc/stackhpc-kayobe-config'
- name: Synchronise 2025.1
- uses: stackhpc/.github/.github/workflows/upstream-sync.yml@main
- with:
- release_series: 2025.1
- upstream: openstack/kayobe-config
- synchronise-master:
- if: github.repository == 'stackhpc/stackhpc-kayobe-config'
- name: Synchronise master
- uses: stackhpc/.github/.github/workflows/upstream-sync.yml@main
- with:
- release_series: master
- upstream: openstack/kayobe-config
diff --git a/etc/kayobe/hooks/overcloud-host-package-update/post.d/20-reset-bls-entries.yml b/etc/kayobe/hooks/overcloud-host-package-update/post.d/20-reset-bls-entries.yml
new file mode 120000
index 0000000000..8a594a37e1
--- /dev/null
+++ b/etc/kayobe/hooks/overcloud-host-package-update/post.d/20-reset-bls-entries.yml
@@ -0,0 +1 @@
+../../../ansible/maintenance/reset-bls-entries.yml
\ No newline at end of file
diff --git a/etc/kayobe/kolla/config/prometheus/etcd.rules b/etc/kayobe/kolla/config/prometheus/etcd.rules
new file mode 100644
index 0000000000..5946ae2fef
--- /dev/null
+++ b/etc/kayobe/kolla/config/prometheus/etcd.rules
@@ -0,0 +1,186 @@
+# Source: https://github.com/samber/awesome-prometheus-alerts
+
+{% raw %}
+groups:
+- name: EmbeddedExporter
+ rules:
+ - alert: EtcdInsufficientMembers
+ expr: count(etcd_server_id) % 2 == 0
+ for: 0m
+ labels:
+ severity: critical
+ annotations:
+ summary: Etcd insufficient Members (instance {{ $labels.instance }})
+ description: "Etcd cluster should have an odd number of members\n VALUE = {{ $value }}\n LABELS = {{ $labels }}"
+
+ - alert: EtcdNoLeader
+ expr: etcd_server_has_leader == 0
+ for: 0m
+ labels:
+ severity: critical
+ annotations:
+ summary: Etcd no Leader (instance {{ $labels.instance }})
+ description: "Etcd cluster have no leader\n VALUE = {{ $value }}\n LABELS = {{ $labels }}"
+
+ - alert: EtcdHighNumberOfLeaderChanges
+ expr: increase(etcd_server_leader_changes_seen_total[10m]) > 2
+ for: 0m
+ labels:
+ severity: warning
+ annotations:
+ summary: Etcd high number of leader changes (instance {{ $labels.instance }})
+ description: "Etcd leader changed {{ $value }} times during 10 minutes\n VALUE = {{ $value }}\n LABELS = {{ $labels }}"
+
+ # Counts server-side failures only. NotFound, AlreadyExists and Cancelled are normal gRPC
+ # responses and are deliberately excluded from the error rate.
+ - alert: EtcdHighNumberOfFailedGRPCRequestsWarning
+ expr: sum(rate(grpc_server_handled_total{grpc_code=~"Unknown|FailedPrecondition|ResourceExhausted|Internal|Unavailable|DataLoss|DeadlineExceeded"}[1m])) BY (grpc_service, grpc_method) / sum(rate(grpc_server_handled_total[1m])) BY (grpc_service, grpc_method) > 0.01 and sum(rate(grpc_server_handled_total[1m])) BY (grpc_service, grpc_method) > 0
+ for: 2m
+ labels:
+ severity: warning
+ annotations:
+ summary: Etcd high number of failed GRPC requests warning (instance {{ $labels.instance }})
+ description: "More than 1% GRPC request failure detected in Etcd\n VALUE = {{ $value }}\n LABELS = {{ $labels }}"
+
+ # Counts server-side failures only. NotFound, AlreadyExists and Cancelled are normal gRPC
+ # responses and are deliberately excluded from the error rate.
+ - alert: EtcdHighNumberOfFailedGRPCRequestsCritical
+ expr: sum(rate(grpc_server_handled_total{grpc_code=~"Unknown|FailedPrecondition|ResourceExhausted|Internal|Unavailable|DataLoss|DeadlineExceeded"}[1m])) BY (grpc_service, grpc_method) / sum(rate(grpc_server_handled_total[1m])) BY (grpc_service, grpc_method) > 0.05 and sum(rate(grpc_server_handled_total[1m])) BY (grpc_service, grpc_method) > 0
+ for: 2m
+ labels:
+ severity: critical
+ annotations:
+ summary: Etcd high number of failed GRPC requests critical (instance {{ $labels.instance }})
+ description: "More than 5% GRPC request failure detected in Etcd\n VALUE = {{ $value }}\n LABELS = {{ $labels }}"
+
+ # Excludes the Defragment method, which is inherently slow and would otherwise trigger
+ # false positives during a manual or automatic etcdctl defrag.
+ - alert: EtcdGRPCRequestsSlow
+ expr: histogram_quantile(0.99, sum(rate(grpc_server_handling_seconds_bucket{grpc_method!="Defragment", grpc_type="unary"}[1m])) by (grpc_service, grpc_method, le)) > 0.15
+ for: 2m
+ labels:
+ severity: warning
+ annotations:
+ summary: Etcd GRPC requests slow (instance {{ $labels.instance }})
+ description: "GRPC requests slowing down, 99th percentile is over 0.15s\n VALUE = {{ $value }}\n LABELS = {{ $labels }}"
+
+ # These etcd_http_* metrics are from the etcd v2 API and do not exist in etcd 3.x. Remove these rules if running etcd 3.x.
+ - alert: EtcdHighNumberOfFailedHTTPRequestsWarning
+ expr: sum(rate(etcd_http_failed_total[1m])) BY (method) / sum(rate(etcd_http_received_total[1m])) BY (method) > 0.01 and sum(rate(etcd_http_received_total[1m])) BY (method) > 0
+ for: 2m
+ labels:
+ severity: warning
+ annotations:
+ summary: Etcd high number of failed HTTP requests warning (instance {{ $labels.instance }})
+ description: "More than 1% HTTP failure detected in Etcd\n VALUE = {{ $value }}\n LABELS = {{ $labels }}"
+
+ # These etcd_http_* metrics are from the etcd v2 API and do not exist in etcd 3.x. Remove these rules if running etcd 3.x.
+ - alert: EtcdHighNumberOfFailedHTTPRequestsCritical
+ expr: sum(rate(etcd_http_failed_total[1m])) BY (method) / sum(rate(etcd_http_received_total[1m])) BY (method) > 0.05 and sum(rate(etcd_http_received_total[1m])) BY (method) > 0
+ for: 2m
+ labels:
+ severity: critical
+ annotations:
+ summary: Etcd high number of failed HTTP requests critical (instance {{ $labels.instance }})
+ description: "More than 5% HTTP failure detected in Etcd\n VALUE = {{ $value }}\n LABELS = {{ $labels }}"
+
+ # This etcd_http_* metric is from the etcd v2 API and does not exist in etcd 3.x. Remove this rule if running etcd 3.x.
+ - alert: EtcdHTTPRequestsSlow
+ expr: histogram_quantile(0.99, rate(etcd_http_successful_duration_seconds_bucket[1m])) > 0.15
+ for: 2m
+ labels:
+ severity: warning
+ annotations:
+ summary: Etcd HTTP requests slow (instance {{ $labels.instance }})
+ description: "HTTP requests slowing down, 99th percentile is over 0.15s\n VALUE = {{ $value }}\n LABELS = {{ $labels }}"
+
+ - alert: EtcdMemberCommunicationSlow
+ expr: histogram_quantile(0.99, sum(rate(etcd_network_peer_round_trip_time_seconds_bucket[5m])) by (instance, le)) > 0.15
+ for: 2m
+ labels:
+ severity: warning
+ annotations:
+ summary: Etcd member communication slow (instance {{ $labels.instance }})
+ description: "Etcd member communication slowing down, 99th percentile is over 0.15s\n VALUE = {{ $value }}\n LABELS = {{ $labels }}"
+
+ - alert: EtcdHighNumberOfFailedProposals
+ expr: increase(etcd_server_proposals_failed_total[1h]) > 5
+ for: 2m
+ labels:
+ severity: warning
+ annotations:
+ summary: Etcd high number of failed proposals (instance {{ $labels.instance }})
+ description: "Etcd server got {{ $value }} failed proposals in the past hour\n VALUE = {{ $value }}\n LABELS = {{ $labels }}"
+
+ - alert: EtcdHighFsyncDurations
+ expr: histogram_quantile(0.99, sum(rate(etcd_disk_wal_fsync_duration_seconds_bucket[5m])) by (instance, le)) > 0.5
+ for: 2m
+ labels:
+ severity: warning
+ annotations:
+ summary: Etcd high fsync durations (instance {{ $labels.instance }})
+ description: "Etcd WAL fsync duration increasing, 99th percentile is over 0.5s\n VALUE = {{ $value }}\n LABELS = {{ $labels }}"
+
+ - alert: EtcdHighCommitDurations
+ expr: histogram_quantile(0.99, sum(rate(etcd_disk_backend_commit_duration_seconds_bucket[5m])) by (instance, le)) > 0.25
+ for: 2m
+ labels:
+ severity: warning
+ annotations:
+ summary: Etcd high commit durations (instance {{ $labels.instance }})
+ description: "Etcd commit duration increasing, 99th percentile is over 0.25s\n VALUE = {{ $value }}\n LABELS = {{ $labels }}"
+
+ - alert: EtcdPeerCommunicationFailures
+ expr: sum(rate(etcd_network_peer_sent_failures_total[1m])) by (To) > 0.01
+ for: 2m
+ labels:
+ severity: warning
+ annotations:
+ summary: Etcd peer communication failures (instance {{ $labels.instance }})
+ description: "Etcd member is experiencing more than 0.01 failed peer sends per second to peer {{ $labels.To }}, which usually indicates the member is unreachable or network-partitioned.\n VALUE = {{ $value }}\n LABELS = {{ $labels }}"
+
+ - alert: EtcdInsufficientMembersForQuorum
+ expr: sum(up{job=~".*etcd.*"} == bool 1) without (instance) < ((count(up{job=~".*etcd.*"}) without (instance) + 1) / 2)
+ for: 3m
+ labels:
+ severity: critical
+ annotations:
+ summary: Etcd insufficient members for quorum (instance {{ $labels.instance }})
+ description: "Only {{ $value }} etcd members are reachable, fewer than required for quorum; the cluster will reject writes until quorum is restored.\n VALUE = {{ $value }}\n LABELS = {{ $labels }}"
+
+ - alert: EtcdDatabaseQuotaLowSpace
+ expr: (last_over_time(etcd_mvcc_db_total_size_in_bytes[5m]) / last_over_time(etcd_server_quota_backend_bytes[5m])) * 100 > 95
+ for: 10m
+ labels:
+ severity: critical
+ annotations:
+ summary: Etcd database quota low space (instance {{ $labels.instance }})
+ description: "Etcd database size is above 95% of the configured storage quota ({{ $value }}%); once the quota is reached, etcd stops accepting writes cluster-wide.\n VALUE = {{ $value }}\n LABELS = {{ $labels }}"
+
+ - alert: EtcdExcessiveDatabaseGrowth
+ expr: predict_linear(etcd_mvcc_db_total_size_in_bytes[4h], 4*60*60) > etcd_server_quota_backend_bytes
+ for: 10m
+ labels:
+ severity: warning
+ annotations:
+ summary: Etcd excessive database growth (instance {{ $labels.instance }})
+ description: "Based on the current write rate over the past 4 hours, the etcd database is predicted to exceed the configured storage quota within the next 4 hours.\n VALUE = {{ $value }}\n LABELS = {{ $labels }}"
+
+ - alert: EtcdDatabaseHighFragmentationRatio
+ expr: (last_over_time(etcd_mvcc_db_total_size_in_use_in_bytes[5m]) / last_over_time(etcd_mvcc_db_total_size_in_bytes[5m])) < 0.5 and etcd_mvcc_db_total_size_in_use_in_bytes > 104857600
+ for: 10m
+ labels:
+ severity: warning
+ annotations:
+ summary: Etcd database high fragmentation ratio (instance {{ $labels.instance }})
+ description: "Etcd database in-use size is below 50% of its allocated size ({{ $value | humanizePercentage }}), indicating fragmentation; run 'etcdctl defrag' to reclaim disk space.\n VALUE = {{ $value }}\n LABELS = {{ $labels }}"
+
+ - alert: EtcdHighFsyncDurationsCritical
+ expr: histogram_quantile(0.99, sum(rate(etcd_disk_wal_fsync_duration_seconds_bucket[5m])) by (instance, le)) > 1
+ for: 2m
+ labels:
+ severity: critical
+ annotations:
+ summary: Etcd high fsync durations critical (instance {{ $labels.instance }})
+ description: "Etcd WAL fsync duration is critically high, 99th percentile is over 1s, which can cause request timeouts and trigger leader elections.\n VALUE = {{ $value }}\n LABELS = {{ $labels }}"
+{% endraw %}
diff --git a/etc/kayobe/ofed.yml b/etc/kayobe/ofed.yml
index 6407446595..057c4797d8 100644
--- a/etc/kayobe/ofed.yml
+++ b/etc/kayobe/ofed.yml
@@ -19,8 +19,8 @@ stackhpc_pulp_rocky_10_doca_version: "{{ stackhpc_pulp_doca_version_matrix[doca_
stackhpc_doca_kernel_version_matrix:
"9.6": 5.14.0.570.21.1.el9.6
"9.7": 5.14.0.611.55.1.el9.7
- "9.8": 5.14.0.687.26.1.el9.8
- "10.2": 6.12.0.211.33.1.el10.2
+ "9.8": 5.14.0.687.30.1.el9.8
+ "10.2": 6.12.0.211.39.1.el10.2
###############################################################################
# Pulp configuration for DOCA OFED
diff --git a/releasenotes/notes/add-etcd-prometheus-alerts-f14710e37d97621f.yaml b/releasenotes/notes/add-etcd-prometheus-alerts-f14710e37d97621f.yaml
new file mode 100644
index 0000000000..54c034ce43
--- /dev/null
+++ b/releasenotes/notes/add-etcd-prometheus-alerts-f14710e37d97621f.yaml
@@ -0,0 +1,5 @@
+---
+features:
+ - |
+ Add alerts for etcd taken from `Awesome Prometheus
+ `__
diff --git a/releasenotes/notes/repo-bump-20260728-6cba5f6d572ebbfb.yaml b/releasenotes/notes/repo-bump-20260728-6cba5f6d572ebbfb.yaml
new file mode 100644
index 0000000000..e2bb0fd3bd
--- /dev/null
+++ b/releasenotes/notes/repo-bump-20260728-6cba5f6d572ebbfb.yaml
@@ -0,0 +1,15 @@
+---
+features:
+ - |
+ Updated OFED kernel modules have been built for the latest Rocky Linux 9.8
+ (``5.14.0.687.30.1.el9.8``) and 10.2 kernels (``6.12.0.211.39.1.el10.2``).
+security:
+ - |
+ The latest Rocky Linux 9.8 kernel (``5.14.0.687.30.1.el9.8``) is now
+ available, which addresses KVM vulnerabilities `CVE-2025-40026
+ `__ and
+ `CVE-2026-63807 `__.
+ - |
+ Updates Docker Engine to `release 29.6.2
+ `__, which addresses
+ multiple vulnerabilities.