Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
58 changes: 52 additions & 6 deletions pkg/cli/admin/upgrade/status/alerts.go
Original file line number Diff line number Diff line change
Expand Up @@ -31,6 +31,7 @@ type AlertAnnotations struct {
Description string `json:"description,omitempty"`
Summary string `json:"summary,omitempty"`
Runbook string `json:"runbook_url,omitempty"`
Message string `json:"message,omitempty"`
}

type Alert struct {
Expand All @@ -57,27 +58,72 @@ func parseAlertDataToInsights(alertData AlertData, startedAt time.Time) []update
var updateInsights []updateInsight

for _, alert := range alerts {
if startedAt.After(alert.ActiveAt) && !allowedAlerts.Contains(alert.Labels.AlertName) {
var alertName string
if alertName = alert.Labels.AlertName; alertName == "" {
continue
}

var description string
startedDuringUpdate := startedAt.Before(alert.ActiveAt)
affectsUpdates := allowedAlerts.Contains(alertName)

if affectsUpdates {
if startedDuringUpdate {
description = "Alert known to affect updates started firing during the update."
} else {
description = "Alert known to affect updates has been firing since before the update started."
}
} else if startedDuringUpdate {
description = "Alert started firing during the update."
} else {
// Do not show alerts that were firing before the update started unless they are on the allowlist
continue
}

if alert.State == "pending" {
continue
}
var level impactLevel
if level = alertImpactLevel(alert.Labels.Severity); level < warningImpactLevel {
continue
}

var runbook string
if runbook = alert.Annotations.Runbook; runbook == "" {
runbook = "<alert does not have a runbook_url annotation>"
}

switch {
case alert.Annotations.Message != "" && alert.Annotations.Description != "":
description += " The alert description is: " + alert.Annotations.Description + " | " + alert.Annotations.Message
case alert.Annotations.Description != "":
description += " The alert description is: " + alert.Annotations.Description

Copy link
Copy Markdown
Member

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

In the deep past (e.g. openshift/cluster-version-operator#547), alerts used message instead of the split summary/description. No worries if you don't want to address that in this pull, and checking the alert fixture data, maybe there are no relevant alerts that still do things the old way, in which case no need to handle it at all.

Copy link
Copy Markdown
Member Author

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

Good catch! 4.16.0-rc.2 payload shows we still have a few manifests that use message so I added handling this in the code. Some of our alerts use all fields (message, description, summary) so I try to include them all, when present. Also added fixtures to exercise this case.

$ rg 'message: ' *prom*yaml
0000_90_cluster-authentication-operator_03_prometheusrule.yaml
17:            message: >-

0000_90_ingress-operator_03_prometheusrules.yaml
25:          message: "HAProxy reloads are failing on {{ $labels.pod }}. Router is not respecting recently created or modified routes"
34:          message: "HAProxy metrics are reporting that HAProxy is down on pod {{ $labels.namespace }} / {{ $labels.pod }}"
43:          message: |
54:          message: |
78:            message: "Ingress {{ $labels.namespace }}/{{ $labels.name }} is missing the IngressClassName for 1 day."
87:            message: "Route {{ $labels.namespace }}/{{ $labels.name }} is owned by an unmanaged Ingress."

0000_50_cluster-storage-operator_12_prometheusrules.yaml
28:          message: "StorageClass count check is failing (there should not be more than one default StorageClass)"
55:            Events of the Pods should contain exact error message: "oc describe pod -n <pod namespace> <pod name>".

case alert.Annotations.Message != "":
description += " The alert description is: " + alert.Annotations.Message
default:
description += " The alert has no description."
}

var summary string
if summary = alert.Annotations.Summary; summary == "" {
summary = alertName
}

updateInsights = append(updateInsights, updateInsight{
startedAt: alert.ActiveAt,
impact: updateInsightImpact{
level: alertImpactLevel(alert.Labels.Severity),
level: level,
impactType: unknownImpactType,
summary: "Alert: " + alert.Annotations.Summary,
description: alert.Annotations.Description,
summary: "Alert is firing: " + summary,
description: description,
},
remediation: updateInsightRemediation{reference: alert.Annotations.Runbook},
remediation: updateInsightRemediation{reference: runbook},
scope: updateInsightScope{
scopeType: scopeTypeCluster,
resources: []scopeResource{{
kind: scopeGroupKind{group: configv1.GroupName, kind: "Alert"},
namespace: alert.Labels.Namespace,
name: alert.Labels.AlertName,
name: alertName,
}},
},
})
Expand Down
50 changes: 34 additions & 16 deletions pkg/cli/admin/upgrade/status/alerts_test.go
Original file line number Diff line number Diff line change
@@ -1,10 +1,10 @@
package status

import (
"reflect"
"testing"
"time"

"github.com/google/go-cmp/cmp"
configv1 "github.com/openshift/api/config/v1"
)

Expand All @@ -15,15 +15,13 @@ func TestParseAlertDataToInsights(t *testing.T) {
tests := []struct {
name string
alertData AlertData
startedAt time.Time
expectedCount int
}{
{
name: "Empty Alerts",
alertData: AlertData{
Data: Data{Alerts: []Alert{}},
},
startedAt: now,
expectedCount: 0,
},
{
Expand All @@ -36,7 +34,6 @@ func TestParseAlertDataToInsights(t *testing.T) {
},
},
},
startedAt: now,
expectedCount: 1,
},
{
Expand All @@ -48,40 +45,59 @@ func TestParseAlertDataToInsights(t *testing.T) {
},
},
},
startedAt: now,
expectedCount: 0,
},
{
name: "Alert Active Before Start Time, Allowed",
alertData: AlertData{
Data: Data{
Alerts: []Alert{
{ActiveAt: now.Add(-20 * time.Minute), Labels: AlertLabels{Severity: "info", Namespace: "default", AlertName: "PodDisruptionBudgetAtLimit"}, Annotations: AlertAnnotations{Summary: "PodDisruptionBudgetAtLimit is at limit"}},
{ActiveAt: now.Add(-20 * time.Minute), Labels: AlertLabels{Severity: "info", Namespace: "default", AlertName: "AlertmanagerReceiversNotConfigured"}, Annotations: AlertAnnotations{Summary: "Receivers (notification integrations) are not configured on Alertmanager"}},
{ActiveAt: now.Add(-20 * time.Minute), Labels: AlertLabels{Severity: "warning", Namespace: "default", AlertName: "PodDisruptionBudgetAtLimit"}, Annotations: AlertAnnotations{Summary: "PodDisruptionBudgetAtLimit is at limit"}},
{ActiveAt: now.Add(-20 * time.Minute), Labels: AlertLabels{Severity: "warning", Namespace: "default", AlertName: "AlertmanagerReceiversNotConfigured"}, Annotations: AlertAnnotations{Summary: "Receivers (notification integrations) are not configured on Alertmanager"}},
},
},
},
startedAt: now,
expectedCount: 1,
},
{
name: "Alert Active Before Start Time, Not Allowed",
alertData: AlertData{
Data: Data{
Alerts: []Alert{
{ActiveAt: now.Add(-20 * time.Minute), Labels: AlertLabels{Severity: "info", Namespace: "default", AlertName: "AlertmanagerReceiversNotConfigured"}, Annotations: AlertAnnotations{Summary: "Receivers (notification integrations) are not configured on Alertmanager"}},
{ActiveAt: now.Add(-20 * time.Minute), Labels: AlertLabels{Severity: "warning", Namespace: "default", AlertName: "AlertmanagerReceiversNotConfigured"}, Annotations: AlertAnnotations{Summary: "Receivers (notification integrations) are not configured on Alertmanager"}},
},
},
},
expectedCount: 0,
},
{
name: "Info Alert Active After Start Time, Not Allowed",
alertData: AlertData{
Data: Data{
Alerts: []Alert{
{ActiveAt: now.Add(10 * time.Minute), Labels: AlertLabels{Severity: "info", Namespace: "default", AlertName: "NodeDown"}, Annotations: AlertAnnotations{Summary: "Node is down"}},
},
},
},
expectedCount: 0,
},
{
name: "Info Alert Active After Start Time, Not Allowed",
alertData: AlertData{
Data: Data{
Alerts: []Alert{
{ActiveAt: now.Add(10 * time.Minute), Labels: AlertLabels{Severity: "info", Namespace: "default", AlertName: "NodeDown"}, Annotations: AlertAnnotations{Summary: "Node is down"}},
},
},
},
startedAt: now,
expectedCount: 0,
},
}

// Execute test cases
for _, tt := range tests {
t.Run(tt.name, func(t *testing.T) {
insights := parseAlertDataToInsights(tt.alertData, tt.startedAt)
insights := parseAlertDataToInsights(tt.alertData, now)
if got := len(insights); got != tt.expectedCount {
t.Errorf("parseAlertDataToInsights() = %v, want %v", got, tt.expectedCount)
}
Expand Down Expand Up @@ -112,10 +128,12 @@ func TestParseAlertDataToInsightsWithData(t *testing.T) {
{
startedAt: now.Add(10 * time.Minute),
impact: updateInsightImpact{
level: alertImpactLevel("critical"),
impactType: unknownImpactType,
summary: "Alert: Node is down",
level: alertImpactLevel("critical"),
impactType: unknownImpactType,
summary: "Alert is firing: Node is down",
description: "Alert started firing during the update. The alert has no description.",
},
remediation: updateInsightRemediation{reference: "<alert does not have a runbook_url annotation>"},
scope: updateInsightScope{
scopeType: scopeTypeCluster,
resources: []scopeResource{
Expand Down Expand Up @@ -147,8 +165,8 @@ func TestParseAlertDataToInsightsWithData(t *testing.T) {
for _, tt := range tests {
t.Run(tt.name, func(t *testing.T) {
insights := parseAlertDataToInsights(tt.alertData, tt.startedAt)
if !reflect.DeepEqual(insights, tt.expectedInsights) {
t.Errorf("parseAlertDataToInsights() got %#v, want %#v", insights, tt.expectedInsights)
if diff := cmp.Diff(tt.expectedInsights, insights, allowUnexportedInsightStructs); diff != "" {
t.Errorf("parseAlertDataToInsights() differs from expected:\n%s", diff)
}
})
}
Expand Down
Original file line number Diff line number Diff line change
Expand Up @@ -74,7 +74,7 @@ Message: Cluster Operator control-plane-machine-set is unavailable (UnavailableR
Description: Missing 1 available replica(s)

Message: Cluster Version version is failing to proceed with the update (ClusterOperatorsDegraded)
Since: 0s
Since: now
Level: Warning
Impact: Update Stalled
Reference: https://github.com/openshift/runbooks/blob/master/alerts/cluster-monitoring-operator/ClusterOperatorDegraded.md
Expand Down
Original file line number Diff line number Diff line change
Expand Up @@ -32,6 +32,6 @@ SINCE LEVEL IMPACT MESSAGE
58m18s Error API Availability Cluster Operator kube-scheduler is degraded (NodeController_MasterNodesReady)
58m38s Error API Availability Cluster Operator etcd is degraded (EtcdEndpoints_ErrorUpdatingEtcdEndpoints::EtcdMembers_UnhealthyMembers::NodeController_MasterNodesReady)
1h0m17s Error API Availability Cluster Operator control-plane-machine-set is unavailable (UnavailableReplicas)
0s Warning Update Stalled Cluster Version version is failing to proceed with the update (ClusterOperatorsDegraded)
now Warning Update Stalled Cluster Version version is failing to proceed with the update (ClusterOperatorsDegraded)

Run with --details=health for additional description and links to related online documentation
Original file line number Diff line number Diff line change
Expand Up @@ -178,7 +178,7 @@
"severity": "warning"
},
"annotations": {
"description": "PodDisruptionBudgetAtLimit. in namespace <>",
"description": "PodDisruptionBudgetAtLimit in namespace <>",
"runbook_url": "https://<mock_data>/PDB.md",
"summary": "PodDisruptionBudgetAtLimit for pods <>"
},
Expand All @@ -187,6 +187,39 @@
"value": "6.708842592592593e-01",
"partialResponseStrategy": "WARN"
},
{
"labels": {
"alertname": "PodDisruptionBudgetAtLimit",
"controller": "alertmanager",
"namespace": "openshift-monitoring",
"severity": "warning"
},
"annotations": {
"summary": "PodDisruptionBudgetAtLimit for some reason does not have runbook and description"
},
"state": "firing",
"activeAt": "2023-11-23T15:39:33.014999722Z",
"value": "6.708842592592593e-01",
"partialResponseStrategy": "WARN"
},
{
"labels": {
"alertname": "PodDisruptionBudgetAtLimitWithMessage",
"controller": "alertmanager",
"namespace": "openshift-monitoring",
"severity": "warning"
},
"annotations": {
"summary": "This alert has a message, description, runbook and summary",
"description": "This alert has a description",
"runbook_url": "https://<mock_data>/runbook.md",
"message": "This alert has a message, too"
},
"state": "firing",
"activeAt": "2023-11-24T15:31:52.75038242Z",
"value": "6.708842592592593e-01",
"partialResponseStrategy": "WARN"
},
{
"labels": {
"alertname": "PrometheusOperatorWatchErrors",
Expand Down Expand Up @@ -221,4 +254,4 @@
}
]
}
}
}
Original file line number Diff line number Diff line change
Expand Up @@ -35,20 +35,38 @@ Message: Cluster Operator machine-config is unavailable (MachineConfigController
clusteroperators.config.openshift.io: machine-config
Description: Cluster not available for [{operator 4.14.0-rc.3}]: ControllerConfig.machineconfiguration.openshift.io "machine-config-controller" is invalid: [status.controllerCertificates[0].notAfter: Required value, status.controllerCertificates[0].notBefore: Required value, status.controllerCertificates[1].notAfter: Required value, status.controllerCertificates[1].notBefore: Required value, status.controllerCertificates[2].notAfter: Required value, status.controllerCertificates[2].notBefore: Required value, status.controllerCertificates[3].notAfter: Required value, status.controllerCertificates[3].notBefore: Required value, status.controllerCertificates[4].notAfter: Required value, status.controllerCertificates[4].notBefore: Required value, status.controllerCertificates[5].notAfter: Required value, status.controllerCertificates[5].notBefore: Required value, status.controllerCertificates[6].notAfter: Required value, status.controllerCertificates[6].notBefore: Required value, status.controllerCertificates[7].notAfter: Required value, status.controllerCertificates[7].notBefore: Required value, status.controllerCertificates[8].notAfter: Required value, status.controllerCertificates[8].notBefore: Required value, status.controllerCertificates[9].notAfter: Required value, status.controllerCertificates[9].notBefore: Required value, <nil>: Invalid value: "null": some validation rules were not checked because the object was invalid; correct the existing errors to complete validation]

Message: Alert: Pod has been in a non-ready state for more than 15 minutes.
Message: Alert is firing: Pod has been in a non-ready state for more than 15 minutes.
Since: 6m35s
Level: Warning
Impact: Unknown
Reference: https://github.com/openshift/runbooks/blob/master/alerts/cluster-monitoring-operator/KubePodNotReady.md
Resources:
alerts.config.openshift.io: openshift-kube-apiserver/KubePodNotReady
Description: Pod openshift-kube-apiserver/kube-apiserver-startup-monitor-ip-10-0-60-26.us-west-1.compute.internal has been in a non-ready state for longer than 15 minutes.
Description: Alert started firing during the update. The alert description is: Pod openshift-kube-apiserver/kube-apiserver-startup-monitor-ip-10-0-60-26.us-west-1.compute.internal has been in a non-ready state for longer than 15 minutes.

Message: Alert: PodDisruptionBudgetAtLimit for pods <>
Message: Alert is firing: This alert has a message, description, runbook and summary
Since: 16m35s
Level: Warning
Impact: Unknown
Reference: https://<mock_data>/runbook.md
Resources:
alerts.config.openshift.io: openshift-monitoring/PodDisruptionBudgetAtLimitWithMessage
Description: Alert started firing during the update. The alert description is: This alert has a description | This alert has a message, too

Message: Alert is firing: PodDisruptionBudgetAtLimit for pods <>
Since: 24h8m54s
Level: Warning
Impact: Unknown
Reference: https://<mock_data>/PDB.md
Resources:
alerts.config.openshift.io: openshift-monitoring/PodDisruptionBudgetAtLimit
Description: PodDisruptionBudgetAtLimit. in namespace <>
Description: Alert known to affect updates has been firing since before the update started. The alert description is: PodDisruptionBudgetAtLimit in namespace <>

Message: Alert is firing: PodDisruptionBudgetAtLimit for some reason does not have runbook and description
Since: 24h8m54s
Level: Warning
Impact: Unknown
Reference: <alert does not have a runbook_url annotation>
Resources:
alerts.config.openshift.io: openshift-monitoring/PodDisruptionBudgetAtLimit
Description: Alert known to affect updates has been firing since before the update started. The alert has no description.
Original file line number Diff line number Diff line change
Expand Up @@ -28,7 +28,9 @@ ip-10-0-99-40.us-east-2.compute.internal Outdated Pending 4.14.0-rc.3
= Update Health =
SINCE LEVEL IMPACT MESSAGE
20m24s Error API Availability Cluster Operator machine-config is unavailable (MachineConfigControllerFailed)
6m35s Warning Unknown Alert: Pod has been in a non-ready state for more than 15 minutes.
24h8m54s Warning Unknown Alert: PodDisruptionBudgetAtLimit for pods <>
6m35s Warning Unknown Alert is firing: Pod has been in a non-ready state for more than 15 minutes.
16m35s Warning Unknown Alert is firing: This alert has a message, description, runbook and summary
24h8m54s Warning Unknown Alert is firing: PodDisruptionBudgetAtLimit for pods <>
24h8m54s Warning Unknown Alert is firing: PodDisruptionBudgetAtLimit for some reason does not have runbook and description

Run with --details=health for additional description and links to related online documentation
Original file line number Diff line number Diff line change
Expand Up @@ -146,7 +146,7 @@ Message: Node build0-gstfj-ci-tests-worker-c-dcz9p is degraded
Description: failed to drain node: build0-gstfj-ci-tests-worker-c-dcz9p after 1 hour. Please see machine-config-controller logs for more information

Message: Cluster Version version is failing to proceed with the update (ClusterOperatorDegraded)
Since: 0s
Since: now
Level: Warning
Impact: Update Stalled
Reference: https://github.com/openshift/runbooks/blob/master/alerts/cluster-monitoring-operator/ClusterOperatorDegraded.md
Expand Down
Original file line number Diff line number Diff line change
Expand Up @@ -44,7 +44,7 @@ SINCE LEVEL IMPACT MESSAGE
- Error Update Stalled Node build0-gstfj-ci-tests-worker-b-jv5bg is degraded
- Error Update Stalled Node build0-gstfj-ci-tests-worker-b-kj6gk is degraded
- Error Update Stalled Node build0-gstfj-ci-tests-worker-c-dcz9p is degraded
0s Warning Update Stalled Cluster Version version is failing to proceed with the update (ClusterOperatorDegraded)
now Warning Update Stalled Cluster Version version is failing to proceed with the update (ClusterOperatorDegraded)
- Warning Update Speed Node build0-gstfj-ci-prowjobs-worker-d-ddnxd is unavailable
- Warning Update Speed Node build0-gstfj-ci-tests-worker-c-jq5rk is unavailable

Expand Down
4 changes: 2 additions & 2 deletions pkg/cli/admin/upgrade/status/health.go
Original file line number Diff line number Diff line change
Expand Up @@ -172,9 +172,9 @@ func assessUpdateInsights(insights []updateInsight, upgradingFor time.Duration,
func shortDuration(d time.Duration) string {
orig := d.String()
switch {
case orig == "0h0m0s":
case orig == "0h0m0s" || orig == "0s":
return "now"
case strings.HasSuffix(orig, "0m0s"):
case strings.HasSuffix(orig, "h0m0s"):
return orig[:len(orig)-4]
case strings.HasSuffix(orig, "m0s"):
return orig[:len(orig)-2]
Expand Down
Loading