Alerting: Add backend support for keep_firing_for (#100750)

What is this feature?

This PR introduces a new alert rule configuration option, keep_firing_for (Prometheus documentation).

keep_firing_for prevents alerts from resolving immediately after the alert condition returns to normal. Instead, they transition into a "Recovering" state and are not considered resolved by the Alertmanager. Once the recovery period ends (or after the next evaluation if it is bigger than keep_firing_for), the alert transitions to "Normal" if it doesn't start alerting again:

Before                                          

+----------+     +----------+                    
| Alerting |---->|  Normal  |                    
+----------+     +----------+                    

-----
After

+----------+      +------------+     +----------+
| Alerting |----->| Recovering |---->|  Normal  |
+----------+      +------------+     +----------+                                                 

Why do we need this feature?

This feature prevents flapping alerts by adding a recovery period. This helps avoid false resolutions caused by brief alert
This commit is contained in:
Alexander Akhmetov
2025-03-18 11:24:48 +01:00
committed by GitHub
parent 9491fa1895
commit 695ac91290
31 changed files with 1280 additions and 53 deletions
@@ -119,7 +119,7 @@ func PrepareAlertStatuses(manager state.AlertInstanceManager, opts AlertStatuses
startsAt := alertState.StartsAt
valString := ""
if alertState.State == eval.Alerting || alertState.State == eval.Pending {
if alertState.State == eval.Alerting || alertState.State == eval.Pending || alertState.State == eval.Recovering {
valString = FormatValues(alertState)
}
@@ -204,6 +204,8 @@ func getStatesFromQuery(v url.Values) ([]eval.State, error) {
// nolint:goconst
case "error":
states = append(states, eval.Error)
case "recovering":
states = append(states, eval.Recovering)
default:
return states, fmt.Errorf("unknown state '%s'", s)
}
@@ -499,6 +501,8 @@ func filterRules(ruleGroup *apimodels.RuleGroup, withStatesFast map[eval.State]s
state = util.Pointer(eval.Alerting)
case "pending":
state = util.Pointer(eval.Pending)
case "recovering":
state = util.Pointer(eval.Recovering)
}
if state != nil {
if _, ok := withStatesFast[*state]; ok {
@@ -541,11 +545,12 @@ func toRuleGroup(log log.Logger, manager state.AlertInstanceManager, sr StatusRe
}
alertingRule := apimodels.AlertingRule{
State: "inactive",
Name: rule.Title,
Query: ruleToQuery(log, rule),
Duration: rule.For.Seconds(),
Annotations: apimodels.LabelsFromMap(rule.Annotations),
State: "inactive",
Name: rule.Title,
Query: ruleToQuery(log, rule),
Duration: rule.For.Seconds(),
KeepFiringFor: rule.KeepFiringFor.Seconds(),
Annotations: apimodels.LabelsFromMap(rule.Annotations),
}
newRule := apimodels.Rule{
@@ -566,7 +571,7 @@ func toRuleGroup(log log.Logger, manager state.AlertInstanceManager, sr StatusRe
for _, alertState := range states {
activeAt := alertState.StartsAt
valString := ""
if alertState.State == eval.Alerting || alertState.State == eval.Pending {
if alertState.State == eval.Alerting || alertState.State == eval.Pending || alertState.State == eval.Recovering {
valString = FormatValues(alertState)
}
stateKey := strings.ToLower(alertState.State.String())
@@ -586,12 +591,19 @@ func toRuleGroup(log log.Logger, manager state.AlertInstanceManager, sr StatusRe
Value: valString,
}
// Set the state of the rule based on the state of its alerts.
// Only update the rule state with 'pending' or 'recovering' if the current state is 'inactive'.
// This prevents overwriting a higher-severity 'firing' state in the case of a rule with multiple alerts.
switch alertState.State {
case eval.Normal:
case eval.Pending:
if alertingRule.State == "inactive" {
alertingRule.State = "pending"
}
case eval.Recovering:
if alertingRule.State == "inactive" {
alertingRule.State = "recovering"
}
case eval.Alerting:
if alertingRule.ActiveAt == nil || alertingRule.ActiveAt.After(activeAt) {
alertingRule.ActiveAt = &activeAt