From c2c58091c59ea3754fe792e04a8b8eb7b9e12f95 Mon Sep 17 00:00:00 2001 From: Bartek Tofel Date: Tue, 29 Sep 2026 08:59:56 +0200 Subject: [PATCH] feat(grafana-alertcheck): rename outcomes and print plain-worded tables check's output now speaks plain words, in both the JSON vocabulary and the human tables: - outcomes: clean -> healthy, newly_bad -> new_failure, persistently_bad -> still_failing, flapping -> unstable, skipped -> paused, unobservable -> not_verified, and terminated_early.kind follows the same rename. A --min-observed deficit that no resolved rule explains is now not_counted instead of being blamed on a paused rule. - RESULTS and VIOLATIONS columns are spelled out (ALERT, VERDICT, BROKEN FOR, CHECKED EVERY, WINDOW COVERED, DETAILS; GRAFANA STATE, GRAFANA HEALTH, INSTANCES). INSTANCES is one word: the old "INSTANCE COUNT" header read as two columns, one of them empty under the count. - THRESHOLDS became LIMITS USED, with each limit named in plain words and explained by a legend under the table. The footer spells out the extra observation time, the evaluation wait and the clock difference from Grafana. The rename reaches the JSON output, so consumers of violations[].outcome, outcomes and terminated_early must move to the new vocabulary. --- grafana-alertcheck/.changeset/v0.1.3.md | 2 + grafana-alertcheck/cmd/table.go | 73 ++++++---- grafana-alertcheck/cmd/table_test.go | 79 ++++++----- grafana-alertcheck/docs/architecture.md | 4 +- .../docs/how-alerts-are-evaluated.md | 26 ++-- grafana-alertcheck/docs/reference/cli.md | 2 +- .../docs/reference/log-format.md | 4 +- grafana-alertcheck/internal/gate/check.go | 12 +- .../internal/gate/check_test.go | 48 +++---- grafana-alertcheck/internal/gate/classify.go | 84 ++++++------ .../internal/gate/classify_test.go | 127 +++++++++++------- .../internal/gate/coverage_test.go | 10 +- grafana-alertcheck/internal/gate/terminal.go | 18 +-- .../internal/gate/terminal_test.go | 41 ++++-- 14 files changed, 304 insertions(+), 226 deletions(-) create mode 100644 grafana-alertcheck/.changeset/v0.1.3.md diff --git a/grafana-alertcheck/.changeset/v0.1.3.md b/grafana-alertcheck/.changeset/v0.1.3.md new file mode 100644 index 000000000..66b3aa5fc --- /dev/null +++ b/grafana-alertcheck/.changeset/v0.1.3.md @@ -0,0 +1,2 @@ +- The `check` output now speaks plain words. Outcome values are renamed: `clean` → `healthy`, `newly_bad` → `new_failure`, `persistently_bad` → `still_failing`, `flapping` → `unstable`, `skipped` → `paused`, `unobservable` → `not_verified`, and the early-exit `terminated_early.kind` follows the same rename. A `--min-observed` deficit that no rule explains is now reported as `not_counted` instead of being blamed on a paused rule. This changes the `--output json` vocabulary and anything downstream of it, including the action's `outcomes` output. +- The human table dropped its internal column names. `RESULTS` is now `ALERT`/`VERDICT`/`BROKEN FOR`/`CHECKED EVERY`/`WINDOW COVERED`/`DETAILS`; `VIOLATIONS` uses `GRAFANA STATE`/`GRAFANA HEALTH` and a single-word `INSTANCES` column (the previous `INSTANCE COUNT` header read as two columns, one of them empty); `THRESHOLDS` became `LIMITS USED`, with limits named in plain words and explained by a legend under the table. The footer now spells out the extra observation time, the evaluation wait and the clock difference from Grafana. diff --git a/grafana-alertcheck/cmd/table.go b/grafana-alertcheck/cmd/table.go index 81d750042..085444b88 100644 --- a/grafana-alertcheck/cmd/table.go +++ b/grafana-alertcheck/cmd/table.go @@ -11,22 +11,29 @@ import ( "github.com/smartcontractkit/chainlink-testing-framework/grafana-alertcheck/internal/gate" ) +// limitsLegend explains each LIMITS USED column in one plain sentence, so the +// table needs no documentation lookup. +const limitsLegend = ` max gap without check — the longest gap between two checks we accept before we say the alert was not watched. + query failing for — how long Grafana may keep failing to run the alert's query before we stop trusting its state. + no evaluation for — how long Grafana may go without evaluating the alert before we stop trusting its state.` + // renderTable is the human table. It always writes to the writer it is given, // which the caller (runCheck) always points at stderr — stdout is reserved for // the machine-readable --output json. // -// Three titled tables, in order (the name column is RULE in all of them — one -// row is one resolved alert rule, never a firing instance): +// Three titled tables, in order (the ALERT column is one resolved alert rule, +// never a firing instance): +// +// 1. RESULTS, one line per rule: verdict, time broken, check cadence, +// whether the window was observed, and any notes; +// 2. VIOLATIONS, one line per distinct violation, with the raw Grafana state +// and health and the number of instances it stands for; +// 3. LIMITS USED, the coverage thresholds that answer "why" on exit 2. Each +// column is named in plain words and explained by the legend below it, so +// the table needs no documentation lookup. // -// 1. RESULTS, one line per rule: outcome, BadFor, pollEvery, proved-or-not -// with the largest gap; -// 2. VIOLATIONS, one line per distinct violation. -// 3. THRESHOLDS, the numbers that answer "why" on exit 2: each non-skipped -// rule's maxGap/healthGrace/evalStaleAfter, followed by the global -// transitionGrace and drainTimeout, and the largest measured clock skew -// alongside its own error bound (RTT/2) — SkewHardLimit is a separate, -// fixed input threshold and is reported next to it, never as if it were -// that bound. +// The global footer then reports the extra observation time, the drain limit +// and the largest measured clock difference, also in plain words. func renderTable(w io.Writer, res gate.Result) error { alertOf := make(map[string]string, len(res.Verdicts)) for _, v := range res.Verdicts { @@ -52,7 +59,7 @@ func renderTable(w io.Writer, res gate.Result) error { fmt.Fprintln(w, "RESULTS") tw := tabwriter.NewWriter(w, 0, 4, 2, ' ', 0) - fmt.Fprintln(tw, "RULE\tOUTCOME\tBADFOR\tPOLLEVERY\tPROVED\tNOTE") + fmt.Fprintln(tw, "ALERT\tVERDICT\tBROKEN FOR\tCHECKED EVERY\tWINDOW COVERED\tDETAILS") for _, v := range sortedVerdicts(res.Verdicts) { fmt.Fprintf(tw, "%s\t%s\t%s\t%s\t%s\t%s\n", v.Alert, v.Outcome, v.BadFor.Round(time.Second), v.PollEvery.Round(time.Second), @@ -65,7 +72,7 @@ func renderTable(w io.Writer, res gate.Result) error { if len(res.Violations) > 0 { fmt.Fprintln(w, "\nVIOLATIONS") vtw := tabwriter.NewWriter(w, 0, 4, 2, ' ', 0) - fmt.Fprintln(vtw, "RULE\tOUTCOME\tSTATE\tHEALTH\tINSTANCE COUNT\tNOTE") + fmt.Fprintln(vtw, "ALERT\tVERDICT\tGRAFANA STATE\tGRAFANA HEALTH\tINSTANCES\tDETAILS") for _, g := range groupedViolations(res.Violations) { fmt.Fprintf(vtw, "%s\t%s\t%s\t%s\t%s\t%s\n", alertLabel(g.v, alertOf), g.v.Outcome, g.v.State, g.v.Health, instanceCount(g), g.v.Note) } @@ -74,14 +81,13 @@ func renderTable(w io.Writer, res gate.Result) error { } } - // The per-rule thresholds answer "why" on exit 2: a table, not the prose - // "rule NAME: maxGap=... healthGrace=... evalStaleAfter=..." that repeated - // the rule name a fourth time. It is separated from the result above by a - // blank line. + // The per-rule limits answer "why" on exit 2. The columns are spelled out + // and explained by limitsLegend right below, so an operator does not have + // to look anything up. fmt.Fprintln(w) - fmt.Fprintln(w, "THRESHOLDS") + fmt.Fprintln(w, "LIMITS USED") ttw := tabwriter.NewWriter(w, 0, 4, 2, ' ', 0) - fmt.Fprintln(ttw, "RULE\tMAXGAP\tHEALTHGRACE\tEVALSTALEAFTER") + fmt.Fprintln(ttw, "ALERT\tMAX GAP WITHOUT CHECK\tQUERY FAILING FOR\tNO EVALUATION FOR") for _, uid := range sortedThresholdUIDs(res.Thresholds, alertOf) { t := res.Thresholds[uid] fmt.Fprintf(ttw, "%s\t%s\t%s\t%s\n", @@ -90,11 +96,17 @@ func renderTable(w io.Writer, res gate.Result) error { if err := ttw.Flush(); err != nil { return fmt.Errorf("render table: %w", err) } + fmt.Fprintln(w, limitsLegend) fmt.Fprintln(w) - fmt.Fprintf(w, "global: transitionGrace=%s (source: %s) drainTimeout=%s\n", - res.Global.TransitionGrace, res.Global.GraceSource, res.Global.DrainTimeout) - fmt.Fprintf(w, "largest measured clock skew: %s (bound ±%s, hard limit %s), grafana %s\n", + if res.Global.TransitionGrace > 0 { + fmt.Fprintf(w, "extra watching after your window: +%s — so an alert that only starts firing at the end is still caught (slowest: %s)\n", + res.Global.TransitionGrace, res.Global.GraceSource) + } else { + fmt.Fprintln(w, "extra watching after your window: none") + } + fmt.Fprintf(w, "max wait for all alerts to finish evaluating: %s\n", res.Global.DrainTimeout) + fmt.Fprintf(w, "clock difference from Grafana: %s, accurate to ±%s (checks fail above %s); Grafana %s\n", res.ClockSkew.Round(time.Millisecond), res.ClockSkewBound.Round(time.Millisecond), gate.SkewHardLimit, res.GrafanaVersion) // The verdict — the single number a terminal operator reads last — sits on @@ -104,8 +116,8 @@ func renderTable(w io.Writer, res gate.Result) error { return nil } -// violationsLabel colours the "violations: N" prefix of the footer: green for a -// clean run, red otherwise. The rest of the line is written uncoloured. +// violationsLabel colours the "violations: N" prefix of the footer: green when +// there are none, red otherwise. The rest of the line is written uncoloured. func violationsLabel(n int, enabled bool) string { s := fmt.Sprintf("violations: %d", n) if !enabled { @@ -117,10 +129,10 @@ func violationsLabel(n int, enabled bool) string { return ansiRed + s + ansiReset } -// provedLabel is the table's PROVED column: "yes" for a clean coverage -// proof, "no" with the reason and largest gap for an unobservable rule, and -// "-" for a rule decide never asked proveCoverage about at all (skipped — -// paused before the window opened). +// provedLabel is the table's WINDOW COVERED column: "yes" for a fully +// observed window, "no" with the reason and largest gap for a not-verified +// rule, and "-" for a rule decide never asked proveCoverage about at all +// (paused before the window opened). func provedLabel(cov gate.CoverageResult) string { if cov.Reason == "" && !cov.Unobservable && !cov.Proved { return "-" @@ -192,8 +204,11 @@ func sameRendered(a, b gate.Violation) bool { return violationSignature(a) == violationSignature(b) } +// instanceCount is the INSTANCES column: how many alert instances one grouped +// violation row stands for. Paused and not-counted rows stand for no instance +// at all, so they render "-". func instanceCount(g violationGroup) string { - if g.v.Outcome == gate.OutcomeSkipped { + if g.v.Outcome == gate.OutcomePaused || g.v.Outcome == gate.OutcomeNotCounted { return "-" } return strconv.Itoa(g.n) diff --git a/grafana-alertcheck/cmd/table_test.go b/grafana-alertcheck/cmd/table_test.go index b050f0d30..acf987423 100644 --- a/grafana-alertcheck/cmd/table_test.go +++ b/grafana-alertcheck/cmd/table_test.go @@ -11,8 +11,8 @@ import ( ) // The golden table test: a fixed Result renders a deterministic, ordered rule -// table, a violations section and a footer carrying the per-rule and global -// thresholds plus the skew and its bound — with no live Check involved. +// table, a violations section and a footer carrying the per-rule limits and +// globals in plain words — with no live Check involved. func TestRenderTable(t *testing.T) { gapAt := time.Date(2026, 1, 1, 12, 0, 0, 0, time.UTC) res := gate.Result{ @@ -20,15 +20,15 @@ func TestRenderTable(t *testing.T) { ClockSkew: 1500 * time.Millisecond, ClockSkewBound: 250 * time.Millisecond, Verdicts: []gate.RuleVerdict{ - {Alert: "Zebra Alert", RuleUID: "uid-z", Outcome: gate.OutcomeClean, PollEvery: 30 * time.Second}, - {Alert: "Ape Alert", RuleUID: "uid-a", Outcome: gate.OutcomeUnobservable, + {Alert: "Zebra Alert", RuleUID: "uid-z", Outcome: gate.OutcomeHealthy, PollEvery: 30 * time.Second}, + {Alert: "Ape Alert", RuleUID: "uid-a", Outcome: gate.OutcomeNotVerified, PollEvery: 30 * time.Second, Note: "gap of 5m0s starting at 2026-01-01T12:00:00Z exceeds maxGap 1m0s"}, - {Alert: "Paused Alert", RuleUID: "uid-p", Outcome: gate.OutcomeSkipped, + {Alert: "Paused Alert", RuleUID: "uid-p", Outcome: gate.OutcomePaused, Note: "paused before the window opened; counts against --min-observed unless --allow-paused is set"}, }, Violations: []gate.Violation{ - {Alert: "Ape Alert", RuleUID: "uid-a", Outcome: gate.OutcomeUnobservable, State: gate.StateFiring, Health: "error", Note: "unobservable"}, - {Alert: "Paused Alert", RuleUID: "uid-p", Outcome: gate.OutcomeSkipped, + {Alert: "Ape Alert", RuleUID: "uid-a", Outcome: gate.OutcomeNotVerified, State: gate.StateFiring, Health: "error", Note: "not verified"}, + {Alert: "Paused Alert", RuleUID: "uid-p", Outcome: gate.OutcomePaused, Note: "paused before the window opened; counts against --min-observed unless --allow-paused is set"}, }, Coverage: map[string]gate.CoverageResult{ @@ -50,42 +50,49 @@ func TestRenderTable(t *testing.T) { require.NoError(t, renderTable(&buf, res)) out := buf.String() - // Rule table: Ape sorts before Zebra sorts before... Paused is skipped and - // carries no coverage entry, so it renders "-" for PROVED. + // The rule table: Ape sorts before Zebra sorts before... Paused carries no + // coverage entry, so it renders "-" for WINDOW COVERED. + require.Contains(t, out, "ALERT") require.Contains(t, out, "Ape Alert") - require.Contains(t, out, "unobservable") + require.Contains(t, out, "not_verified") require.Contains(t, out, "heartbeat_gap") require.Contains(t, out, "largest gap 5m0s") require.Contains(t, out, "Zebra Alert") - require.Contains(t, out, "clean") + require.Contains(t, out, "healthy") + require.Contains(t, out, "WINDOW COVERED") // The violations section must show up even without --output json, and must - // carry the --allow-paused hint text verbatim. + // carry the --allow-paused hint text verbatim. INSTANCES is a single word + // so the count under it cannot read as a second, empty column. require.Contains(t, out, "VIOLATIONS") require.Contains(t, out, "--allow-paused") - require.Contains(t, out, "STATE") - require.Contains(t, out, "HEALTH") - require.Contains(t, out, "INSTANCE COUNT") + require.Contains(t, out, "GRAFANA STATE") + require.Contains(t, out, "GRAFANA HEALTH") + require.Contains(t, out, "INSTANCES") require.Contains(t, out, string(gate.StateFiring)) require.Contains(t, out, "error") - // The footer: per-rule thresholds are a table (RULE/MAXGAP/HEALTHGRACE/ - // EVALSTALEAFTER) rather than prose, followed by the global thresholds and - // the violations count with the skew and its own bound rather than the - // fixed hard limit. - require.Contains(t, out, "MAXGAP") - require.Contains(t, out, "HEALTHGRACE") - require.Contains(t, out, "EVALSTALEAFTER") - require.Contains(t, out, "global: transitionGrace=5m0s (source: Ape Alert (for=5m)) drainTimeout=2m0s") - require.Contains(t, out, "largest measured clock skew: 1.5s (bound ±250ms, hard limit 1m0s)") + // The limits table names each threshold in plain words and explains it + // right below, so an operator does not have to consult the docs. + require.Contains(t, out, "LIMITS USED") + require.Contains(t, out, "MAX GAP WITHOUT CHECK") + require.Contains(t, out, "QUERY FAILING FOR") + require.Contains(t, out, "NO EVALUATION FOR") + require.Contains(t, out, "the longest gap between two checks") + require.Contains(t, out, "without evaluating the alert") + + // The global footer in plain words. + require.Contains(t, out, "extra watching after your window: +5m0s") + require.Contains(t, out, "slowest: Ape Alert (for=5m)") + require.Contains(t, out, "max wait for all alerts to finish evaluating: 2m0s") + require.Contains(t, out, "clock difference from Grafana: 1.5s, accurate to ±250ms (checks fail above 1m0s); Grafana 13.1.0") require.Contains(t, out, "violations: 2") - require.Contains(t, out, "13.1.0") } // The "-" case: a rule decide never asked proveCoverage about (paused before // the window opened) has an empty CoverageResult and must not be reported as -// either proved or unobservable. -func TestProvedLabel_Skipped(t *testing.T) { +// either covered or not verified. +func TestProvedLabel_Paused(t *testing.T) { require.Equal(t, "-", provedLabel(gate.CoverageResult{})) } @@ -93,12 +100,12 @@ func TestProvedLabel_Skipped(t *testing.T) { // rendered signature, each with a count. func TestGroupedViolations(t *testing.T) { in := []gate.Violation{ - {Alert: "OCR2 Consensus failure", RuleUID: "uid-o", Outcome: gate.OutcomePersistentlyBad, State: gate.StateFiring, Health: "ok"}, - {Alert: "OCR2 Consensus failure", RuleUID: "uid-o", Outcome: gate.OutcomePersistentlyBad, State: gate.StateFiring, Health: "ok"}, - {Alert: "OCR2 Consensus failure", RuleUID: "uid-o", Outcome: gate.OutcomePersistentlyBad, State: gate.StateFiring, Health: "ok"}, - {Alert: "OCR2 Consensus failure", RuleUID: "uid-o", Outcome: gate.OutcomePersistentlyBad, State: gate.StateFiring, Health: "error"}, - {Alert: "Other Alert", RuleUID: "uid-p", Outcome: gate.OutcomeNewlyBad, State: gate.StateFiring, Health: "ok", Note: "x"}, - {Alert: "Other Alert", RuleUID: "uid-p", Outcome: gate.OutcomeNewlyBad, State: gate.StateFiring, Health: "ok", Note: "x"}, + {Alert: "OCR2 Consensus failure", RuleUID: "uid-o", Outcome: gate.OutcomeStillFailing, State: gate.StateFiring, Health: "ok"}, + {Alert: "OCR2 Consensus failure", RuleUID: "uid-o", Outcome: gate.OutcomeStillFailing, State: gate.StateFiring, Health: "ok"}, + {Alert: "OCR2 Consensus failure", RuleUID: "uid-o", Outcome: gate.OutcomeStillFailing, State: gate.StateFiring, Health: "ok"}, + {Alert: "OCR2 Consensus failure", RuleUID: "uid-o", Outcome: gate.OutcomeStillFailing, State: gate.StateFiring, Health: "error"}, + {Alert: "Other Alert", RuleUID: "uid-p", Outcome: gate.OutcomeNewFailure, State: gate.StateFiring, Health: "ok", Note: "x"}, + {Alert: "Other Alert", RuleUID: "uid-p", Outcome: gate.OutcomeNewFailure, State: gate.StateFiring, Health: "ok", Note: "x"}, } got := groupedViolations(in) @@ -108,7 +115,7 @@ func TestGroupedViolations(t *testing.T) { for _, g := range got { counts[g.v.Health+"|"+string(g.v.Outcome)] = g.n } - require.Equal(t, 3, counts["ok|"+string(gate.OutcomePersistentlyBad)]) - require.Equal(t, 1, counts["error|"+string(gate.OutcomePersistentlyBad)]) - require.Equal(t, 2, counts["ok|"+string(gate.OutcomeNewlyBad)]) + require.Equal(t, 3, counts["ok|"+string(gate.OutcomeStillFailing)]) + require.Equal(t, 1, counts["error|"+string(gate.OutcomeStillFailing)]) + require.Equal(t, 2, counts["ok|"+string(gate.OutcomeNewFailure)]) } diff --git a/grafana-alertcheck/docs/architecture.md b/grafana-alertcheck/docs/architecture.md index 9c8b2aa38..fa2971178 100644 --- a/grafana-alertcheck/docs/architecture.md +++ b/grafana-alertcheck/docs/architecture.md @@ -15,11 +15,11 @@ This page documents the invariants and seams a maintainer must not break. It exi The gate must fail if it cannot get an answer. Every rule below is a specific instance of that: - **An error is never a pass.** A pass is exactly `len(Violations) == 0 && err == nil`. Every error path leaves `err` non-nil, and the CLI maps that to exit `2` unconditionally. -- **Inability beats violation.** Any `unobservable` rule is exit `2`, even alongside a real violation found first. +- **Inability beats violation.** Any `not_verified` rule is exit `2`, even alongside a real violation found first. - **Absent never means normal.** An instance that leaves the bad set is looked up in the *same* response: present as `normal` → cleared; absent (or `MissingSeries`) → vanished (a discontinuity, not a recovery). - **Staleness is absolute.** `grafana_now − lastEvaluation` is compared against a threshold, never "did it increase since the last poll" — a delta check reports stale on ~half the polls of a healthy rule (we poll at half of `intervalSeconds` of each rule). - **`grafana_now` is the response `Date` header.** Never the runner clock, in any comparison against a Grafana timestamp. -- **An early exit can never be a pass.** `check` may stop collecting before `to + transitionGrace` (fail-fast), but only on a *monotone* terminal verdict: an inability that has already happened, or a post-`from` bad onset (which the full classifier would call `newly_bad`/`flapping`). The one outcome that forgives an observed bad state, `recovered`, is reserved for bad-at-`from`, so a preexisting condition is never terminal. `--no-fail-fast` removes the guard entirely. +- **An early exit can never be a pass.** `check` may stop collecting before `to + transitionGrace` (fail-fast), but only on a *monotone* terminal verdict: an inability that has already happened, or a post-`from` bad onset (which the full classifier would call `new_failure`/`unstable`). The one outcome that forgives an observed bad state, `recovered`, is reserved for bad-at-`from`, so a preexisting condition is never terminal. `--no-fail-fast` removes the guard entirely. - **No replay.** No run-id key, no artifact download, no state between attempts. A retry is a new piece of work and observation. ## The pure-function seam diff --git a/grafana-alertcheck/docs/how-alerts-are-evaluated.md b/grafana-alertcheck/docs/how-alerts-are-evaluated.md index 46fa2142d..58ac89c14 100644 --- a/grafana-alertcheck/docs/how-alerts-are-evaluated.md +++ b/grafana-alertcheck/docs/how-alerts-are-evaluated.md @@ -32,13 +32,15 @@ For each instance the gate builds a timeline of bad spans over `[from, to]`, the | Outcome | Shape | Exit | | ------- | ----- | ---- | -| `clean` | Good throughout, observed throughout | → 0 | -| `newly_bad` | Entered a bad state **inside** the window | → 1 | -| `persistently_bad` | Bad at `from`, still bad at `to` | → 1 | +| `healthy` | Good throughout, observed throughout | → 0 | +| `new_failure` | Entered a bad state **inside** the window | → 1 | +| `still_failing` | Bad at `from`, still bad at `to` | → 1 | | `recovered` | Bad at `from`, cleared before `to`, stayed clear | → 0 | -| `flapping` | Cleared, then became bad again | → 1 | -| `skipped` | Paused **before** the window opened | reported, not observable | -| `unobservable` | Coverage gap / sustained `health=error` / stale / absent | → 2 | +| `unstable` | Cleared, then became bad again | → 1 | +| `paused` | Paused **before** the window opened | counts against `--min-observed` unless `--allow-paused` | +| `not_verified` | The window could not be observed: a gap, sustained `health=error`, a stale evaluation, or an absent rule | → 2 | + +A `--min-observed` deficit that no rule explains is reported as `not_counted`: it is not a verdict on any alert. `recovered` has **no deadline** — an alert that clears at minute 58 of a 60-minute window still passes. The total bad time is reported as `BadFor`. @@ -57,11 +59,11 @@ When an instance leaves the bad set, the gate looks it up **in the same response - Present as `normal` → `cleared` (a real recovery). - Absent, or present as `normal (MissingSeries)` → `vanished` (a discontinuity, **not** a recovery). -A vanished instance that was bad stays `persistently_bad`. A metric that stops being emitted is not evidence of health — this is deliberate and can surprise users whose fix is to remove a metric rather than drive it to a good value. +A vanished instance that was bad stays `still_failing`. A metric that stops being emitted is not evidence of health — this is deliberate and can surprise users whose fix is to remove a metric rather than drive it to a good value. ## Coverage proof -Before classifying, `check` must **prove** continuous coverage of `[from, to]` for each alert. Nine checks run; any failure makes the rule `unobservable`: +Before classifying, `check` must **prove** continuous coverage of `[from, to]` for each alert. Nine checks run; any failure makes the rule `not_verified`: 1. **Sentinel** — a clean recorder stop, timestamped at or after `to + transitionGrace`. A recorder that died mid-window looks exactly like a coverage gap and is one. 2. **`from` bounds** — `from` earlier than the recording start is unprovable. @@ -69,20 +71,20 @@ Before classifying, `check` must **prove** continuous coverage of `[from, to]` f 4. **`health=error`** — a contiguous run longer than `healthGrace` consumes coverage; a short blip is a note. 5. **`health=nodata`** — a note, never fatal (unless `--nodata-is-unobservable`). 6. **Liveness** — `grafana_now − lastEvaluation` must not exceed `evalStaleAfter`. This is an **absolute** check, never a "did it increase since the last poll" delta. -7. **In-window pause** — a poll reporting `isPaused` mid-window is `unobservable` (the primary pause detector). +7. **In-window pause** — a poll reporting `isPaused` mid-window is `not_verified` (the primary pause detector). 8. **Rule absent** — an authoritative `2xx` with no matching rule. 9. **`KeepLast`** — a note naming a stale-state blind spot. ## Health: `error` vs `nodata` -- `health=error` means the query **failed** — a malfunction. Sustained past `healthGrace`, it makes the rule `unobservable`. +- `health=error` means the query **failed** — a malfunction. Sustained past `healthGrace`, it makes the rule `not_verified`. - `health=nodata` means the query **ran and returned no series** — indistinguishable from a quiet system. It is not fatal by default; most of a fleet runs `no_data_state: OK`. ## Early exit (fail-fast) `check` does not have to wait for the whole window to know the run has failed. As soon as it observes a condition that cannot become a pass, it stops and classifies the sub-window it did see: -- a **post-`from` bad onset** — the full classifier would call it `newly_bad` (or `flapping`), which fails whether or not it later clears; or +- a **post-`from` bad onset** — the full classifier would call it `new_failure` (or `unstable`), which fails whether or not it later clears; or - an **inability** — a heartbeat gap, a sustained `health=error` run, a stale evaluation, an in-window pause, or an absent rule. A preexisting bad instance is deliberately **not** terminal: if it clears before `to` the full run would call it `recovered`, which passes. @@ -91,6 +93,6 @@ Fail-fast is on by default and always preserves the failure: an early run can ex ## The drain wait and `transitionGrace` -A condition that arises just before `to` becomes `firing` only at the first evaluation after its `for` elapses. `transitionGrace` (derived from the watched rules' `for` values) extends the classification bound past `to` so such a surfacing condition is caught. After collection, a **drain wait** polls until each rule has evaluated through `to + transitionGrace` (bounded by `drainTimeout`); a rule that never does is `unobservable`. +A condition that arises just before `to` becomes `firing` only at the first evaluation after its `for` elapses. `transitionGrace` (derived from the watched rules' `for` values) extends the classification bound past `to` so such a surfacing condition is caught. After collection, a **drain wait** polls until each rule has evaluated through `to + transitionGrace` (bounded by `drainTimeout`); a rule that never does is `not_verified`. Run time = `(to − from) + transitionGrace + drainTimeout`. This is printed at start, and the grace is warned about when it exceeds a quarter of the window — the window may be too short for the alert's `for`. diff --git a/grafana-alertcheck/docs/reference/cli.md b/grafana-alertcheck/docs/reference/cli.md index f059cc83c..312ea4860 100644 --- a/grafana-alertcheck/docs/reference/cli.md +++ b/grafana-alertcheck/docs/reference/cli.md @@ -102,7 +102,7 @@ Datasource-managed and recording rules are refused with a specific error. A name ## Output and exit codes -The human table goes to **stderr**: `RESULTS` (one row per rule), `VIOLATIONS` (one per distinct rule/outcome/state/health/note signature, with a `COUNT` of the instances it stands for — instance identity is only in the JSON), and `THRESHOLDS` (each rule's `maxGap`/`healthGrace`/`evalStaleAfter` plus global `transitionGrace`/`drainTimeout` and the largest measured clock skew). `--output json` writes the result to stdout. +The human table goes to **stderr**: `RESULTS` (one row per rule, with the verdict, time broken, check cadence and whether the window was observed), `VIOLATIONS` (one per distinct rule/verdict/state/health/note signature, with an `INSTANCES` count of the instances it stands for — instance identity is only in the JSON), and `LIMITS USED` (each rule's observation limits in plain words, explained by a legend under the table, plus the extra observation time, the evaluation wait and the largest measured clock difference). The JSON outcome values are `healthy`, `new_failure`, `still_failing`, `recovered`, `unstable`, `paused`, `not_verified` and the synthetic `not_counted`. `--output json` writes the result to stdout. | Code | Meaning | | ---- | ------- | diff --git a/grafana-alertcheck/docs/reference/log-format.md b/grafana-alertcheck/docs/reference/log-format.md index 9dc27c29c..0bf1830af 100644 --- a/grafana-alertcheck/docs/reference/log-format.md +++ b/grafana-alertcheck/docs/reference/log-format.md @@ -49,7 +49,7 @@ The header must be line 1, appear once, and carry `schema_version` `1` (any othe ``` - `url` and `rules` are the log's identity — `check` validates them against the current environment and a fresh ruler read. -- `is_paused` records the pause state at record start (the moment `skipped` means). +- `is_paused` records the pause state at record start (the moment `paused` means). - `poll_every_seconds` is the cadence the recording **actually used** (after any `--poll-interval` override). `check` derives `maxGap` from it, never from `interval_seconds`. - `for_seconds`, `interval_seconds`, `no_data_state`, `exec_err_state` are forensic only — `check` re-resolves definitions and never reads them back. @@ -95,4 +95,4 @@ Instance keys are the JSON encoding of the labels map (with stable key order), s { "type": "stopped", "at": "2026-09-07T10:10:30Z" } ``` -`at` is the recorder's own stop time. `check` compares it against `to + transitionGrace`; absent or earlier is `unobservable` — never a pass. +`at` is the recorder's own stop time. `check` compares it against `to + transitionGrace`; absent or earlier is `not_verified` — never a pass. diff --git a/grafana-alertcheck/internal/gate/check.go b/grafana-alertcheck/internal/gate/check.go index 63bc1dab2..9faa8f1f1 100644 --- a/grafana-alertcheck/internal/gate/check.go +++ b/grafana-alertcheck/internal/gate/check.go @@ -685,13 +685,13 @@ type drainVerdict struct { // drainWait is the final liveness check: did each rule evaluate through the // end of the window? A rule that cannot answer within drainTimeout is -// unobservable, never a pass. It returns one verdict per rule it could not +// not_verified, never a pass. It returns one verdict per rule it could not // clear (keyed by UID); an error only for a hard failure of the wait itself. // // Two kinds of rule are excluded up front because draining them could not // change a verdict: a rule the HEADER says was paused at the window open (the // header, not the late-resolved definitions — see Header.pausedAtStart), and a -// rule whose last poll says Found == false (already unobservable via rule_absent). +// rule whose last poll says Found == false (already not_verified via rule_absent). func drainWait(ctx context.Context, cfg Config, src Source, defs []Definition, pausedAtStart map[string]bool, rt map[string]RuleTimings, polls []Poll, windowEnd time.Time, timeout time.Duration) (map[string]drainVerdict, error) { @@ -862,17 +862,17 @@ func mergeDrainTimeouts(res Result, drained map[string]drainVerdict) (Result, er cov.Notes = append(cov.Notes, verdict.note) res.Coverage[uid] = cov - if res.Verdicts[i].Outcome != OutcomeUnobservable { + if res.Verdicts[i].Outcome != OutcomeNotVerified { names = append(names, fmt.Sprintf("%s (%s)", res.Verdicts[i].Alert, verdict.reason)) } - res.Verdicts[i].Outcome = OutcomeUnobservable + res.Verdicts[i].Outcome = OutcomeNotVerified res.Verdicts[i].Note = strings.Join(cov.Notes, "; ") } if len(names) == 0 { - // Every drained rule was already unobservable for an earlier reason, + // Every drained rule was already not_verified for an earlier reason, // so decide's own error already stops the run. Adding a second error // saying the same thing would only make the message longer. return res, nil } - return res, fmt.Errorf("gate: %d rule(s) unobservable at the drain wait: %s", len(names), strings.Join(names, "; ")) + return res, fmt.Errorf("gate: %d rule(s) not verified at the drain wait: %s", len(names), strings.Join(names, "; ")) } diff --git a/grafana-alertcheck/internal/gate/check_test.go b/grafana-alertcheck/internal/gate/check_test.go index 2ba8747b0..2dc6cb1a6 100644 --- a/grafana-alertcheck/internal/gate/check_test.go +++ b/grafana-alertcheck/internal/gate/check_test.go @@ -268,7 +268,7 @@ func TestCheckSingleStepCleanWindowPasses(t *testing.T) { // A pass is exactly this shape. require.Empty(t, res.Violations) require.Len(t, res.Verdicts, 1) - require.Equal(t, OutcomeClean, res.Verdicts[0].Outcome) + require.Equal(t, OutcomeHealthy, res.Verdicts[0].Outcome) cov := res.Coverage[checkUID] require.True(t, cov.Proved) require.False(t, cov.Unobservable) @@ -340,7 +340,7 @@ func TestCheckSingleStepContinuousHealthErrorIsUnobservable(t *testing.T) { res, err := check(context.Background(), cfg, src) require.Error(t, err, "continuous health=error must be unobservable") require.Len(t, res.Verdicts, 1) - require.Equal(t, OutcomeUnobservable, res.Verdicts[0].Outcome) + require.Equal(t, OutcomeNotVerified, res.Verdicts[0].Outcome) require.Equal(t, ReasonHealthError, res.Coverage[def.UID].Reason) } @@ -361,7 +361,7 @@ func TestCheckSingleStepFiringInstanceReportsWithoutExitingEarly(t *testing.T) { res, err := check(context.Background(), cfg, src) require.NoError(t, err, "a violation is exit 1, not an error") require.Len(t, res.Violations, 1) - require.Equal(t, OutcomePersistentlyBad, res.Violations[0].Outcome) + require.Equal(t, OutcomeStillFailing, res.Violations[0].Outcome) require.False(t, clock.Now().Before(cfg.To.Add(checkGrace)), "exited early; collection must run to to+grace") } @@ -387,12 +387,12 @@ func TestCheckSingleStepNewOnsetExitsEarlyByDefault(t *testing.T) { res, err := check(context.Background(), cfg, src) require.NoError(t, err, "a violation is exit 1, not an error") require.Len(t, res.Violations, 1) - require.Equal(t, OutcomeNewlyBad, res.Violations[0].Outcome) + require.Equal(t, OutcomeNewFailure, res.Violations[0].Outcome) require.True(t, clock.Now().Before(cfg.To.Add(checkGrace)), "the runner was released only after to+grace; fail-fast did not fire") require.NotNil(t, res.TerminatedEarly) require.Equal(t, TerminationViolation, res.TerminatedEarly.Kind) - require.Equal(t, OutcomeNewlyBad, res.TerminatedEarly.Outcome) + require.Equal(t, OutcomeNewFailure, res.TerminatedEarly.Outcome) require.True(t, res.To.Equal(cfg.To), "the requested window is still reported") require.Contains(t, notesOf(cfg), "fail-fast") } @@ -414,7 +414,7 @@ func TestCheckSingleStepPreexistingBadDoesNotExitEarly(t *testing.T) { res, err := check(context.Background(), cfg, src) require.NoError(t, err) require.Len(t, res.Violations, 1) - require.Equal(t, OutcomePersistentlyBad, res.Violations[0].Outcome) + require.Equal(t, OutcomeStillFailing, res.Violations[0].Outcome) require.Nil(t, res.TerminatedEarly, "a preexisting condition can still recover, so it is not terminal") require.False(t, clock.Now().Before(cfg.To.Add(checkGrace)), "exited early; collection must run to to+grace for a preexisting bad instance") @@ -442,7 +442,7 @@ func TestCheckSingleStepNewOnsetNoFailFastRunsToTheEnd(t *testing.T) { res, err := check(context.Background(), cfg, src) require.NoError(t, err) require.Len(t, res.Violations, 1) - require.Equal(t, OutcomeNewlyBad, res.Violations[0].Outcome) + require.Equal(t, OutcomeNewFailure, res.Violations[0].Outcome) require.Nil(t, res.TerminatedEarly) require.False(t, clock.Now().Before(cfg.To.Add(checkGrace)), "with --no-fail-fast the loop must run to to+grace") @@ -650,7 +650,7 @@ func TestCheckRecorderModeCleanWindowPasses(t *testing.T) { require.NoError(t, err) require.Empty(t, res.Violations) require.Len(t, res.Verdicts, 1) - require.Equal(t, OutcomeClean, res.Verdicts[0].Outcome) + require.Equal(t, OutcomeHealthy, res.Verdicts[0].Outcome) require.Equal(t, "13.1.0", res.GrafanaVersion) // The collection loop still waited out to+transitionGrace even though the // recorder had already finished. @@ -701,7 +701,7 @@ func TestCheckRecorderModeExitsEarlyOnANewOnset(t *testing.T) { require.NoError(t, err) require.NotNil(t, res.TerminatedEarly) require.Equal(t, TerminationViolation, res.TerminatedEarly.Kind) - require.Equal(t, OutcomeNewlyBad, res.TerminatedEarly.Outcome) + require.Equal(t, OutcomeNewFailure, res.TerminatedEarly.Outcome) require.Len(t, res.Violations, 1) require.True(t, clock.Now().Before(cfg.To.Add(checkGrace)), "the runner was held to the window; fail-fast did not fire") @@ -721,7 +721,7 @@ func TestCheckRecorderModeNoFailFastRunsToTheEnd(t *testing.T) { require.NoError(t, err) require.Nil(t, res.TerminatedEarly) require.Len(t, res.Violations, 1) - require.Equal(t, OutcomeNewlyBad, res.Violations[0].Outcome) + require.Equal(t, OutcomeNewFailure, res.Violations[0].Outcome) require.False(t, clock.Now().Before(cfg.To.Add(checkGrace)), "with --no-fail-fast the loop must run to to+grace") } @@ -803,7 +803,7 @@ func TestCheckRecorderModeFromSameSecondAsStartedAtPasses(t *testing.T) { require.NoError(t, err) require.Empty(t, res.Violations) require.Len(t, res.Verdicts, 1) - require.Equal(t, OutcomeClean, res.Verdicts[0].Outcome) + require.Equal(t, OutcomeHealthy, res.Verdicts[0].Outcome) } // The coverage proof failed: a hole in the middle of the recording is not @@ -835,7 +835,7 @@ func TestCheckFailClosedOnCoverageGap(t *testing.T) { res, err := check(context.Background(), cfg, newCheckSource(nil)) require.Error(t, err, "the coverage gap to fail closed") require.Equal(t, ReasonHeartbeatGap, res.Coverage[checkUID].Reason) - require.Equal(t, OutcomeUnobservable, res.Verdicts[0].Outcome) + require.Equal(t, OutcomeNotVerified, res.Verdicts[0].Outcome) } // An episode fully between the deploy and the start of the check: recorder @@ -868,7 +868,7 @@ func TestCheckRecorderModeFindsAGapImmediatelyAfterTheDeploy(t *testing.T) { res, err := check(context.Background(), cfg, newCheckSource(nil)) require.Error(t, err, "a hole right after the deploy hides whatever happened there") require.Len(t, res.Verdicts, 1) - require.Equal(t, OutcomeUnobservable, res.Verdicts[0].Outcome, "never clean") + require.Equal(t, OutcomeNotVerified, res.Verdicts[0].Outcome, "never clean") } // The drain limit passed. The recording itself is clean, so this isolates the @@ -897,7 +897,7 @@ func TestCheckFailClosedOnDrainTimeout(t *testing.T) { res, err := check(context.Background(), cfg, src) require.Error(t, err, "the drain limit to fail closed") require.Equal(t, ReasonDrainTimeout, res.Coverage[checkUID].Reason) - require.Equal(t, OutcomeUnobservable, res.Verdicts[0].Outcome) + require.Equal(t, OutcomeNotVerified, res.Verdicts[0].Outcome) require.Contains(t, res.Verdicts[0].Note, "drain limit") require.GreaterOrEqual(t, clock.Now().Sub(windowEnd), checkDrainLimit, "the rule never evaluates through the window, so the drain wait must run its full limit") @@ -996,10 +996,10 @@ func pausedAfterWindowCheck(t *testing.T, allowPaused bool) (Result, Config, err func TestCheckPausingARuleAfterTheWindowDoesNotMakeItSkipped(t *testing.T) { res, _, err := pausedAfterWindowCheck(t, false) require.NoError(t, err) - require.Equal(t, OutcomeNewlyBad, res.Verdicts[0].Outcome, + require.Equal(t, OutcomeNewFailure, res.Verdicts[0].Outcome, "the rule was active for the whole window and fired inside it") require.Len(t, res.Violations, 1) - require.Equal(t, OutcomeNewlyBad, res.Violations[0].Outcome) + require.Equal(t, OutcomeNewFailure, res.Violations[0].Outcome) require.NotContains(t, res.Verdicts[0].Note, "paused before the window opened") } @@ -1047,7 +1047,7 @@ func TestCheckHeaderPausedRuleStaysSkipped(t *testing.T) { res, err := run(false) require.NoError(t, err, "a skipped rule is a known condition, not an inability") - require.Equal(t, OutcomeSkipped, res.Verdicts[0].Outcome) + require.Equal(t, OutcomePaused, res.Verdicts[0].Outcome) _, ok := res.Coverage[checkUID] require.False(t, ok, "a skipped rule has no coverage to prove") require.Len(t, res.Violations, 1, "the MinObserved shortfall") @@ -1214,7 +1214,7 @@ func TestCheckDeadPidWithNoSentinelIsUnobservable(t *testing.T) { res, err := check(context.Background(), cfg, newCheckSource(nil)) require.Error(t, err, "no sentinel means the recorder never proved it ran to the end") require.Len(t, res.Verdicts, 1) - require.Equal(t, OutcomeUnobservable, res.Verdicts[0].Outcome) + require.Equal(t, OutcomeNotVerified, res.Verdicts[0].Outcome) } // An incomplete last line gives exit 2. log_test.go's TestReadLogRejectsBadLogs @@ -1332,8 +1332,8 @@ func TestMergeDrainTimeoutsNamesEveryUnobservableRule(t *testing.T) { "b": {Unobservable: true, Reason: ReasonHeartbeatGap, Notes: []string{"rule \"B\": gap"}}, }, Verdicts: []RuleVerdict{ - {Alert: "A", RuleUID: "a", Outcome: OutcomeClean}, - {Alert: "B", RuleUID: "b", Outcome: OutcomeUnobservable}, + {Alert: "A", RuleUID: "a", Outcome: OutcomeHealthy}, + {Alert: "B", RuleUID: "b", Outcome: OutcomeNotVerified}, }, } @@ -1341,9 +1341,9 @@ func TestMergeDrainTimeoutsNamesEveryUnobservableRule(t *testing.T) { "a": {reason: ReasonDrainTimeout, note: "rule \"A\": did not evaluate through the end within the drain limit"}, "b": {reason: ReasonDrainTimeout, note: "rule \"B\": did not evaluate through the end within the drain limit"}, }) - require.Error(t, err, "naming the newly unobservable rule") - require.Contains(t, err.Error(), "unobservable at the drain wait") - // Only A is newly unobservable; B was already, so naming it twice would + require.Error(t, err, "naming the newly not_verified rule") + require.Contains(t, err.Error(), "not verified at the drain wait") + // Only A is newly not_verified; B was already, so naming it twice would // only lengthen the message. require.Contains(t, err.Error(), "A ("+string(ReasonDrainTimeout)+")") require.NotContains(t, err.Error(), "B (") @@ -1351,7 +1351,7 @@ func TestMergeDrainTimeoutsNamesEveryUnobservableRule(t *testing.T) { // B keeps the reason the coverage proof gave it — the FIRST reason wins, // as it does inside proveCoverage. require.Equal(t, ReasonHeartbeatGap, merged.Coverage["b"].Reason) - require.Equal(t, OutcomeUnobservable, merged.Verdicts[0].Outcome) + require.Equal(t, OutcomeNotVerified, merged.Verdicts[0].Outcome) } // ReadLogHeader is the one read of a log a writer may still hold, so its diff --git a/grafana-alertcheck/internal/gate/classify.go b/grafana-alertcheck/internal/gate/classify.go index 41e2a5f7b..ab31edb3b 100644 --- a/grafana-alertcheck/internal/gate/classify.go +++ b/grafana-alertcheck/internal/gate/classify.go @@ -19,26 +19,30 @@ const ReasonNodata UnobservableReason = "nodata" type Outcome string const ( - OutcomeClean Outcome = "clean" - OutcomeNewlyBad Outcome = "newly_bad" - OutcomeRecovered Outcome = "recovered" - OutcomePersistentlyBad Outcome = "persistently_bad" - OutcomeFlapping Outcome = "flapping" - OutcomeSkipped Outcome = "skipped" - OutcomeUnobservable Outcome = "unobservable" + OutcomeHealthy Outcome = "healthy" + OutcomeNewFailure Outcome = "new_failure" + OutcomeRecovered Outcome = "recovered" + OutcomeStillFailing Outcome = "still_failing" + OutcomeUnstable Outcome = "unstable" + OutcomePaused Outcome = "paused" + OutcomeNotVerified Outcome = "not_verified" + // OutcomeNotCounted is synthetic: decide uses it for a --min-observed + // deficit row that no resolved rule explains. It is not a verdict on an + // alert, so it is deliberately not named after one. + OutcomeNotCounted Outcome = "not_counted" ) // PreexistingPolicy governs only the ONE ambiguous case in the outcome table: -// an instance that was already bad when the window opened. A newly_bad or -// flapping instance is a fail under every policy, so this type only ever -// changes how `recovered` and `persistently_bad` are judged (isViolation +// an instance that was already bad when the window opened. A new_failure or +// unstable instance is a fail under every policy, so this type only ever +// changes how `recovered` and `still_failing` are judged (isViolation // below). type PreexistingPolicy string const ( // PreexistingFailUnlessRecovered is the default: a preexisting instance // that clears and stays clear is a pass (`recovered`); one that never - // clears is still a fail (`persistently_bad`). + // clears is still a fail (`still_failing`). PreexistingFailUnlessRecovered PreexistingPolicy = "fail-unless-recovered" // PreexistingFail makes ANY preexisting instance a fail, even one that // recovers — for a user who wants no benefit of the doubt for a @@ -46,7 +50,7 @@ const ( PreexistingFail PreexistingPolicy = "fail" // PreexistingIgnore disregards a preexisting instance entirely, whether // it recovers or stays bad for the whole window: only a genuinely NEW - // bad episode (newly_bad or flapping) can fail the rule. + // bad episode (new_failure or unstable) can fail the rule. PreexistingIgnore PreexistingPolicy = "ignore" ) @@ -294,7 +298,7 @@ func classifyRule(def Definition, polls []Poll, from, windowEnd time.Time, badSt slices.Sort(order) var ( - outcome = OutcomeClean + outcome = OutcomeHealthy badFor []episode viols []Violation ) @@ -311,17 +315,17 @@ func classifyRule(def Definition, polls []Poll, from, windowEnd time.Time, badSt var instOutcome Outcome switch { case len(tl.episodes) > 1: - instOutcome = OutcomeFlapping + instOutcome = OutcomeUnstable case tl.preexisting: if tl.episodes[0].closedByRealClear { instOutcome = OutcomeRecovered } else { - instOutcome = OutcomePersistentlyBad + instOutcome = OutcomeStillFailing } default: // A genuinely new onset fails whether or not it clears in-window; // only a preexisting condition earns `recovered`. - instOutcome = OutcomeNewlyBad + instOutcome = OutcomeNewFailure } if outcomeRank(instOutcome) > outcomeRank(outcome) { @@ -353,17 +357,17 @@ func classifyRule(def Definition, polls []Poll, from, windowEnd time.Time, badSt } // isViolation decides whether one instance's outcome counts against the run, -// once the preexisting policy is applied. newly_bad and flapping always do: +// once the preexisting policy is applied. new_failure and unstable always do: // both contain a genuinely new bad episode, so no policy forgives them. -// recovered and persistently_bad are, by classifyRule's construction, -// ALWAYS preexisting (a non-preexisting single episode is newly_bad instead, +// recovered and still_failing are, by classifyRule's construction, +// ALWAYS preexisting (a non-preexisting single episode is new_failure instead, // regardless of whether it clears) — so these are the only two policy can // change, and isViolation needs no separate preexisting flag to know that. func isViolation(o Outcome, pol PreexistingPolicy) bool { switch o { - case OutcomeNewlyBad, OutcomeFlapping: + case OutcomeNewFailure, OutcomeUnstable: return true - case OutcomePersistentlyBad: + case OutcomeStillFailing: return pol != PreexistingIgnore case OutcomeRecovered: return pol == PreexistingFail @@ -375,25 +379,25 @@ func isViolation(o Outcome, pol PreexistingPolicy) bool { // outcomeRank orders outcomes for classifyRule's worst-of reduction across a // rule's instances: // -// unobservable > {flapping, persistently_bad, newly_bad} > recovered > -// skipped > clean +// not_verified > {unstable, still_failing, new_failure} > recovered > +// paused > healthy // -// with unobservable and skipped applied outside this function (decide owns -// both: unobservable from CoverageResult, skipped from the log header). The +// with not_verified and paused applied outside this function (decide owns +// both: not_verified from CoverageResult, paused from the log header). The // three fail values are not ranked against each other by anything that reads // this, so their relative order here is an arbitrary but fixed tie-break, not // a claim that one is worse than another. func outcomeRank(o Outcome) int { switch o { - case OutcomeFlapping: + case OutcomeUnstable: return 4 - case OutcomePersistentlyBad: + case OutcomeStillFailing: return 3 - case OutcomeNewlyBad: + case OutcomeNewFailure: return 2 case OutcomeRecovered: return 1 - default: // OutcomeClean + default: // OutcomeHealthy return 0 } } @@ -483,7 +487,7 @@ func applyNodataPolicy(def Definition, polls []Poll, cov *CoverageResult, t Rule // decide is the pure seam between the collected evidence and the CLI's exit // code, and carries nearly the whole test suite because of it. It combines // proveCoverage's nine checks with classifyRule's timelines under one Policy, -// and owns the inability-beats-violation rule: any unobservable rule makes +// and owns the inability-beats-violation rule: any not-verified rule makes // decide return a non-nil error, which the CLI maps to exit 2 unconditionally // — never to 0 or 1, and never suppressed by a real violation found alongside // it. @@ -532,13 +536,13 @@ func decide(h Header, polls []Poll, sentinel *time.Time, defs []Definition, windowEnd := pol.To.Add(gt.transitionGrace) var ( - skippedRules []Definition + pausedRules []Definition watchedCount int anyUnobservable bool unobservableNames []string ) - // `skipped` is decided from the header, never from defs: defs are resolved + // `paused` is decided from the header, never from defs: defs are resolved // after the window closed, so Definition.IsPaused describes the present, // while Header.pausedAtStart describes the window open — the only moment // "paused before the window opened" can mean. @@ -546,9 +550,9 @@ func decide(h Header, polls []Poll, sentinel *time.Time, defs []Definition, for _, def := range defs { if pausedAtStart[def.UID] { - skippedRules = append(skippedRules, def) + pausedRules = append(pausedRules, def) result.Verdicts = append(result.Verdicts, RuleVerdict{ - Alert: def.Title, RuleUID: def.UID, Outcome: OutcomeSkipped, + Alert: def.Title, RuleUID: def.UID, Outcome: OutcomePaused, PollEvery: rt[def.UID].pollEvery, Note: "paused before the window opened", }) @@ -571,7 +575,7 @@ func decide(h Header, polls []Poll, sentinel *time.Time, defs []Definition, outcome, badFor, viols := classifyRule(def, polls, pol.From, windowEnd, badStates, pol.Preexisting) if cov.Unobservable { - outcome = OutcomeUnobservable + outcome = OutcomeNotVerified anyUnobservable = true unobservableNames = append(unobservableNames, fmt.Sprintf("%s (%s)", def.Title, cov.Reason)) } @@ -588,9 +592,9 @@ func decide(h Header, polls []Poll, sentinel *time.Time, defs []Definition, counted := watchedCount var attributable []Definition if pol.AllowPaused { - counted += len(skippedRules) + counted += len(pausedRules) } else { - attributable = skippedRules + attributable = pausedRules } if shortfall := minObserved - counted; shortfall > 0 { attributed := 0 @@ -603,7 +607,7 @@ func decide(h Header, polls []Poll, sentinel *time.Time, defs []Definition, // prints Note verbatim rather than re-deriving the hint, so the // exact wording here is what an operator reads. result.Violations = append(result.Violations, Violation{ - Alert: def.Title, RuleUID: def.UID, Outcome: OutcomeSkipped, + Alert: def.Title, RuleUID: def.UID, Outcome: OutcomePaused, Note: "paused before the window opened; counts against --min-observed unless --allow-paused is set", }) attributed++ @@ -615,14 +619,14 @@ func decide(h Header, polls []Poll, sentinel *time.Time, defs []Definition, // rule state read from a real poll, and this Violation never // touched one. result.Violations = append(result.Violations, Violation{ - Outcome: OutcomeSkipped, + Outcome: OutcomeNotCounted, Note: fmt.Sprintf("min-observed %d exceeds the %d rule(s) counted as observed", minObserved, counted), }) } } if anyUnobservable { - return result, fmt.Errorf("gate: %d rule(s) unobservable: %s", len(unobservableNames), strings.Join(unobservableNames, "; ")) + return result, fmt.Errorf("gate: %d rule(s) not verified: %s", len(unobservableNames), strings.Join(unobservableNames, "; ")) } return result, nil } diff --git a/grafana-alertcheck/internal/gate/classify_test.go b/grafana-alertcheck/internal/gate/classify_test.go index 0e6b16c8e..14cc95248 100644 --- a/grafana-alertcheck/internal/gate/classify_test.go +++ b/grafana-alertcheck/internal/gate/classify_test.go @@ -1,6 +1,7 @@ package gate import ( + "encoding/json" "testing" "time" @@ -48,7 +49,7 @@ func pausedHeader(startedAt time.Time, pausedUIDs ...string) Header { return h } -// --- clean / newly_bad --- +// --- healthy / new_failure --- func TestClassifyRule_NoEvidenceIsClean(t *testing.T) { from := time.Date(2026, 1, 1, 0, 0, 0, 0, time.UTC) @@ -57,7 +58,7 @@ func TestClassifyRule_NoEvidenceIsClean(t *testing.T) { polls := []Poll{quietPoll("r1", from), quietPoll("r1", to)} outcome, badFor, viols := classifyRule(def, polls, from, to, defaultBad, PreexistingFailUnlessRecovered) - require.Equal(t, OutcomeClean, outcome) + require.Equal(t, OutcomeHealthy, outcome) require.Zero(t, badFor) require.Empty(t, viols) } @@ -74,10 +75,10 @@ func TestClassifyRule_NewOnsetInsideWindowIsNewlyBad(t *testing.T) { abnormalPoll("r1", to, StateFiring, lbl("a"), onset), } outcome, badFor, viols := classifyRule(def, polls, from, to, defaultBad, PreexistingFailUnlessRecovered) - require.Equal(t, OutcomeNewlyBad, outcome) + require.Equal(t, OutcomeNewFailure, outcome) require.Equal(t, to.Sub(onset), badFor) require.Len(t, viols, 1) - require.Equal(t, OutcomeNewlyBad, viols[0].Outcome) + require.Equal(t, OutcomeNewFailure, viols[0].Outcome) } // A genuinely new bad episode fails even if it clears again before the window @@ -96,11 +97,11 @@ func TestClassifyRule_NewOnsetThatClearsStillFails(t *testing.T) { quietPoll("r1", to), } outcome, _, viols := classifyRule(def, polls, from, to, defaultBad, PreexistingFailUnlessRecovered) - require.Equal(t, OutcomeNewlyBad, outcome, "even though it cleared") + require.Equal(t, OutcomeNewFailure, outcome, "even though it cleared") require.Len(t, viols, 1) } -// --- recovered / persistently_bad (preexisting) --- +// --- recovered / still_failing (preexisting) --- func TestClassifyRule_PreexistingThatRecoversIsRecoveredAndNotAViolation(t *testing.T) { from := time.Date(2026, 1, 1, 0, 0, 0, 0, time.UTC) @@ -151,13 +152,13 @@ func TestClassifyRule_PreexistingStillBadAtWindowEndIsPersistentlyBad(t *testing abnormalPoll("r1", to, StateFiring, lbl("a"), from.Add(-time.Hour)), } outcome, badFor, viols := classifyRule(def, polls, from, to, defaultBad, PreexistingFailUnlessRecovered) - require.Equal(t, OutcomePersistentlyBad, outcome) + require.Equal(t, OutcomeStillFailing, outcome) require.Equal(t, to.Sub(from), badFor) require.Len(t, viols, 1) - require.Equal(t, OutcomePersistentlyBad, viols[0].Outcome) + require.Equal(t, OutcomeStillFailing, viols[0].Outcome) } -// --- flapping --- +// --- unstable --- func TestClassifyRule_ClearThenBadAgainIsFlapping(t *testing.T) { from := time.Date(2026, 1, 1, 0, 0, 0, 0, time.UTC) @@ -172,12 +173,12 @@ func TestClassifyRule_ClearThenBadAgainIsFlapping(t *testing.T) { quietPoll("r1", to), } outcome, _, viols := classifyRule(def, polls, from, to, defaultBad, PreexistingFailUnlessRecovered) - require.Equal(t, OutcomeFlapping, outcome) + require.Equal(t, OutcomeUnstable, outcome) require.Len(t, viols, 1) - require.Equal(t, OutcomeFlapping, viols[0].Outcome, "always a fail regardless of policy") + require.Equal(t, OutcomeUnstable, viols[0].Outcome, "always a fail regardless of policy") } -// A clear and then a second bad state gives flapping, wherever the second bad +// A clear and then a second bad state gives unstable, wherever the second bad // state lands. A table over where the second onset falls — immediately after // the clear, mid-window, and right at the // last instant before windowEnd — closes the boundary this single fixed @@ -206,9 +207,9 @@ func TestClassifyRule_FlappingAtEveryTimingOfTheSecondOnset(t *testing.T) { quietPoll("r1", to), } outcome, _, viols := classifyRule(def, polls, from, to, defaultBad, PreexistingFailUnlessRecovered) - require.Equalf(t, OutcomeFlapping, outcome, "second onset at %s", tc.secondOnset) + require.Equalf(t, OutcomeUnstable, outcome, "second onset at %s", tc.secondOnset) require.Len(t, viols, 1) - require.Equal(t, OutcomeFlapping, viols[0].Outcome) + require.Equal(t, OutcomeUnstable, viols[0].Outcome) }) } } @@ -227,7 +228,7 @@ func TestClassifyRule_VanishedWhileBadStaysPersistentlyBad(t *testing.T) { quietPoll("r1", to), } outcome, badFor, viols := classifyRule(def, polls, from, to, defaultBad, PreexistingFailUnlessRecovered) - require.Equal(t, OutcomePersistentlyBad, outcome, "a vanish must never read as a recovery") + require.Equal(t, OutcomeStillFailing, outcome, "a vanish must never read as a recovery") require.Equal(t, to.Sub(from), badFor, "the freeze must hold the episode open to windowEnd") require.Len(t, viols, 1) } @@ -246,7 +247,7 @@ func TestClassifyRule_VanishedWhileNeverBadIsUninteresting(t *testing.T) { quietPoll("r1", to), } outcome, badFor, viols := classifyRule(def, polls, from, to, defaultBad, PreexistingFailUnlessRecovered) - require.Equal(t, OutcomeClean, outcome) + require.Equal(t, OutcomeHealthy, outcome) require.Zero(t, badFor) require.Empty(t, viols) } @@ -281,7 +282,7 @@ func TestClassifyRule_PreexistingPolicyIgnoreForgivesPersistentlyBad(t *testing. abnormalPoll("r1", to, StateFiring, lbl("a"), from.Add(-time.Hour)), } outcome, _, viols := classifyRule(def, polls, from, to, defaultBad, PreexistingIgnore) - require.Equal(t, OutcomePersistentlyBad, outcome, "the descriptive outcome does not change under policy=ignore") + require.Equal(t, OutcomeStillFailing, outcome, "the descriptive outcome does not change under policy=ignore") require.Empty(t, viols, "policy=ignore disregards a preexisting instance even if it never recovers") } @@ -297,7 +298,7 @@ func TestClassifyRule_PreexistingPolicyIgnoreStillFailsANewOnset(t *testing.T) { abnormalPoll("r1", to, StateFiring, lbl("a"), onset), } outcome, _, viols := classifyRule(def, polls, from, to, defaultBad, PreexistingIgnore) - require.Equal(t, OutcomeNewlyBad, outcome) + require.Equal(t, OutcomeNewFailure, outcome) require.Len(t, viols, 1, "ignore only forgives PREEXISTING badness") } @@ -323,9 +324,9 @@ func TestClassifyRule_WorstOfMultipleInstancesWins(t *testing.T) { }, } outcome, _, viols := classifyRule(def, polls, from, to, defaultBad, PreexistingFailUnlessRecovered) - require.Equal(t, OutcomePersistentlyBad, outcome, "the worse of {recovered, persistently_bad}") + require.Equal(t, OutcomeStillFailing, outcome, "the worse of {recovered, still_failing}") require.Len(t, viols, 1) - require.Equal(t, OutcomePersistentlyBad, viols[0].Outcome) + require.Equal(t, OutcomeStillFailing, viols[0].Outcome) } // --- decide(): skipped rules, unobservable, MinObserved, exit mapping --- @@ -344,9 +345,9 @@ func TestDecide_SkippedRuleNeverReachesProveCoverage(t *testing.T) { // The HEADER is what says paused — decide reads skipped from there, not // from def.IsPaused, which is a post-window reading (Header.pausedAtStart). res, err := decide(pausedHeader(from.Add(-time.Hour), "r1"), nil, nil, defs, rt, gt, pol) - require.NoError(t, err, "a rule paused before the window is skipped, not unobservable") + require.NoError(t, err, "a rule paused before the window is paused, not not_verified") require.Len(t, res.Verdicts, 1) - require.Equal(t, OutcomeSkipped, res.Verdicts[0].Outcome) + require.Equal(t, OutcomePaused, res.Verdicts[0].Outcome) _, ok := res.Coverage["r1"] require.False(t, ok, "a skipped rule has no coverage to prove") } @@ -365,11 +366,11 @@ func TestDecide_UnobservableRuleAlwaysReturnsAnError(t *testing.T) { res, err := decide(Header{StartedAt: from.Add(-time.Hour)}, nil, nil, defs, rt, gt, pol) require.Error(t, err, "an unobservable rule must always fail the run") require.Len(t, res.Verdicts, 1) - require.Equal(t, OutcomeUnobservable, res.Verdicts[0].Outcome) + require.Equal(t, OutcomeNotVerified, res.Verdicts[0].Outcome) } // Any unobservable rule means exit 2, with no exception — even alongside a -// real newly_bad. +// real new_failure. func TestDecide_UnobservableWinsEvenAlongsideARealViolation(t *testing.T) { from := time.Date(2026, 1, 1, 0, 0, 0, 0, time.UTC) to := from.Add(10 * time.Minute) @@ -407,11 +408,11 @@ func TestDecide_UnobservableWinsEvenAlongsideARealViolation(t *testing.T) { gotBad = v.Outcome } } - require.Equal(t, OutcomeUnobservable, gotBroken) - require.Equal(t, OutcomeNewlyBad, gotBad, + require.Equal(t, OutcomeNotVerified, gotBroken) + require.Equal(t, OutcomeNewFailure, gotBad, "classification still runs and is still visible in Verdicts") require.NotEmpty(t, res.Violations, - "the newly_bad instance still reported even though the run fails on the unobservable rule") + "the new_failure instance still reported even though the run fails on the not_verified rule") } // A clean verdict with a coverage gap must never give exit 0, and recovered @@ -429,7 +430,7 @@ func TestDecide_UnobservableRuleWinsOverEveryFavorableOutcome(t *testing.T) { wantOutcome Outcome }{ { - name: "clean", + name: "healthy", goodPolls: func() []Poll { var polls []Poll for ts := from; !ts.After(to); ts = ts.Add(30 * time.Second) { @@ -437,7 +438,7 @@ func TestDecide_UnobservableRuleWinsOverEveryFavorableOutcome(t *testing.T) { } return polls }(), - wantOutcome: OutcomeClean, + wantOutcome: OutcomeHealthy, }, { // Dense 30s-spaced polls throughout, so "good"'s own coverage @@ -466,9 +467,9 @@ func TestDecide_UnobservableRuleWinsOverEveryFavorableOutcome(t *testing.T) { wantOutcome: OutcomeRecovered, }, { - name: "skipped", + name: "paused", pausedAtStart: true, - wantOutcome: OutcomeSkipped, + wantOutcome: OutcomePaused, }, } @@ -502,7 +503,7 @@ func TestDecide_UnobservableRuleWinsOverEveryFavorableOutcome(t *testing.T) { } } require.Equal(t, tc.wantOutcome, gotGood) - require.Equal(t, OutcomeUnobservable, gotBroken) + require.Equal(t, OutcomeNotVerified, gotBroken) }) } } @@ -540,7 +541,7 @@ func TestDecide_RecoveredOutcomeOverriddenByItsOwnCoverageGap(t *testing.T) { res, err := decide(Header{StartedAt: from.Add(-time.Hour)}, polls, &sentinel, defs, rt, gt, pol) require.Error(t, err, "r1's own coverage gap must fail the run even though it recovered") require.Len(t, res.Verdicts, 1) - require.Equal(t, OutcomeUnobservable, res.Verdicts[0].Outcome, "never recovered") + require.Equal(t, OutcomeNotVerified, res.Verdicts[0].Outcome, "never recovered") require.False(t, res.Coverage["r1"].Proved) } @@ -562,7 +563,7 @@ func TestDecide_CleanWindowIsAPass(t *testing.T) { res, err := decide(Header{StartedAt: from.Add(-time.Hour)}, polls, &sentinel, defs, rt, gt, pol) require.NoError(t, err) require.Empty(t, res.Violations, "a pass is exactly len(Violations)==0 && err==nil") - require.Equal(t, OutcomeClean, res.Verdicts[0].Outcome) + require.Equal(t, OutcomeHealthy, res.Verdicts[0].Outcome) } // A pause and then an unpause inside the window, with an episode that would @@ -604,7 +605,7 @@ func TestDecide_PauseThenUnpauseWithHiddenEpisodeGivesUnobservableNotClean(t *te res, err := decide(Header{StartedAt: from.Add(-time.Hour)}, polls, &sentinel, defs, rt, gt, pol) require.Error(t, err, "the pause-then-unpause blind interval must fail closed") require.Len(t, res.Verdicts, 1) - require.Equal(t, OutcomeUnobservable, res.Verdicts[0].Outcome, "never clean") + require.Equal(t, OutcomeNotVerified, res.Verdicts[0].Outcome, "never clean") require.False(t, res.Coverage["r1"].Proved) } @@ -632,10 +633,10 @@ func TestDecide_SkippedOnlyShortfallProducesAViolationWithoutAnError(t *testing. sentinel := to res, err := decide(pausedHeader(from.Add(-time.Hour), "paused"), polls, &sentinel, defs, rt, gt, pol) - require.NoError(t, err, "a shortfall caused only by a skipped rule is exit 1, not exit 2") + require.NoError(t, err, "a shortfall caused only by a paused rule is exit 1, not exit 2") require.Len(t, res.Violations, 1) v := res.Violations[0] - require.Equal(t, OutcomeSkipped, v.Outcome) + require.Equal(t, OutcomePaused, v.Outcome) require.Equal(t, "paused", v.RuleUID) require.Equal(t, "Paused", v.Alert) require.NotEmpty(t, v.Note, @@ -665,7 +666,8 @@ func TestDecide_ExplicitMinObservedShortfallWithNoPausedRuleStillProducesAViolat require.NoError(t, err, "an unmet MinObserved is exit 1, never exit 2") require.Len(t, res.Violations, 2, "the shortfall (3-1=2) must surface directly rather than pass silently") for _, v := range res.Violations { - require.Equal(t, OutcomeSkipped, v.Outcome) + require.Equal(t, OutcomeNotCounted, v.Outcome, + "no resolved rule explains the deficit, so it must not be blamed on a paused one") } } @@ -740,7 +742,7 @@ func TestDecide_NodataIsANoteByDefault(t *testing.T) { // An instance whose true onset (ActiveAt) falls strictly inside the window — // even though the first poll that happens to observe it already shows it bad — -// must never be treated as preexisting. If it then clears, that is newly_bad +// must never be treated as preexisting. If it then clears, that is new_failure // (exit 1), not recovered (exit 0). func TestClassifyRule_OnsetBetweenFromAndFirstPollIsNewlyBadNotRecovered(t *testing.T) { from := time.Date(2026, 1, 1, 0, 0, 0, 0, time.UTC) @@ -757,10 +759,10 @@ func TestClassifyRule_OnsetBetweenFromAndFirstPollIsNewlyBadNotRecovered(t *test quietPoll("r1", to), } outcome, badFor, viols := classifyRule(def, polls, from, to, defaultBad, PreexistingFailUnlessRecovered) - require.Equal(t, OutcomeNewlyBad, outcome, + require.Equal(t, OutcomeNewFailure, outcome, "the onset is after `from`, so it is not preexisting even though the FIRST in-window poll already observes it bad") require.Len(t, viols, 1) - require.Equal(t, OutcomeNewlyBad, viols[0].Outcome) + require.Equal(t, OutcomeNewFailure, viols[0].Outcome) require.Equal(t, clearAt.Sub(onset), badFor, "BadFor must count from the true onset, not from `from`") } @@ -799,7 +801,7 @@ func TestClassifyRule_SkewTranslatesActiveAtAcrossTheWindowBoundary(t *testing.T // Grafana's clock reads 90s ahead of the runner's (skew = +90s). The // poll's raw GrafanaNow/ActiveAt both sit 90s past `from` in Grafana's // domain, but translate to exactly `from` in the runner domain — genuinely - // preexisting once translated, and wrongly "newly_bad" if the skew is + // preexisting once translated, and wrongly "new_failure" if the skew is // ignored. skew := 90 * time.Second rawActiveAt := from.Add(skew) @@ -813,7 +815,7 @@ func TestClassifyRule_SkewTranslatesActiveAtAcrossTheWindowBoundary(t *testing.T stillBad.LastEvaluation = to.Add(skew) outcome, badFor, _ := classifyRule(def, []Poll{poll, stillBad}, from, to, defaultBad, PreexistingFailUnlessRecovered) - require.Equal(t, OutcomePersistentlyBad, outcome, + require.Equal(t, OutcomeStillFailing, outcome, "a +90s skew must translate ActiveAt back to exactly `from`") require.Equal(t, to.Sub(from), badFor) } @@ -893,7 +895,7 @@ func TestClassifyRule_ClearedEventPastWindowEndClampsToWindowEnd(t *testing.T) { // reading of the upper boundary: an instance whose runner-domain onset lands // only slightly past windowEnd (to + transitionGrace) is reachable at all only // because inWindowPolls widens the boundary outward by the skew bound, so the -// gate cannot PROVE it belongs to the next window. It is charged as newly_bad — +// gate cannot PROVE it belongs to the next window. It is charged as new_failure — // with BadFor truncated to zero — rather than silently forgiven as clean. func TestClassifyRule_OnsetJustPastWindowEndIsNewlyBadNotClean(t *testing.T) { from := time.Date(2026, 1, 1, 0, 0, 0, 0, time.UTC) @@ -914,13 +916,13 @@ func TestClassifyRule_OnsetJustPastWindowEndIsNewlyBadNotClean(t *testing.T) { } outcome, badFor, viols := classifyRule(def, []Poll{poll}, from, windowEnd, defaultBad, PreexistingFailUnlessRecovered) - require.Equal(t, OutcomeNewlyBad, outcome, + require.Equal(t, OutcomeNewFailure, outcome, "an onset past windowEnd seen only via the skew bound must fail closed") require.Zero(t, badFor, "the zero-length episode must truncate to the window end") require.Len(t, viols, 1) } -// A clear after `to` gives persistently_bad. classifyRule filters +// A clear after `to` gives still_failing. classifyRule filters // its input to [from, windowEnd] itself (inWindowPolls), so a Cleared event // GENUINELY past windowEnd — well beyond any skew bound, unlike the clamp // case above — never reaches the timeline at all: the instance is still bad @@ -936,10 +938,10 @@ func TestClassifyRule_ClearAfterWindowEndIsPersistentlyBad(t *testing.T) { clearedPoll("r1", to.Add(time.Hour), key), // far past `to`, not a boundary case } outcome, badFor, viols := classifyRule(def, polls, from, to, defaultBad, PreexistingFailUnlessRecovered) - require.Equal(t, OutcomePersistentlyBad, outcome, "a clear outside the window must not read as a recovery") + require.Equal(t, OutcomeStillFailing, outcome, "a clear outside the window must not read as a recovery") require.Equal(t, to.Sub(from), badFor) require.Len(t, viols, 1) - require.Equal(t, OutcomePersistentlyBad, viols[0].Outcome) + require.Equal(t, OutcomeStillFailing, viols[0].Outcome) } // TestClassifyRule_CloseBeforeOpenClampsToZeroNotNegative pins the @@ -966,7 +968,7 @@ func TestClassifyRule_CloseBeforeOpenClampsToZeroNotNegative(t *testing.T) { quietPoll("r1", to), } outcome, badFor, viols := classifyRule(def, polls, from, to, defaultBad, PreexistingFailUnlessRecovered) - require.Equal(t, OutcomeNewlyBad, outcome) + require.Equal(t, OutcomeNewFailure, outcome) require.GreaterOrEqual(t, badFor, time.Duration(0), "a non-negative duration even though the closing poll's translated time landed before the opening poll's") require.Zero(t, badFor, "the clamp collapses the inverted span to a zero-length episode") @@ -990,3 +992,30 @@ func TestMergeDurations_OverlappingEpisodesCountOnce(t *testing.T) { func TestMergeDurations_Empty(t *testing.T) { require.Zero(t, mergeDurations(nil)) } + +// --- published vocabulary --- + +// Outcome strings are published JSON, but every other test compares against +// the constants, so a typo in a constant's literal would pass them all. Pin +// the literals themselves. +func TestOutcomeJSONVocabulary(t *testing.T) { + want := map[Outcome]string{ + OutcomeHealthy: "healthy", + OutcomeNewFailure: "new_failure", + OutcomeRecovered: "recovered", + OutcomeStillFailing: "still_failing", + OutcomeUnstable: "unstable", + OutcomePaused: "paused", + OutcomeNotVerified: "not_verified", + OutcomeNotCounted: "not_counted", + } + require.Len(t, want, 8, "every Outcome constant must be pinned here") + + for outcome, literal := range want { + t.Run(string(outcome), func(t *testing.T) { + raw, err := json.Marshal(outcome) + require.NoError(t, err) + require.Equal(t, `"`+literal+`"`, string(raw)) + }) + } +} diff --git a/grafana-alertcheck/internal/gate/coverage_test.go b/grafana-alertcheck/internal/gate/coverage_test.go index 7126372ed..226b46243 100644 --- a/grafana-alertcheck/internal/gate/coverage_test.go +++ b/grafana-alertcheck/internal/gate/coverage_test.go @@ -79,7 +79,7 @@ func TestProveCoverage_SentinelBeforeGraceIsUnobservable(t *testing.T) { dres, err := decide(Header{StartedAt: from.Add(-time.Hour)}, nil, &sentinel, defs, drt, gt, pol) require.Error(t, err, "a sentinel short of to+grace must fail the run") require.Len(t, dres.Verdicts, 1) - require.Equal(t, OutcomeUnobservable, dres.Verdicts[0].Outcome) + require.Equal(t, OutcomeNotVerified, dres.Verdicts[0].Outcome) } func TestProveCoverage_SentinelExactlyAtGraceIsFine(t *testing.T) { @@ -123,7 +123,7 @@ func TestProveCoverage_FromBeforeRecordIsUnobservable(t *testing.T) { dres, err := decide(Header{StartedAt: started}, nil, &sentinel, defs, drt, gt, pol) require.Error(t, err, "`from` before the recording started must fail the run") require.Len(t, dres.Verdicts, 1) - require.Equal(t, OutcomeUnobservable, dres.Verdicts[0].Outcome) + require.Equal(t, OutcomeNotVerified, dres.Verdicts[0].Outcome) } // The from-bounds check compares at whole-second granularity: a whole-second @@ -688,15 +688,15 @@ func TestProveCoverage_MultipleFailuresReasonIsFirstButAllNoted(t *testing.T) { "a later failure must still be recorded, not swallowed once Reason is already set") } -// --- Skipped rules --- +// --- Paused rules --- // A known limit of this function's contract, not a bug in it: a rule paused // BEFORE the window opened is never scheduled or polled (watch.go), so it // reaches proveCoverage with zero polls at all. proveCoverage has no notion of -// "skipped" — that classification belongs to the definitions +// "paused" — that classification belongs to the definitions // (LoggedRule.IsPaused / Definition.IsPaused), never to the polls — so it // reports the whole window as one big heartbeat_gap instead. decide is what -// reads skipped status from the header and never calls this function for such +// reads paused status from the header and never calls this function for such // a rule; this pins the behavior it relies on not reaching. func TestProveCoverage_SkippedRuleWithZeroPollsPinnedAsHeartbeatGap(t *testing.T) { from := time.Date(2026, 1, 1, 0, 0, 0, 0, time.UTC) diff --git a/grafana-alertcheck/internal/gate/terminal.go b/grafana-alertcheck/internal/gate/terminal.go index e41b9c923..c8b4197bb 100644 --- a/grafana-alertcheck/internal/gate/terminal.go +++ b/grafana-alertcheck/internal/gate/terminal.go @@ -9,10 +9,10 @@ import ( type TerminationKind string const ( - // TerminationViolation is a post-`from` bad onset (newly_bad or flapping). + // TerminationViolation is a post-`from` bad onset (new_failure or unstable). TerminationViolation TerminationKind = "violation" - // TerminationUnobservable is an inability that has already happened. - TerminationUnobservable TerminationKind = "unobservable" + // TerminationNotVerified is an inability that has already happened. + TerminationNotVerified TerminationKind = "not_verified" ) // Termination is why a fail-fast run stopped before the window closed. At is @@ -30,10 +30,10 @@ type Termination struct { // observed so far, treating at as the provisional end of the window. PURE. // // Only two conditions qualify, because only they can never become a pass: an -// inability that already happened, and a post-`from` bad onset (newly_bad or -// flapping). `recovered` forgives an observed bad state and is reserved for +// inability that already happened, and a post-`from` bad onset (new_failure or +// unstable). `recovered` forgives an observed bad state and is reserved for // bad-at-`from`, so a preexisting condition is deliberately not terminal. -// unobservable beats violation, as it does at the end of a full run. +// not_verified beats violation, as it does at the end of a full run. func terminalVerdict(h Header, polls []Poll, defs []Definition, rt map[string]RuleTimings, pol Policy, from, at time.Time) (Termination, bool) { @@ -55,17 +55,17 @@ func terminalVerdict(h Header, polls []Poll, defs []Definition, rt map[string]Ru } if cov.Unobservable { return Termination{ - Kind: TerminationUnobservable, + Kind: TerminationNotVerified, Alert: def.Title, RuleUID: def.UID, - Outcome: OutcomeUnobservable, + Outcome: OutcomeNotVerified, Reason: cov.Reason, At: at, }, true } outcome, _, _ := classifyRule(def, polls, from, at, badStates, pol.Preexisting) - if outcome == OutcomeNewlyBad || outcome == OutcomeFlapping { + if outcome == OutcomeNewFailure || outcome == OutcomeUnstable { if violation == nil { v := Termination{ Kind: TerminationViolation, diff --git a/grafana-alertcheck/internal/gate/terminal_test.go b/grafana-alertcheck/internal/gate/terminal_test.go index 7f46bfd46..a89b71437 100644 --- a/grafana-alertcheck/internal/gate/terminal_test.go +++ b/grafana-alertcheck/internal/gate/terminal_test.go @@ -1,6 +1,7 @@ package gate import ( + "encoding/json" "testing" "time" @@ -57,7 +58,7 @@ func TestTerminalVerdict(t *testing.T) { polls: func() []Poll { return badFromPolls(checkUID, from, at, from.Add(time.Minute)) }, want: true, wantKind: TerminationViolation, - wantOut: OutcomeNewlyBad, + wantOut: OutcomeNewFailure, }, { // Bad before `from` can still become `recovered` (a pass). @@ -72,17 +73,17 @@ func TestTerminalVerdict(t *testing.T) { want: false, }, { - name: "heartbeat gap is terminal unobservable", + name: "heartbeat gap is terminal not_verified", polls: func() []Poll { return append(denseHealthyPolls(checkUID, from, from.Add(time.Minute), checkPollEvery), quietPoll(checkUID, at)) }, want: true, - wantKind: TerminationUnobservable, + wantKind: TerminationNotVerified, wantReason: ReasonHeartbeatGap, }, { - name: "sustained health=error is terminal unobservable", + name: "sustained health=error is terminal not_verified", polls: func() []Poll { var out []Poll for ts := from; !ts.After(at); ts = ts.Add(checkPollEvery) { @@ -93,29 +94,29 @@ func TestTerminalVerdict(t *testing.T) { return out }, want: true, - wantKind: TerminationUnobservable, + wantKind: TerminationNotVerified, wantReason: ReasonHealthError, }, { - name: "in-window pause is terminal unobservable", + name: "in-window pause is terminal not_verified", polls: func() []Poll { out := denseHealthyPolls(checkUID, from, at, checkPollEvery) out = append(out, Poll{RuleUID: checkUID, GrafanaNow: from.Add(time.Minute), Found: true, Health: "ok", IsPaused: true}) return out }, want: true, - wantKind: TerminationUnobservable, + wantKind: TerminationNotVerified, wantReason: ReasonPausedInWindow, }, { - name: "absent rule is terminal unobservable", + name: "absent rule is terminal not_verified", polls: func() []Poll { out := denseHealthyPolls(checkUID, from, at, checkPollEvery) out = append(out, Poll{RuleUID: checkUID, GrafanaNow: from.Add(time.Minute)}) return out }, want: true, - wantKind: TerminationUnobservable, + wantKind: TerminationNotVerified, wantReason: ReasonRuleAbsent, }, } @@ -158,7 +159,7 @@ func TestTerminalVerdictNodataIsTerminalOnlyWhenConfigured(t *testing.T) { term, ok := terminalVerdict(terminalHeader(from), polls, []Definition{def}, rt, Policy{From: from, To: at, NodataIsUnobservable: true}, from, at) require.True(t, ok) - require.Equal(t, TerminationUnobservable, term.Kind) + require.Equal(t, TerminationNotVerified, term.Kind) require.Equal(t, ReasonNodata, term.Reason) } @@ -180,7 +181,25 @@ func TestTerminalVerdictUnobservableBeatsViolation(t *testing.T) { term, ok := terminalVerdict(terminalHeader(from), polls, []Definition{violating, gapped}, rt, Policy{From: from, To: at}, from, at) require.True(t, ok) - require.Equal(t, TerminationUnobservable, term.Kind) + require.Equal(t, TerminationNotVerified, term.Kind) require.Equal(t, "rule-two", term.RuleUID) require.Equal(t, ReasonHeartbeatGap, term.Reason) } + +// Termination kinds are published JSON too, and the assertions above compare +// against the constants. Pin the literals. +func TestTerminationKindJSONVocabulary(t *testing.T) { + want := map[TerminationKind]string{ + TerminationViolation: "violation", + TerminationNotVerified: "not_verified", + } + require.Len(t, want, 2, "every TerminationKind constant must be pinned here") + + for kind, literal := range want { + t.Run(string(kind), func(t *testing.T) { + raw, err := json.Marshal(kind) + require.NoError(t, err) + require.Equal(t, `"`+literal+`"`, string(raw)) + }) + } +}