diff --git a/grafana-alertcheck/.changeset/v0.1.3.md b/grafana-alertcheck/.changeset/v0.1.3.md new file mode 100644 index 000000000..66b3aa5fc --- /dev/null +++ b/grafana-alertcheck/.changeset/v0.1.3.md @@ -0,0 +1,2 @@ +- The `check` output now speaks plain words. Outcome values are renamed: `clean` → `healthy`, `newly_bad` → `new_failure`, `persistently_bad` → `still_failing`, `flapping` → `unstable`, `skipped` → `paused`, `unobservable` → `not_verified`, and the early-exit `terminated_early.kind` follows the same rename. A `--min-observed` deficit that no rule explains is now reported as `not_counted` instead of being blamed on a paused rule. This changes the `--output json` vocabulary and anything downstream of it, including the action's `outcomes` output. +- The human table dropped its internal column names. `RESULTS` is now `ALERT`/`VERDICT`/`BROKEN FOR`/`CHECKED EVERY`/`WINDOW COVERED`/`DETAILS`; `VIOLATIONS` uses `GRAFANA STATE`/`GRAFANA HEALTH` and a single-word `INSTANCES` column (the previous `INSTANCE COUNT` header read as two columns, one of them empty); `THRESHOLDS` became `LIMITS USED`, with limits named in plain words and explained by a legend under the table. The footer now spells out the extra observation time, the evaluation wait and the clock difference from Grafana. diff --git a/grafana-alertcheck/cmd/table.go b/grafana-alertcheck/cmd/table.go index 81d750042..085444b88 100644 --- a/grafana-alertcheck/cmd/table.go +++ b/grafana-alertcheck/cmd/table.go @@ -11,22 +11,29 @@ import ( "github.com/smartcontractkit/chainlink-testing-framework/grafana-alertcheck/internal/gate" ) +// limitsLegend explains each LIMITS USED column in one plain sentence, so the +// table needs no documentation lookup. +const limitsLegend = ` max gap without check — the longest gap between two checks we accept before we say the alert was not watched. + query failing for — how long Grafana may keep failing to run the alert's query before we stop trusting its state. + no evaluation for — how long Grafana may go without evaluating the alert before we stop trusting its state.` + // renderTable is the human table. It always writes to the writer it is given, // which the caller (runCheck) always points at stderr — stdout is reserved for // the machine-readable --output json. // -// Three titled tables, in order (the name column is RULE in all of them — one -// row is one resolved alert rule, never a firing instance): +// Three titled tables, in order (the ALERT column is one resolved alert rule, +// never a firing instance): +// +// 1. RESULTS, one line per rule: verdict, time broken, check cadence, +// whether the window was observed, and any notes; +// 2. VIOLATIONS, one line per distinct violation, with the raw Grafana state +// and health and the number of instances it stands for; +// 3. LIMITS USED, the coverage thresholds that answer "why" on exit 2. Each +// column is named in plain words and explained by the legend below it, so +// the table needs no documentation lookup. // -// 1. RESULTS, one line per rule: outcome, BadFor, pollEvery, proved-or-not -// with the largest gap; -// 2. VIOLATIONS, one line per distinct violation. -// 3. THRESHOLDS, the numbers that answer "why" on exit 2: each non-skipped -// rule's maxGap/healthGrace/evalStaleAfter, followed by the global -// transitionGrace and drainTimeout, and the largest measured clock skew -// alongside its own error bound (RTT/2) — SkewHardLimit is a separate, -// fixed input threshold and is reported next to it, never as if it were -// that bound. +// The global footer then reports the extra observation time, the drain limit +// and the largest measured clock difference, also in plain words. func renderTable(w io.Writer, res gate.Result) error { alertOf := make(map[string]string, len(res.Verdicts)) for _, v := range res.Verdicts { @@ -52,7 +59,7 @@ func renderTable(w io.Writer, res gate.Result) error { fmt.Fprintln(w, "RESULTS") tw := tabwriter.NewWriter(w, 0, 4, 2, ' ', 0) - fmt.Fprintln(tw, "RULE\tOUTCOME\tBADFOR\tPOLLEVERY\tPROVED\tNOTE") + fmt.Fprintln(tw, "ALERT\tVERDICT\tBROKEN FOR\tCHECKED EVERY\tWINDOW COVERED\tDETAILS") for _, v := range sortedVerdicts(res.Verdicts) { fmt.Fprintf(tw, "%s\t%s\t%s\t%s\t%s\t%s\n", v.Alert, v.Outcome, v.BadFor.Round(time.Second), v.PollEvery.Round(time.Second), @@ -65,7 +72,7 @@ func renderTable(w io.Writer, res gate.Result) error { if len(res.Violations) > 0 { fmt.Fprintln(w, "\nVIOLATIONS") vtw := tabwriter.NewWriter(w, 0, 4, 2, ' ', 0) - fmt.Fprintln(vtw, "RULE\tOUTCOME\tSTATE\tHEALTH\tINSTANCE COUNT\tNOTE") + fmt.Fprintln(vtw, "ALERT\tVERDICT\tGRAFANA STATE\tGRAFANA HEALTH\tINSTANCES\tDETAILS") for _, g := range groupedViolations(res.Violations) { fmt.Fprintf(vtw, "%s\t%s\t%s\t%s\t%s\t%s\n", alertLabel(g.v, alertOf), g.v.Outcome, g.v.State, g.v.Health, instanceCount(g), g.v.Note) } @@ -74,14 +81,13 @@ func renderTable(w io.Writer, res gate.Result) error { } } - // The per-rule thresholds answer "why" on exit 2: a table, not the prose - // "rule NAME: maxGap=... healthGrace=... evalStaleAfter=..." that repeated - // the rule name a fourth time. It is separated from the result above by a - // blank line. + // The per-rule limits answer "why" on exit 2. The columns are spelled out + // and explained by limitsLegend right below, so an operator does not have + // to look anything up. fmt.Fprintln(w) - fmt.Fprintln(w, "THRESHOLDS") + fmt.Fprintln(w, "LIMITS USED") ttw := tabwriter.NewWriter(w, 0, 4, 2, ' ', 0) - fmt.Fprintln(ttw, "RULE\tMAXGAP\tHEALTHGRACE\tEVALSTALEAFTER") + fmt.Fprintln(ttw, "ALERT\tMAX GAP WITHOUT CHECK\tQUERY FAILING FOR\tNO EVALUATION FOR") for _, uid := range sortedThresholdUIDs(res.Thresholds, alertOf) { t := res.Thresholds[uid] fmt.Fprintf(ttw, "%s\t%s\t%s\t%s\n", @@ -90,11 +96,17 @@ func renderTable(w io.Writer, res gate.Result) error { if err := ttw.Flush(); err != nil { return fmt.Errorf("render table: %w", err) } + fmt.Fprintln(w, limitsLegend) fmt.Fprintln(w) - fmt.Fprintf(w, "global: transitionGrace=%s (source: %s) drainTimeout=%s\n", - res.Global.TransitionGrace, res.Global.GraceSource, res.Global.DrainTimeout) - fmt.Fprintf(w, "largest measured clock skew: %s (bound ±%s, hard limit %s), grafana %s\n", + if res.Global.TransitionGrace > 0 { + fmt.Fprintf(w, "extra watching after your window: +%s — so an alert that only starts firing at the end is still caught (slowest: %s)\n", + res.Global.TransitionGrace, res.Global.GraceSource) + } else { + fmt.Fprintln(w, "extra watching after your window: none") + } + fmt.Fprintf(w, "max wait for all alerts to finish evaluating: %s\n", res.Global.DrainTimeout) + fmt.Fprintf(w, "clock difference from Grafana: %s, accurate to ±%s (checks fail above %s); Grafana %s\n", res.ClockSkew.Round(time.Millisecond), res.ClockSkewBound.Round(time.Millisecond), gate.SkewHardLimit, res.GrafanaVersion) // The verdict — the single number a terminal operator reads last — sits on @@ -104,8 +116,8 @@ func renderTable(w io.Writer, res gate.Result) error { return nil } -// violationsLabel colours the "violations: N" prefix of the footer: green for a -// clean run, red otherwise. The rest of the line is written uncoloured. +// violationsLabel colours the "violations: N" prefix of the footer: green when +// there are none, red otherwise. The rest of the line is written uncoloured. func violationsLabel(n int, enabled bool) string { s := fmt.Sprintf("violations: %d", n) if !enabled { @@ -117,10 +129,10 @@ func violationsLabel(n int, enabled bool) string { return ansiRed + s + ansiReset } -// provedLabel is the table's PROVED column: "yes" for a clean coverage -// proof, "no" with the reason and largest gap for an unobservable rule, and -// "-" for a rule decide never asked proveCoverage about at all (skipped — -// paused before the window opened). +// provedLabel is the table's WINDOW COVERED column: "yes" for a fully +// observed window, "no" with the reason and largest gap for a not-verified +// rule, and "-" for a rule decide never asked proveCoverage about at all +// (paused before the window opened). func provedLabel(cov gate.CoverageResult) string { if cov.Reason == "" && !cov.Unobservable && !cov.Proved { return "-" @@ -192,8 +204,11 @@ func sameRendered(a, b gate.Violation) bool { return violationSignature(a) == violationSignature(b) } +// instanceCount is the INSTANCES column: how many alert instances one grouped +// violation row stands for. Paused and not-counted rows stand for no instance +// at all, so they render "-". func instanceCount(g violationGroup) string { - if g.v.Outcome == gate.OutcomeSkipped { + if g.v.Outcome == gate.OutcomePaused || g.v.Outcome == gate.OutcomeNotCounted { return "-" } return strconv.Itoa(g.n) diff --git a/grafana-alertcheck/cmd/table_test.go b/grafana-alertcheck/cmd/table_test.go index b050f0d30..acf987423 100644 --- a/grafana-alertcheck/cmd/table_test.go +++ b/grafana-alertcheck/cmd/table_test.go @@ -11,8 +11,8 @@ import ( ) // The golden table test: a fixed Result renders a deterministic, ordered rule -// table, a violations section and a footer carrying the per-rule and global -// thresholds plus the skew and its bound — with no live Check involved. +// table, a violations section and a footer carrying the per-rule limits and +// globals in plain words — with no live Check involved. func TestRenderTable(t *testing.T) { gapAt := time.Date(2026, 1, 1, 12, 0, 0, 0, time.UTC) res := gate.Result{ @@ -20,15 +20,15 @@ func TestRenderTable(t *testing.T) { ClockSkew: 1500 * time.Millisecond, ClockSkewBound: 250 * time.Millisecond, Verdicts: []gate.RuleVerdict{ - {Alert: "Zebra Alert", RuleUID: "uid-z", Outcome: gate.OutcomeClean, PollEvery: 30 * time.Second}, - {Alert: "Ape Alert", RuleUID: "uid-a", Outcome: gate.OutcomeUnobservable, + {Alert: "Zebra Alert", RuleUID: "uid-z", Outcome: gate.OutcomeHealthy, PollEvery: 30 * time.Second}, + {Alert: "Ape Alert", RuleUID: "uid-a", Outcome: gate.OutcomeNotVerified, PollEvery: 30 * time.Second, Note: "gap of 5m0s starting at 2026-01-01T12:00:00Z exceeds maxGap 1m0s"}, - {Alert: "Paused Alert", RuleUID: "uid-p", Outcome: gate.OutcomeSkipped, + {Alert: "Paused Alert", RuleUID: "uid-p", Outcome: gate.OutcomePaused, Note: "paused before the window opened; counts against --min-observed unless --allow-paused is set"}, }, Violations: []gate.Violation{ - {Alert: "Ape Alert", RuleUID: "uid-a", Outcome: gate.OutcomeUnobservable, State: gate.StateFiring, Health: "error", Note: "unobservable"}, - {Alert: "Paused Alert", RuleUID: "uid-p", Outcome: gate.OutcomeSkipped, + {Alert: "Ape Alert", RuleUID: "uid-a", Outcome: gate.OutcomeNotVerified, State: gate.StateFiring, Health: "error", Note: "not verified"}, + {Alert: "Paused Alert", RuleUID: "uid-p", Outcome: gate.OutcomePaused, Note: "paused before the window opened; counts against --min-observed unless --allow-paused is set"}, }, Coverage: map[string]gate.CoverageResult{ @@ -50,42 +50,49 @@ func TestRenderTable(t *testing.T) { require.NoError(t, renderTable(&buf, res)) out := buf.String() - // Rule table: Ape sorts before Zebra sorts before... Paused is skipped and - // carries no coverage entry, so it renders "-" for PROVED. + // The rule table: Ape sorts before Zebra sorts before... Paused carries no + // coverage entry, so it renders "-" for WINDOW COVERED. + require.Contains(t, out, "ALERT") require.Contains(t, out, "Ape Alert") - require.Contains(t, out, "unobservable") + require.Contains(t, out, "not_verified") require.Contains(t, out, "heartbeat_gap") require.Contains(t, out, "largest gap 5m0s") require.Contains(t, out, "Zebra Alert") - require.Contains(t, out, "clean") + require.Contains(t, out, "healthy") + require.Contains(t, out, "WINDOW COVERED") // The violations section must show up even without --output json, and must - // carry the --allow-paused hint text verbatim. + // carry the --allow-paused hint text verbatim. INSTANCES is a single word + // so the count under it cannot read as a second, empty column. require.Contains(t, out, "VIOLATIONS") require.Contains(t, out, "--allow-paused") - require.Contains(t, out, "STATE") - require.Contains(t, out, "HEALTH") - require.Contains(t, out, "INSTANCE COUNT") + require.Contains(t, out, "GRAFANA STATE") + require.Contains(t, out, "GRAFANA HEALTH") + require.Contains(t, out, "INSTANCES") require.Contains(t, out, string(gate.StateFiring)) require.Contains(t, out, "error") - // The footer: per-rule thresholds are a table (RULE/MAXGAP/HEALTHGRACE/ - // EVALSTALEAFTER) rather than prose, followed by the global thresholds and - // the violations count with the skew and its own bound rather than the - // fixed hard limit. - require.Contains(t, out, "MAXGAP") - require.Contains(t, out, "HEALTHGRACE") - require.Contains(t, out, "EVALSTALEAFTER") - require.Contains(t, out, "global: transitionGrace=5m0s (source: Ape Alert (for=5m)) drainTimeout=2m0s") - require.Contains(t, out, "largest measured clock skew: 1.5s (bound ±250ms, hard limit 1m0s)") + // The limits table names each threshold in plain words and explains it + // right below, so an operator does not have to consult the docs. + require.Contains(t, out, "LIMITS USED") + require.Contains(t, out, "MAX GAP WITHOUT CHECK") + require.Contains(t, out, "QUERY FAILING FOR") + require.Contains(t, out, "NO EVALUATION FOR") + require.Contains(t, out, "the longest gap between two checks") + require.Contains(t, out, "without evaluating the alert") + + // The global footer in plain words. + require.Contains(t, out, "extra watching after your window: +5m0s") + require.Contains(t, out, "slowest: Ape Alert (for=5m)") + require.Contains(t, out, "max wait for all alerts to finish evaluating: 2m0s") + require.Contains(t, out, "clock difference from Grafana: 1.5s, accurate to ±250ms (checks fail above 1m0s); Grafana 13.1.0") require.Contains(t, out, "violations: 2") - require.Contains(t, out, "13.1.0") } // The "-" case: a rule decide never asked proveCoverage about (paused before // the window opened) has an empty CoverageResult and must not be reported as -// either proved or unobservable. -func TestProvedLabel_Skipped(t *testing.T) { +// either covered or not verified. +func TestProvedLabel_Paused(t *testing.T) { require.Equal(t, "-", provedLabel(gate.CoverageResult{})) } @@ -93,12 +100,12 @@ func TestProvedLabel_Skipped(t *testing.T) { // rendered signature, each with a count. func TestGroupedViolations(t *testing.T) { in := []gate.Violation{ - {Alert: "OCR2 Consensus failure", RuleUID: "uid-o", Outcome: gate.OutcomePersistentlyBad, State: gate.StateFiring, Health: "ok"}, - {Alert: "OCR2 Consensus failure", RuleUID: "uid-o", Outcome: gate.OutcomePersistentlyBad, State: gate.StateFiring, Health: "ok"}, - {Alert: "OCR2 Consensus failure", RuleUID: "uid-o", Outcome: gate.OutcomePersistentlyBad, State: gate.StateFiring, Health: "ok"}, - {Alert: "OCR2 Consensus failure", RuleUID: "uid-o", Outcome: gate.OutcomePersistentlyBad, State: gate.StateFiring, Health: "error"}, - {Alert: "Other Alert", RuleUID: "uid-p", Outcome: gate.OutcomeNewlyBad, State: gate.StateFiring, Health: "ok", Note: "x"}, - {Alert: "Other Alert", RuleUID: "uid-p", Outcome: gate.OutcomeNewlyBad, State: gate.StateFiring, Health: "ok", Note: "x"}, + {Alert: "OCR2 Consensus failure", RuleUID: "uid-o", Outcome: gate.OutcomeStillFailing, State: gate.StateFiring, Health: "ok"}, + {Alert: "OCR2 Consensus failure", RuleUID: "uid-o", Outcome: gate.OutcomeStillFailing, State: gate.StateFiring, Health: "ok"}, + {Alert: "OCR2 Consensus failure", RuleUID: "uid-o", Outcome: gate.OutcomeStillFailing, State: gate.StateFiring, Health: "ok"}, + {Alert: "OCR2 Consensus failure", RuleUID: "uid-o", Outcome: gate.OutcomeStillFailing, State: gate.StateFiring, Health: "error"}, + {Alert: "Other Alert", RuleUID: "uid-p", Outcome: gate.OutcomeNewFailure, State: gate.StateFiring, Health: "ok", Note: "x"}, + {Alert: "Other Alert", RuleUID: "uid-p", Outcome: gate.OutcomeNewFailure, State: gate.StateFiring, Health: "ok", Note: "x"}, } got := groupedViolations(in) @@ -108,7 +115,7 @@ func TestGroupedViolations(t *testing.T) { for _, g := range got { counts[g.v.Health+"|"+string(g.v.Outcome)] = g.n } - require.Equal(t, 3, counts["ok|"+string(gate.OutcomePersistentlyBad)]) - require.Equal(t, 1, counts["error|"+string(gate.OutcomePersistentlyBad)]) - require.Equal(t, 2, counts["ok|"+string(gate.OutcomeNewlyBad)]) + require.Equal(t, 3, counts["ok|"+string(gate.OutcomeStillFailing)]) + require.Equal(t, 1, counts["error|"+string(gate.OutcomeStillFailing)]) + require.Equal(t, 2, counts["ok|"+string(gate.OutcomeNewFailure)]) } diff --git a/grafana-alertcheck/docs/architecture.md b/grafana-alertcheck/docs/architecture.md index 9c8b2aa38..fa2971178 100644 --- a/grafana-alertcheck/docs/architecture.md +++ b/grafana-alertcheck/docs/architecture.md @@ -15,11 +15,11 @@ This page documents the invariants and seams a maintainer must not break. It exi The gate must fail if it cannot get an answer. Every rule below is a specific instance of that: - **An error is never a pass.** A pass is exactly `len(Violations) == 0 && err == nil`. Every error path leaves `err` non-nil, and the CLI maps that to exit `2` unconditionally. -- **Inability beats violation.** Any `unobservable` rule is exit `2`, even alongside a real violation found first. +- **Inability beats violation.** Any `not_verified` rule is exit `2`, even alongside a real violation found first. - **Absent never means normal.** An instance that leaves the bad set is looked up in the *same* response: present as `normal` → cleared; absent (or `MissingSeries`) → vanished (a discontinuity, not a recovery). - **Staleness is absolute.** `grafana_now − lastEvaluation` is compared against a threshold, never "did it increase since the last poll" — a delta check reports stale on ~half the polls of a healthy rule (we poll at half of `intervalSeconds` of each rule). - **`grafana_now` is the response `Date` header.** Never the runner clock, in any comparison against a Grafana timestamp. -- **An early exit can never be a pass.** `check` may stop collecting before `to + transitionGrace` (fail-fast), but only on a *monotone* terminal verdict: an inability that has already happened, or a post-`from` bad onset (which the full classifier would call `newly_bad`/`flapping`). The one outcome that forgives an observed bad state, `recovered`, is reserved for bad-at-`from`, so a preexisting condition is never terminal. `--no-fail-fast` removes the guard entirely. +- **An early exit can never be a pass.** `check` may stop collecting before `to + transitionGrace` (fail-fast), but only on a *monotone* terminal verdict: an inability that has already happened, or a post-`from` bad onset (which the full classifier would call `new_failure`/`unstable`). The one outcome that forgives an observed bad state, `recovered`, is reserved for bad-at-`from`, so a preexisting condition is never terminal. `--no-fail-fast` removes the guard entirely. - **No replay.** No run-id key, no artifact download, no state between attempts. A retry is a new piece of work and observation. ## The pure-function seam diff --git a/grafana-alertcheck/docs/how-alerts-are-evaluated.md b/grafana-alertcheck/docs/how-alerts-are-evaluated.md index 46fa2142d..58ac89c14 100644 --- a/grafana-alertcheck/docs/how-alerts-are-evaluated.md +++ b/grafana-alertcheck/docs/how-alerts-are-evaluated.md @@ -32,13 +32,15 @@ For each instance the gate builds a timeline of bad spans over `[from, to]`, the | Outcome | Shape | Exit | | ------- | ----- | ---- | -| `clean` | Good throughout, observed throughout | → 0 | -| `newly_bad` | Entered a bad state **inside** the window | → 1 | -| `persistently_bad` | Bad at `from`, still bad at `to` | → 1 | +| `healthy` | Good throughout, observed throughout | → 0 | +| `new_failure` | Entered a bad state **inside** the window | → 1 | +| `still_failing` | Bad at `from`, still bad at `to` | → 1 | | `recovered` | Bad at `from`, cleared before `to`, stayed clear | → 0 | -| `flapping` | Cleared, then became bad again | → 1 | -| `skipped` | Paused **before** the window opened | reported, not observable | -| `unobservable` | Coverage gap / sustained `health=error` / stale / absent | → 2 | +| `unstable` | Cleared, then became bad again | → 1 | +| `paused` | Paused **before** the window opened | counts against `--min-observed` unless `--allow-paused` | +| `not_verified` | The window could not be observed: a gap, sustained `health=error`, a stale evaluation, or an absent rule | → 2 | + +A `--min-observed` deficit that no rule explains is reported as `not_counted`: it is not a verdict on any alert. `recovered` has **no deadline** — an alert that clears at minute 58 of a 60-minute window still passes. The total bad time is reported as `BadFor`. @@ -57,11 +59,11 @@ When an instance leaves the bad set, the gate looks it up **in the same response - Present as `normal` → `cleared` (a real recovery). - Absent, or present as `normal (MissingSeries)` → `vanished` (a discontinuity, **not** a recovery). -A vanished instance that was bad stays `persistently_bad`. A metric that stops being emitted is not evidence of health — this is deliberate and can surprise users whose fix is to remove a metric rather than drive it to a good value. +A vanished instance that was bad stays `still_failing`. A metric that stops being emitted is not evidence of health — this is deliberate and can surprise users whose fix is to remove a metric rather than drive it to a good value. ## Coverage proof -Before classifying, `check` must **prove** continuous coverage of `[from, to]` for each alert. Nine checks run; any failure makes the rule `unobservable`: +Before classifying, `check` must **prove** continuous coverage of `[from, to]` for each alert. Nine checks run; any failure makes the rule `not_verified`: 1. **Sentinel** — a clean recorder stop, timestamped at or after `to + transitionGrace`. A recorder that died mid-window looks exactly like a coverage gap and is one. 2. **`from` bounds** — `from` earlier than the recording start is unprovable. @@ -69,20 +71,20 @@ Before classifying, `check` must **prove** continuous coverage of `[from, to]` f 4. **`health=error`** — a contiguous run longer than `healthGrace` consumes coverage; a short blip is a note. 5. **`health=nodata`** — a note, never fatal (unless `--nodata-is-unobservable`). 6. **Liveness** — `grafana_now − lastEvaluation` must not exceed `evalStaleAfter`. This is an **absolute** check, never a "did it increase since the last poll" delta. -7. **In-window pause** — a poll reporting `isPaused` mid-window is `unobservable` (the primary pause detector). +7. **In-window pause** — a poll reporting `isPaused` mid-window is `not_verified` (the primary pause detector). 8. **Rule absent** — an authoritative `2xx` with no matching rule. 9. **`KeepLast`** — a note naming a stale-state blind spot. ## Health: `error` vs `nodata` -- `health=error` means the query **failed** — a malfunction. Sustained past `healthGrace`, it makes the rule `unobservable`. +- `health=error` means the query **failed** — a malfunction. Sustained past `healthGrace`, it makes the rule `not_verified`. - `health=nodata` means the query **ran and returned no series** — indistinguishable from a quiet system. It is not fatal by default; most of a fleet runs `no_data_state: OK`. ## Early exit (fail-fast) `check` does not have to wait for the whole window to know the run has failed. As soon as it observes a condition that cannot become a pass, it stops and classifies the sub-window it did see: -- a **post-`from` bad onset** — the full classifier would call it `newly_bad` (or `flapping`), which fails whether or not it later clears; or +- a **post-`from` bad onset** — the full classifier would call it `new_failure` (or `unstable`), which fails whether or not it later clears; or - an **inability** — a heartbeat gap, a sustained `health=error` run, a stale evaluation, an in-window pause, or an absent rule. A preexisting bad instance is deliberately **not** terminal: if it clears before `to` the full run would call it `recovered`, which passes. @@ -91,6 +93,6 @@ Fail-fast is on by default and always preserves the failure: an early run can ex ## The drain wait and `transitionGrace` -A condition that arises just before `to` becomes `firing` only at the first evaluation after its `for` elapses. `transitionGrace` (derived from the watched rules' `for` values) extends the classification bound past `to` so such a surfacing condition is caught. After collection, a **drain wait** polls until each rule has evaluated through `to + transitionGrace` (bounded by `drainTimeout`); a rule that never does is `unobservable`. +A condition that arises just before `to` becomes `firing` only at the first evaluation after its `for` elapses. `transitionGrace` (derived from the watched rules' `for` values) extends the classification bound past `to` so such a surfacing condition is caught. After collection, a **drain wait** polls until each rule has evaluated through `to + transitionGrace` (bounded by `drainTimeout`); a rule that never does is `not_verified`. Run time = `(to − from) + transitionGrace + drainTimeout`. This is printed at start, and the grace is warned about when it exceeds a quarter of the window — the window may be too short for the alert's `for`. diff --git a/grafana-alertcheck/docs/reference/cli.md b/grafana-alertcheck/docs/reference/cli.md index f059cc83c..312ea4860 100644 --- a/grafana-alertcheck/docs/reference/cli.md +++ b/grafana-alertcheck/docs/reference/cli.md @@ -102,7 +102,7 @@ Datasource-managed and recording rules are refused with a specific error. A name ## Output and exit codes -The human table goes to **stderr**: `RESULTS` (one row per rule), `VIOLATIONS` (one per distinct rule/outcome/state/health/note signature, with a `COUNT` of the instances it stands for — instance identity is only in the JSON), and `THRESHOLDS` (each rule's `maxGap`/`healthGrace`/`evalStaleAfter` plus global `transitionGrace`/`drainTimeout` and the largest measured clock skew). `--output json` writes the result to stdout. +The human table goes to **stderr**: `RESULTS` (one row per rule, with the verdict, time broken, check cadence and whether the window was observed), `VIOLATIONS` (one per distinct rule/verdict/state/health/note signature, with an `INSTANCES` count of the instances it stands for — instance identity is only in the JSON), and `LIMITS USED` (each rule's observation limits in plain words, explained by a legend under the table, plus the extra observation time, the evaluation wait and the largest measured clock difference). The JSON outcome values are `healthy`, `new_failure`, `still_failing`, `recovered`, `unstable`, `paused`, `not_verified` and the synthetic `not_counted`. `--output json` writes the result to stdout. | Code | Meaning | | ---- | ------- | diff --git a/grafana-alertcheck/docs/reference/log-format.md b/grafana-alertcheck/docs/reference/log-format.md index 9dc27c29c..0bf1830af 100644 --- a/grafana-alertcheck/docs/reference/log-format.md +++ b/grafana-alertcheck/docs/reference/log-format.md @@ -49,7 +49,7 @@ The header must be line 1, appear once, and carry `schema_version` `1` (any othe ``` - `url` and `rules` are the log's identity — `check` validates them against the current environment and a fresh ruler read. -- `is_paused` records the pause state at record start (the moment `skipped` means). +- `is_paused` records the pause state at record start (the moment `paused` means). - `poll_every_seconds` is the cadence the recording **actually used** (after any `--poll-interval` override). `check` derives `maxGap` from it, never from `interval_seconds`. - `for_seconds`, `interval_seconds`, `no_data_state`, `exec_err_state` are forensic only — `check` re-resolves definitions and never reads them back. @@ -95,4 +95,4 @@ Instance keys are the JSON encoding of the labels map (with stable key order), s { "type": "stopped", "at": "2026-09-07T10:10:30Z" } ``` -`at` is the recorder's own stop time. `check` compares it against `to + transitionGrace`; absent or earlier is `unobservable` — never a pass. +`at` is the recorder's own stop time. `check` compares it against `to + transitionGrace`; absent or earlier is `not_verified` — never a pass. diff --git a/grafana-alertcheck/internal/gate/check.go b/grafana-alertcheck/internal/gate/check.go index 63bc1dab2..9faa8f1f1 100644 --- a/grafana-alertcheck/internal/gate/check.go +++ b/grafana-alertcheck/internal/gate/check.go @@ -685,13 +685,13 @@ type drainVerdict struct { // drainWait is the final liveness check: did each rule evaluate through the // end of the window? A rule that cannot answer within drainTimeout is -// unobservable, never a pass. It returns one verdict per rule it could not +// not_verified, never a pass. It returns one verdict per rule it could not // clear (keyed by UID); an error only for a hard failure of the wait itself. // // Two kinds of rule are excluded up front because draining them could not // change a verdict: a rule the HEADER says was paused at the window open (the // header, not the late-resolved definitions — see Header.pausedAtStart), and a -// rule whose last poll says Found == false (already unobservable via rule_absent). +// rule whose last poll says Found == false (already not_verified via rule_absent). func drainWait(ctx context.Context, cfg Config, src Source, defs []Definition, pausedAtStart map[string]bool, rt map[string]RuleTimings, polls []Poll, windowEnd time.Time, timeout time.Duration) (map[string]drainVerdict, error) { @@ -862,17 +862,17 @@ func mergeDrainTimeouts(res Result, drained map[string]drainVerdict) (Result, er cov.Notes = append(cov.Notes, verdict.note) res.Coverage[uid] = cov - if res.Verdicts[i].Outcome != OutcomeUnobservable { + if res.Verdicts[i].Outcome != OutcomeNotVerified { names = append(names, fmt.Sprintf("%s (%s)", res.Verdicts[i].Alert, verdict.reason)) } - res.Verdicts[i].Outcome = OutcomeUnobservable + res.Verdicts[i].Outcome = OutcomeNotVerified res.Verdicts[i].Note = strings.Join(cov.Notes, "; ") } if len(names) == 0 { - // Every drained rule was already unobservable for an earlier reason, + // Every drained rule was already not_verified for an earlier reason, // so decide's own error already stops the run. Adding a second error // saying the same thing would only make the message longer. return res, nil } - return res, fmt.Errorf("gate: %d rule(s) unobservable at the drain wait: %s", len(names), strings.Join(names, "; ")) + return res, fmt.Errorf("gate: %d rule(s) not verified at the drain wait: %s", len(names), strings.Join(names, "; ")) } diff --git a/grafana-alertcheck/internal/gate/check_test.go b/grafana-alertcheck/internal/gate/check_test.go index 2ba8747b0..2dc6cb1a6 100644 --- a/grafana-alertcheck/internal/gate/check_test.go +++ b/grafana-alertcheck/internal/gate/check_test.go @@ -268,7 +268,7 @@ func TestCheckSingleStepCleanWindowPasses(t *testing.T) { // A pass is exactly this shape. require.Empty(t, res.Violations) require.Len(t, res.Verdicts, 1) - require.Equal(t, OutcomeClean, res.Verdicts[0].Outcome) + require.Equal(t, OutcomeHealthy, res.Verdicts[0].Outcome) cov := res.Coverage[checkUID] require.True(t, cov.Proved) require.False(t, cov.Unobservable) @@ -340,7 +340,7 @@ func TestCheckSingleStepContinuousHealthErrorIsUnobservable(t *testing.T) { res, err := check(context.Background(), cfg, src) require.Error(t, err, "continuous health=error must be unobservable") require.Len(t, res.Verdicts, 1) - require.Equal(t, OutcomeUnobservable, res.Verdicts[0].Outcome) + require.Equal(t, OutcomeNotVerified, res.Verdicts[0].Outcome) require.Equal(t, ReasonHealthError, res.Coverage[def.UID].Reason) } @@ -361,7 +361,7 @@ func TestCheckSingleStepFiringInstanceReportsWithoutExitingEarly(t *testing.T) { res, err := check(context.Background(), cfg, src) require.NoError(t, err, "a violation is exit 1, not an error") require.Len(t, res.Violations, 1) - require.Equal(t, OutcomePersistentlyBad, res.Violations[0].Outcome) + require.Equal(t, OutcomeStillFailing, res.Violations[0].Outcome) require.False(t, clock.Now().Before(cfg.To.Add(checkGrace)), "exited early; collection must run to to+grace") } @@ -387,12 +387,12 @@ func TestCheckSingleStepNewOnsetExitsEarlyByDefault(t *testing.T) { res, err := check(context.Background(), cfg, src) require.NoError(t, err, "a violation is exit 1, not an error") require.Len(t, res.Violations, 1) - require.Equal(t, OutcomeNewlyBad, res.Violations[0].Outcome) + require.Equal(t, OutcomeNewFailure, res.Violations[0].Outcome) require.True(t, clock.Now().Before(cfg.To.Add(checkGrace)), "the runner was released only after to+grace; fail-fast did not fire") require.NotNil(t, res.TerminatedEarly) require.Equal(t, TerminationViolation, res.TerminatedEarly.Kind) - require.Equal(t, OutcomeNewlyBad, res.TerminatedEarly.Outcome) + require.Equal(t, OutcomeNewFailure, res.TerminatedEarly.Outcome) require.True(t, res.To.Equal(cfg.To), "the requested window is still reported") require.Contains(t, notesOf(cfg), "fail-fast") } @@ -414,7 +414,7 @@ func TestCheckSingleStepPreexistingBadDoesNotExitEarly(t *testing.T) { res, err := check(context.Background(), cfg, src) require.NoError(t, err) require.Len(t, res.Violations, 1) - require.Equal(t, OutcomePersistentlyBad, res.Violations[0].Outcome) + require.Equal(t, OutcomeStillFailing, res.Violations[0].Outcome) require.Nil(t, res.TerminatedEarly, "a preexisting condition can still recover, so it is not terminal") require.False(t, clock.Now().Before(cfg.To.Add(checkGrace)), "exited early; collection must run to to+grace for a preexisting bad instance") @@ -442,7 +442,7 @@ func TestCheckSingleStepNewOnsetNoFailFastRunsToTheEnd(t *testing.T) { res, err := check(context.Background(), cfg, src) require.NoError(t, err) require.Len(t, res.Violations, 1) - require.Equal(t, OutcomeNewlyBad, res.Violations[0].Outcome) + require.Equal(t, OutcomeNewFailure, res.Violations[0].Outcome) require.Nil(t, res.TerminatedEarly) require.False(t, clock.Now().Before(cfg.To.Add(checkGrace)), "with --no-fail-fast the loop must run to to+grace") @@ -650,7 +650,7 @@ func TestCheckRecorderModeCleanWindowPasses(t *testing.T) { require.NoError(t, err) require.Empty(t, res.Violations) require.Len(t, res.Verdicts, 1) - require.Equal(t, OutcomeClean, res.Verdicts[0].Outcome) + require.Equal(t, OutcomeHealthy, res.Verdicts[0].Outcome) require.Equal(t, "13.1.0", res.GrafanaVersion) // The collection loop still waited out to+transitionGrace even though the // recorder had already finished. @@ -701,7 +701,7 @@ func TestCheckRecorderModeExitsEarlyOnANewOnset(t *testing.T) { require.NoError(t, err) require.NotNil(t, res.TerminatedEarly) require.Equal(t, TerminationViolation, res.TerminatedEarly.Kind) - require.Equal(t, OutcomeNewlyBad, res.TerminatedEarly.Outcome) + require.Equal(t, OutcomeNewFailure, res.TerminatedEarly.Outcome) require.Len(t, res.Violations, 1) require.True(t, clock.Now().Before(cfg.To.Add(checkGrace)), "the runner was held to the window; fail-fast did not fire") @@ -721,7 +721,7 @@ func TestCheckRecorderModeNoFailFastRunsToTheEnd(t *testing.T) { require.NoError(t, err) require.Nil(t, res.TerminatedEarly) require.Len(t, res.Violations, 1) - require.Equal(t, OutcomeNewlyBad, res.Violations[0].Outcome) + require.Equal(t, OutcomeNewFailure, res.Violations[0].Outcome) require.False(t, clock.Now().Before(cfg.To.Add(checkGrace)), "with --no-fail-fast the loop must run to to+grace") } @@ -803,7 +803,7 @@ func TestCheckRecorderModeFromSameSecondAsStartedAtPasses(t *testing.T) { require.NoError(t, err) require.Empty(t, res.Violations) require.Len(t, res.Verdicts, 1) - require.Equal(t, OutcomeClean, res.Verdicts[0].Outcome) + require.Equal(t, OutcomeHealthy, res.Verdicts[0].Outcome) } // The coverage proof failed: a hole in the middle of the recording is not @@ -835,7 +835,7 @@ func TestCheckFailClosedOnCoverageGap(t *testing.T) { res, err := check(context.Background(), cfg, newCheckSource(nil)) require.Error(t, err, "the coverage gap to fail closed") require.Equal(t, ReasonHeartbeatGap, res.Coverage[checkUID].Reason) - require.Equal(t, OutcomeUnobservable, res.Verdicts[0].Outcome) + require.Equal(t, OutcomeNotVerified, res.Verdicts[0].Outcome) } // An episode fully between the deploy and the start of the check: recorder @@ -868,7 +868,7 @@ func TestCheckRecorderModeFindsAGapImmediatelyAfterTheDeploy(t *testing.T) { res, err := check(context.Background(), cfg, newCheckSource(nil)) require.Error(t, err, "a hole right after the deploy hides whatever happened there") require.Len(t, res.Verdicts, 1) - require.Equal(t, OutcomeUnobservable, res.Verdicts[0].Outcome, "never clean") + require.Equal(t, OutcomeNotVerified, res.Verdicts[0].Outcome, "never clean") } // The drain limit passed. The recording itself is clean, so this isolates the @@ -897,7 +897,7 @@ func TestCheckFailClosedOnDrainTimeout(t *testing.T) { res, err := check(context.Background(), cfg, src) require.Error(t, err, "the drain limit to fail closed") require.Equal(t, ReasonDrainTimeout, res.Coverage[checkUID].Reason) - require.Equal(t, OutcomeUnobservable, res.Verdicts[0].Outcome) + require.Equal(t, OutcomeNotVerified, res.Verdicts[0].Outcome) require.Contains(t, res.Verdicts[0].Note, "drain limit") require.GreaterOrEqual(t, clock.Now().Sub(windowEnd), checkDrainLimit, "the rule never evaluates through the window, so the drain wait must run its full limit") @@ -996,10 +996,10 @@ func pausedAfterWindowCheck(t *testing.T, allowPaused bool) (Result, Config, err func TestCheckPausingARuleAfterTheWindowDoesNotMakeItSkipped(t *testing.T) { res, _, err := pausedAfterWindowCheck(t, false) require.NoError(t, err) - require.Equal(t, OutcomeNewlyBad, res.Verdicts[0].Outcome, + require.Equal(t, OutcomeNewFailure, res.Verdicts[0].Outcome, "the rule was active for the whole window and fired inside it") require.Len(t, res.Violations, 1) - require.Equal(t, OutcomeNewlyBad, res.Violations[0].Outcome) + require.Equal(t, OutcomeNewFailure, res.Violations[0].Outcome) require.NotContains(t, res.Verdicts[0].Note, "paused before the window opened") } @@ -1047,7 +1047,7 @@ func TestCheckHeaderPausedRuleStaysSkipped(t *testing.T) { res, err := run(false) require.NoError(t, err, "a skipped rule is a known condition, not an inability") - require.Equal(t, OutcomeSkipped, res.Verdicts[0].Outcome) + require.Equal(t, OutcomePaused, res.Verdicts[0].Outcome) _, ok := res.Coverage[checkUID] require.False(t, ok, "a skipped rule has no coverage to prove") require.Len(t, res.Violations, 1, "the MinObserved shortfall") @@ -1214,7 +1214,7 @@ func TestCheckDeadPidWithNoSentinelIsUnobservable(t *testing.T) { res, err := check(context.Background(), cfg, newCheckSource(nil)) require.Error(t, err, "no sentinel means the recorder never proved it ran to the end") require.Len(t, res.Verdicts, 1) - require.Equal(t, OutcomeUnobservable, res.Verdicts[0].Outcome) + require.Equal(t, OutcomeNotVerified, res.Verdicts[0].Outcome) } // An incomplete last line gives exit 2. log_test.go's TestReadLogRejectsBadLogs @@ -1332,8 +1332,8 @@ func TestMergeDrainTimeoutsNamesEveryUnobservableRule(t *testing.T) { "b": {Unobservable: true, Reason: ReasonHeartbeatGap, Notes: []string{"rule \"B\": gap"}}, }, Verdicts: []RuleVerdict{ - {Alert: "A", RuleUID: "a", Outcome: OutcomeClean}, - {Alert: "B", RuleUID: "b", Outcome: OutcomeUnobservable}, + {Alert: "A", RuleUID: "a", Outcome: OutcomeHealthy}, + {Alert: "B", RuleUID: "b", Outcome: OutcomeNotVerified}, }, } @@ -1341,9 +1341,9 @@ func TestMergeDrainTimeoutsNamesEveryUnobservableRule(t *testing.T) { "a": {reason: ReasonDrainTimeout, note: "rule \"A\": did not evaluate through the end within the drain limit"}, "b": {reason: ReasonDrainTimeout, note: "rule \"B\": did not evaluate through the end within the drain limit"}, }) - require.Error(t, err, "naming the newly unobservable rule") - require.Contains(t, err.Error(), "unobservable at the drain wait") - // Only A is newly unobservable; B was already, so naming it twice would + require.Error(t, err, "naming the newly not_verified rule") + require.Contains(t, err.Error(), "not verified at the drain wait") + // Only A is newly not_verified; B was already, so naming it twice would // only lengthen the message. require.Contains(t, err.Error(), "A ("+string(ReasonDrainTimeout)+")") require.NotContains(t, err.Error(), "B (") @@ -1351,7 +1351,7 @@ func TestMergeDrainTimeoutsNamesEveryUnobservableRule(t *testing.T) { // B keeps the reason the coverage proof gave it — the FIRST reason wins, // as it does inside proveCoverage. require.Equal(t, ReasonHeartbeatGap, merged.Coverage["b"].Reason) - require.Equal(t, OutcomeUnobservable, merged.Verdicts[0].Outcome) + require.Equal(t, OutcomeNotVerified, merged.Verdicts[0].Outcome) } // ReadLogHeader is the one read of a log a writer may still hold, so its diff --git a/grafana-alertcheck/internal/gate/classify.go b/grafana-alertcheck/internal/gate/classify.go index 41e2a5f7b..ab31edb3b 100644 --- a/grafana-alertcheck/internal/gate/classify.go +++ b/grafana-alertcheck/internal/gate/classify.go @@ -19,26 +19,30 @@ const ReasonNodata UnobservableReason = "nodata" type Outcome string const ( - OutcomeClean Outcome = "clean" - OutcomeNewlyBad Outcome = "newly_bad" - OutcomeRecovered Outcome = "recovered" - OutcomePersistentlyBad Outcome = "persistently_bad" - OutcomeFlapping Outcome = "flapping" - OutcomeSkipped Outcome = "skipped" - OutcomeUnobservable Outcome = "unobservable" + OutcomeHealthy Outcome = "healthy" + OutcomeNewFailure Outcome = "new_failure" + OutcomeRecovered Outcome = "recovered" + OutcomeStillFailing Outcome = "still_failing" + OutcomeUnstable Outcome = "unstable" + OutcomePaused Outcome = "paused" + OutcomeNotVerified Outcome = "not_verified" + // OutcomeNotCounted is synthetic: decide uses it for a --min-observed + // deficit row that no resolved rule explains. It is not a verdict on an + // alert, so it is deliberately not named after one. + OutcomeNotCounted Outcome = "not_counted" ) // PreexistingPolicy governs only the ONE ambiguous case in the outcome table: -// an instance that was already bad when the window opened. A newly_bad or -// flapping instance is a fail under every policy, so this type only ever -// changes how `recovered` and `persistently_bad` are judged (isViolation +// an instance that was already bad when the window opened. A new_failure or +// unstable instance is a fail under every policy, so this type only ever +// changes how `recovered` and `still_failing` are judged (isViolation // below). type PreexistingPolicy string const ( // PreexistingFailUnlessRecovered is the default: a preexisting instance // that clears and stays clear is a pass (`recovered`); one that never - // clears is still a fail (`persistently_bad`). + // clears is still a fail (`still_failing`). PreexistingFailUnlessRecovered PreexistingPolicy = "fail-unless-recovered" // PreexistingFail makes ANY preexisting instance a fail, even one that // recovers — for a user who wants no benefit of the doubt for a @@ -46,7 +50,7 @@ const ( PreexistingFail PreexistingPolicy = "fail" // PreexistingIgnore disregards a preexisting instance entirely, whether // it recovers or stays bad for the whole window: only a genuinely NEW - // bad episode (newly_bad or flapping) can fail the rule. + // bad episode (new_failure or unstable) can fail the rule. PreexistingIgnore PreexistingPolicy = "ignore" ) @@ -294,7 +298,7 @@ func classifyRule(def Definition, polls []Poll, from, windowEnd time.Time, badSt slices.Sort(order) var ( - outcome = OutcomeClean + outcome = OutcomeHealthy badFor []episode viols []Violation ) @@ -311,17 +315,17 @@ func classifyRule(def Definition, polls []Poll, from, windowEnd time.Time, badSt var instOutcome Outcome switch { case len(tl.episodes) > 1: - instOutcome = OutcomeFlapping + instOutcome = OutcomeUnstable case tl.preexisting: if tl.episodes[0].closedByRealClear { instOutcome = OutcomeRecovered } else { - instOutcome = OutcomePersistentlyBad + instOutcome = OutcomeStillFailing } default: // A genuinely new onset fails whether or not it clears in-window; // only a preexisting condition earns `recovered`. - instOutcome = OutcomeNewlyBad + instOutcome = OutcomeNewFailure } if outcomeRank(instOutcome) > outcomeRank(outcome) { @@ -353,17 +357,17 @@ func classifyRule(def Definition, polls []Poll, from, windowEnd time.Time, badSt } // isViolation decides whether one instance's outcome counts against the run, -// once the preexisting policy is applied. newly_bad and flapping always do: +// once the preexisting policy is applied. new_failure and unstable always do: // both contain a genuinely new bad episode, so no policy forgives them. -// recovered and persistently_bad are, by classifyRule's construction, -// ALWAYS preexisting (a non-preexisting single episode is newly_bad instead, +// recovered and still_failing are, by classifyRule's construction, +// ALWAYS preexisting (a non-preexisting single episode is new_failure instead, // regardless of whether it clears) — so these are the only two policy can // change, and isViolation needs no separate preexisting flag to know that. func isViolation(o Outcome, pol PreexistingPolicy) bool { switch o { - case OutcomeNewlyBad, OutcomeFlapping: + case OutcomeNewFailure, OutcomeUnstable: return true - case OutcomePersistentlyBad: + case OutcomeStillFailing: return pol != PreexistingIgnore case OutcomeRecovered: return pol == PreexistingFail @@ -375,25 +379,25 @@ func isViolation(o Outcome, pol PreexistingPolicy) bool { // outcomeRank orders outcomes for classifyRule's worst-of reduction across a // rule's instances: // -// unobservable > {flapping, persistently_bad, newly_bad} > recovered > -// skipped > clean +// not_verified > {unstable, still_failing, new_failure} > recovered > +// paused > healthy // -// with unobservable and skipped applied outside this function (decide owns -// both: unobservable from CoverageResult, skipped from the log header). The +// with not_verified and paused applied outside this function (decide owns +// both: not_verified from CoverageResult, paused from the log header). The // three fail values are not ranked against each other by anything that reads // this, so their relative order here is an arbitrary but fixed tie-break, not // a claim that one is worse than another. func outcomeRank(o Outcome) int { switch o { - case OutcomeFlapping: + case OutcomeUnstable: return 4 - case OutcomePersistentlyBad: + case OutcomeStillFailing: return 3 - case OutcomeNewlyBad: + case OutcomeNewFailure: return 2 case OutcomeRecovered: return 1 - default: // OutcomeClean + default: // OutcomeHealthy return 0 } } @@ -483,7 +487,7 @@ func applyNodataPolicy(def Definition, polls []Poll, cov *CoverageResult, t Rule // decide is the pure seam between the collected evidence and the CLI's exit // code, and carries nearly the whole test suite because of it. It combines // proveCoverage's nine checks with classifyRule's timelines under one Policy, -// and owns the inability-beats-violation rule: any unobservable rule makes +// and owns the inability-beats-violation rule: any not-verified rule makes // decide return a non-nil error, which the CLI maps to exit 2 unconditionally // — never to 0 or 1, and never suppressed by a real violation found alongside // it. @@ -532,13 +536,13 @@ func decide(h Header, polls []Poll, sentinel *time.Time, defs []Definition, windowEnd := pol.To.Add(gt.transitionGrace) var ( - skippedRules []Definition + pausedRules []Definition watchedCount int anyUnobservable bool unobservableNames []string ) - // `skipped` is decided from the header, never from defs: defs are resolved + // `paused` is decided from the header, never from defs: defs are resolved // after the window closed, so Definition.IsPaused describes the present, // while Header.pausedAtStart describes the window open — the only moment // "paused before the window opened" can mean. @@ -546,9 +550,9 @@ func decide(h Header, polls []Poll, sentinel *time.Time, defs []Definition, for _, def := range defs { if pausedAtStart[def.UID] { - skippedRules = append(skippedRules, def) + pausedRules = append(pausedRules, def) result.Verdicts = append(result.Verdicts, RuleVerdict{ - Alert: def.Title, RuleUID: def.UID, Outcome: OutcomeSkipped, + Alert: def.Title, RuleUID: def.UID, Outcome: OutcomePaused, PollEvery: rt[def.UID].pollEvery, Note: "paused before the window opened", }) @@ -571,7 +575,7 @@ func decide(h Header, polls []Poll, sentinel *time.Time, defs []Definition, outcome, badFor, viols := classifyRule(def, polls, pol.From, windowEnd, badStates, pol.Preexisting) if cov.Unobservable { - outcome = OutcomeUnobservable + outcome = OutcomeNotVerified anyUnobservable = true unobservableNames = append(unobservableNames, fmt.Sprintf("%s (%s)", def.Title, cov.Reason)) } @@ -588,9 +592,9 @@ func decide(h Header, polls []Poll, sentinel *time.Time, defs []Definition, counted := watchedCount var attributable []Definition if pol.AllowPaused { - counted += len(skippedRules) + counted += len(pausedRules) } else { - attributable = skippedRules + attributable = pausedRules } if shortfall := minObserved - counted; shortfall > 0 { attributed := 0 @@ -603,7 +607,7 @@ func decide(h Header, polls []Poll, sentinel *time.Time, defs []Definition, // prints Note verbatim rather than re-deriving the hint, so the // exact wording here is what an operator reads. result.Violations = append(result.Violations, Violation{ - Alert: def.Title, RuleUID: def.UID, Outcome: OutcomeSkipped, + Alert: def.Title, RuleUID: def.UID, Outcome: OutcomePaused, Note: "paused before the window opened; counts against --min-observed unless --allow-paused is set", }) attributed++ @@ -615,14 +619,14 @@ func decide(h Header, polls []Poll, sentinel *time.Time, defs []Definition, // rule state read from a real poll, and this Violation never // touched one. result.Violations = append(result.Violations, Violation{ - Outcome: OutcomeSkipped, + Outcome: OutcomeNotCounted, Note: fmt.Sprintf("min-observed %d exceeds the %d rule(s) counted as observed", minObserved, counted), }) } } if anyUnobservable { - return result, fmt.Errorf("gate: %d rule(s) unobservable: %s", len(unobservableNames), strings.Join(unobservableNames, "; ")) + return result, fmt.Errorf("gate: %d rule(s) not verified: %s", len(unobservableNames), strings.Join(unobservableNames, "; ")) } return result, nil } diff --git a/grafana-alertcheck/internal/gate/classify_test.go b/grafana-alertcheck/internal/gate/classify_test.go index 0e6b16c8e..14cc95248 100644 --- a/grafana-alertcheck/internal/gate/classify_test.go +++ b/grafana-alertcheck/internal/gate/classify_test.go @@ -1,6 +1,7 @@ package gate import ( + "encoding/json" "testing" "time" @@ -48,7 +49,7 @@ func pausedHeader(startedAt time.Time, pausedUIDs ...string) Header { return h } -// --- clean / newly_bad --- +// --- healthy / new_failure --- func TestClassifyRule_NoEvidenceIsClean(t *testing.T) { from := time.Date(2026, 1, 1, 0, 0, 0, 0, time.UTC) @@ -57,7 +58,7 @@ func TestClassifyRule_NoEvidenceIsClean(t *testing.T) { polls := []Poll{quietPoll("r1", from), quietPoll("r1", to)} outcome, badFor, viols := classifyRule(def, polls, from, to, defaultBad, PreexistingFailUnlessRecovered) - require.Equal(t, OutcomeClean, outcome) + require.Equal(t, OutcomeHealthy, outcome) require.Zero(t, badFor) require.Empty(t, viols) } @@ -74,10 +75,10 @@ func TestClassifyRule_NewOnsetInsideWindowIsNewlyBad(t *testing.T) { abnormalPoll("r1", to, StateFiring, lbl("a"), onset), } outcome, badFor, viols := classifyRule(def, polls, from, to, defaultBad, PreexistingFailUnlessRecovered) - require.Equal(t, OutcomeNewlyBad, outcome) + require.Equal(t, OutcomeNewFailure, outcome) require.Equal(t, to.Sub(onset), badFor) require.Len(t, viols, 1) - require.Equal(t, OutcomeNewlyBad, viols[0].Outcome) + require.Equal(t, OutcomeNewFailure, viols[0].Outcome) } // A genuinely new bad episode fails even if it clears again before the window @@ -96,11 +97,11 @@ func TestClassifyRule_NewOnsetThatClearsStillFails(t *testing.T) { quietPoll("r1", to), } outcome, _, viols := classifyRule(def, polls, from, to, defaultBad, PreexistingFailUnlessRecovered) - require.Equal(t, OutcomeNewlyBad, outcome, "even though it cleared") + require.Equal(t, OutcomeNewFailure, outcome, "even though it cleared") require.Len(t, viols, 1) } -// --- recovered / persistently_bad (preexisting) --- +// --- recovered / still_failing (preexisting) --- func TestClassifyRule_PreexistingThatRecoversIsRecoveredAndNotAViolation(t *testing.T) { from := time.Date(2026, 1, 1, 0, 0, 0, 0, time.UTC) @@ -151,13 +152,13 @@ func TestClassifyRule_PreexistingStillBadAtWindowEndIsPersistentlyBad(t *testing abnormalPoll("r1", to, StateFiring, lbl("a"), from.Add(-time.Hour)), } outcome, badFor, viols := classifyRule(def, polls, from, to, defaultBad, PreexistingFailUnlessRecovered) - require.Equal(t, OutcomePersistentlyBad, outcome) + require.Equal(t, OutcomeStillFailing, outcome) require.Equal(t, to.Sub(from), badFor) require.Len(t, viols, 1) - require.Equal(t, OutcomePersistentlyBad, viols[0].Outcome) + require.Equal(t, OutcomeStillFailing, viols[0].Outcome) } -// --- flapping --- +// --- unstable --- func TestClassifyRule_ClearThenBadAgainIsFlapping(t *testing.T) { from := time.Date(2026, 1, 1, 0, 0, 0, 0, time.UTC) @@ -172,12 +173,12 @@ func TestClassifyRule_ClearThenBadAgainIsFlapping(t *testing.T) { quietPoll("r1", to), } outcome, _, viols := classifyRule(def, polls, from, to, defaultBad, PreexistingFailUnlessRecovered) - require.Equal(t, OutcomeFlapping, outcome) + require.Equal(t, OutcomeUnstable, outcome) require.Len(t, viols, 1) - require.Equal(t, OutcomeFlapping, viols[0].Outcome, "always a fail regardless of policy") + require.Equal(t, OutcomeUnstable, viols[0].Outcome, "always a fail regardless of policy") } -// A clear and then a second bad state gives flapping, wherever the second bad +// A clear and then a second bad state gives unstable, wherever the second bad // state lands. A table over where the second onset falls — immediately after // the clear, mid-window, and right at the // last instant before windowEnd — closes the boundary this single fixed @@ -206,9 +207,9 @@ func TestClassifyRule_FlappingAtEveryTimingOfTheSecondOnset(t *testing.T) { quietPoll("r1", to), } outcome, _, viols := classifyRule(def, polls, from, to, defaultBad, PreexistingFailUnlessRecovered) - require.Equalf(t, OutcomeFlapping, outcome, "second onset at %s", tc.secondOnset) + require.Equalf(t, OutcomeUnstable, outcome, "second onset at %s", tc.secondOnset) require.Len(t, viols, 1) - require.Equal(t, OutcomeFlapping, viols[0].Outcome) + require.Equal(t, OutcomeUnstable, viols[0].Outcome) }) } } @@ -227,7 +228,7 @@ func TestClassifyRule_VanishedWhileBadStaysPersistentlyBad(t *testing.T) { quietPoll("r1", to), } outcome, badFor, viols := classifyRule(def, polls, from, to, defaultBad, PreexistingFailUnlessRecovered) - require.Equal(t, OutcomePersistentlyBad, outcome, "a vanish must never read as a recovery") + require.Equal(t, OutcomeStillFailing, outcome, "a vanish must never read as a recovery") require.Equal(t, to.Sub(from), badFor, "the freeze must hold the episode open to windowEnd") require.Len(t, viols, 1) } @@ -246,7 +247,7 @@ func TestClassifyRule_VanishedWhileNeverBadIsUninteresting(t *testing.T) { quietPoll("r1", to), } outcome, badFor, viols := classifyRule(def, polls, from, to, defaultBad, PreexistingFailUnlessRecovered) - require.Equal(t, OutcomeClean, outcome) + require.Equal(t, OutcomeHealthy, outcome) require.Zero(t, badFor) require.Empty(t, viols) } @@ -281,7 +282,7 @@ func TestClassifyRule_PreexistingPolicyIgnoreForgivesPersistentlyBad(t *testing. abnormalPoll("r1", to, StateFiring, lbl("a"), from.Add(-time.Hour)), } outcome, _, viols := classifyRule(def, polls, from, to, defaultBad, PreexistingIgnore) - require.Equal(t, OutcomePersistentlyBad, outcome, "the descriptive outcome does not change under policy=ignore") + require.Equal(t, OutcomeStillFailing, outcome, "the descriptive outcome does not change under policy=ignore") require.Empty(t, viols, "policy=ignore disregards a preexisting instance even if it never recovers") } @@ -297,7 +298,7 @@ func TestClassifyRule_PreexistingPolicyIgnoreStillFailsANewOnset(t *testing.T) { abnormalPoll("r1", to, StateFiring, lbl("a"), onset), } outcome, _, viols := classifyRule(def, polls, from, to, defaultBad, PreexistingIgnore) - require.Equal(t, OutcomeNewlyBad, outcome) + require.Equal(t, OutcomeNewFailure, outcome) require.Len(t, viols, 1, "ignore only forgives PREEXISTING badness") } @@ -323,9 +324,9 @@ func TestClassifyRule_WorstOfMultipleInstancesWins(t *testing.T) { }, } outcome, _, viols := classifyRule(def, polls, from, to, defaultBad, PreexistingFailUnlessRecovered) - require.Equal(t, OutcomePersistentlyBad, outcome, "the worse of {recovered, persistently_bad}") + require.Equal(t, OutcomeStillFailing, outcome, "the worse of {recovered, still_failing}") require.Len(t, viols, 1) - require.Equal(t, OutcomePersistentlyBad, viols[0].Outcome) + require.Equal(t, OutcomeStillFailing, viols[0].Outcome) } // --- decide(): skipped rules, unobservable, MinObserved, exit mapping --- @@ -344,9 +345,9 @@ func TestDecide_SkippedRuleNeverReachesProveCoverage(t *testing.T) { // The HEADER is what says paused — decide reads skipped from there, not // from def.IsPaused, which is a post-window reading (Header.pausedAtStart). res, err := decide(pausedHeader(from.Add(-time.Hour), "r1"), nil, nil, defs, rt, gt, pol) - require.NoError(t, err, "a rule paused before the window is skipped, not unobservable") + require.NoError(t, err, "a rule paused before the window is paused, not not_verified") require.Len(t, res.Verdicts, 1) - require.Equal(t, OutcomeSkipped, res.Verdicts[0].Outcome) + require.Equal(t, OutcomePaused, res.Verdicts[0].Outcome) _, ok := res.Coverage["r1"] require.False(t, ok, "a skipped rule has no coverage to prove") } @@ -365,11 +366,11 @@ func TestDecide_UnobservableRuleAlwaysReturnsAnError(t *testing.T) { res, err := decide(Header{StartedAt: from.Add(-time.Hour)}, nil, nil, defs, rt, gt, pol) require.Error(t, err, "an unobservable rule must always fail the run") require.Len(t, res.Verdicts, 1) - require.Equal(t, OutcomeUnobservable, res.Verdicts[0].Outcome) + require.Equal(t, OutcomeNotVerified, res.Verdicts[0].Outcome) } // Any unobservable rule means exit 2, with no exception — even alongside a -// real newly_bad. +// real new_failure. func TestDecide_UnobservableWinsEvenAlongsideARealViolation(t *testing.T) { from := time.Date(2026, 1, 1, 0, 0, 0, 0, time.UTC) to := from.Add(10 * time.Minute) @@ -407,11 +408,11 @@ func TestDecide_UnobservableWinsEvenAlongsideARealViolation(t *testing.T) { gotBad = v.Outcome } } - require.Equal(t, OutcomeUnobservable, gotBroken) - require.Equal(t, OutcomeNewlyBad, gotBad, + require.Equal(t, OutcomeNotVerified, gotBroken) + require.Equal(t, OutcomeNewFailure, gotBad, "classification still runs and is still visible in Verdicts") require.NotEmpty(t, res.Violations, - "the newly_bad instance still reported even though the run fails on the unobservable rule") + "the new_failure instance still reported even though the run fails on the not_verified rule") } // A clean verdict with a coverage gap must never give exit 0, and recovered @@ -429,7 +430,7 @@ func TestDecide_UnobservableRuleWinsOverEveryFavorableOutcome(t *testing.T) { wantOutcome Outcome }{ { - name: "clean", + name: "healthy", goodPolls: func() []Poll { var polls []Poll for ts := from; !ts.After(to); ts = ts.Add(30 * time.Second) { @@ -437,7 +438,7 @@ func TestDecide_UnobservableRuleWinsOverEveryFavorableOutcome(t *testing.T) { } return polls }(), - wantOutcome: OutcomeClean, + wantOutcome: OutcomeHealthy, }, { // Dense 30s-spaced polls throughout, so "good"'s own coverage @@ -466,9 +467,9 @@ func TestDecide_UnobservableRuleWinsOverEveryFavorableOutcome(t *testing.T) { wantOutcome: OutcomeRecovered, }, { - name: "skipped", + name: "paused", pausedAtStart: true, - wantOutcome: OutcomeSkipped, + wantOutcome: OutcomePaused, }, } @@ -502,7 +503,7 @@ func TestDecide_UnobservableRuleWinsOverEveryFavorableOutcome(t *testing.T) { } } require.Equal(t, tc.wantOutcome, gotGood) - require.Equal(t, OutcomeUnobservable, gotBroken) + require.Equal(t, OutcomeNotVerified, gotBroken) }) } } @@ -540,7 +541,7 @@ func TestDecide_RecoveredOutcomeOverriddenByItsOwnCoverageGap(t *testing.T) { res, err := decide(Header{StartedAt: from.Add(-time.Hour)}, polls, &sentinel, defs, rt, gt, pol) require.Error(t, err, "r1's own coverage gap must fail the run even though it recovered") require.Len(t, res.Verdicts, 1) - require.Equal(t, OutcomeUnobservable, res.Verdicts[0].Outcome, "never recovered") + require.Equal(t, OutcomeNotVerified, res.Verdicts[0].Outcome, "never recovered") require.False(t, res.Coverage["r1"].Proved) } @@ -562,7 +563,7 @@ func TestDecide_CleanWindowIsAPass(t *testing.T) { res, err := decide(Header{StartedAt: from.Add(-time.Hour)}, polls, &sentinel, defs, rt, gt, pol) require.NoError(t, err) require.Empty(t, res.Violations, "a pass is exactly len(Violations)==0 && err==nil") - require.Equal(t, OutcomeClean, res.Verdicts[0].Outcome) + require.Equal(t, OutcomeHealthy, res.Verdicts[0].Outcome) } // A pause and then an unpause inside the window, with an episode that would @@ -604,7 +605,7 @@ func TestDecide_PauseThenUnpauseWithHiddenEpisodeGivesUnobservableNotClean(t *te res, err := decide(Header{StartedAt: from.Add(-time.Hour)}, polls, &sentinel, defs, rt, gt, pol) require.Error(t, err, "the pause-then-unpause blind interval must fail closed") require.Len(t, res.Verdicts, 1) - require.Equal(t, OutcomeUnobservable, res.Verdicts[0].Outcome, "never clean") + require.Equal(t, OutcomeNotVerified, res.Verdicts[0].Outcome, "never clean") require.False(t, res.Coverage["r1"].Proved) } @@ -632,10 +633,10 @@ func TestDecide_SkippedOnlyShortfallProducesAViolationWithoutAnError(t *testing. sentinel := to res, err := decide(pausedHeader(from.Add(-time.Hour), "paused"), polls, &sentinel, defs, rt, gt, pol) - require.NoError(t, err, "a shortfall caused only by a skipped rule is exit 1, not exit 2") + require.NoError(t, err, "a shortfall caused only by a paused rule is exit 1, not exit 2") require.Len(t, res.Violations, 1) v := res.Violations[0] - require.Equal(t, OutcomeSkipped, v.Outcome) + require.Equal(t, OutcomePaused, v.Outcome) require.Equal(t, "paused", v.RuleUID) require.Equal(t, "Paused", v.Alert) require.NotEmpty(t, v.Note, @@ -665,7 +666,8 @@ func TestDecide_ExplicitMinObservedShortfallWithNoPausedRuleStillProducesAViolat require.NoError(t, err, "an unmet MinObserved is exit 1, never exit 2") require.Len(t, res.Violations, 2, "the shortfall (3-1=2) must surface directly rather than pass silently") for _, v := range res.Violations { - require.Equal(t, OutcomeSkipped, v.Outcome) + require.Equal(t, OutcomeNotCounted, v.Outcome, + "no resolved rule explains the deficit, so it must not be blamed on a paused one") } } @@ -740,7 +742,7 @@ func TestDecide_NodataIsANoteByDefault(t *testing.T) { // An instance whose true onset (ActiveAt) falls strictly inside the window — // even though the first poll that happens to observe it already shows it bad — -// must never be treated as preexisting. If it then clears, that is newly_bad +// must never be treated as preexisting. If it then clears, that is new_failure // (exit 1), not recovered (exit 0). func TestClassifyRule_OnsetBetweenFromAndFirstPollIsNewlyBadNotRecovered(t *testing.T) { from := time.Date(2026, 1, 1, 0, 0, 0, 0, time.UTC) @@ -757,10 +759,10 @@ func TestClassifyRule_OnsetBetweenFromAndFirstPollIsNewlyBadNotRecovered(t *test quietPoll("r1", to), } outcome, badFor, viols := classifyRule(def, polls, from, to, defaultBad, PreexistingFailUnlessRecovered) - require.Equal(t, OutcomeNewlyBad, outcome, + require.Equal(t, OutcomeNewFailure, outcome, "the onset is after `from`, so it is not preexisting even though the FIRST in-window poll already observes it bad") require.Len(t, viols, 1) - require.Equal(t, OutcomeNewlyBad, viols[0].Outcome) + require.Equal(t, OutcomeNewFailure, viols[0].Outcome) require.Equal(t, clearAt.Sub(onset), badFor, "BadFor must count from the true onset, not from `from`") } @@ -799,7 +801,7 @@ func TestClassifyRule_SkewTranslatesActiveAtAcrossTheWindowBoundary(t *testing.T // Grafana's clock reads 90s ahead of the runner's (skew = +90s). The // poll's raw GrafanaNow/ActiveAt both sit 90s past `from` in Grafana's // domain, but translate to exactly `from` in the runner domain — genuinely - // preexisting once translated, and wrongly "newly_bad" if the skew is + // preexisting once translated, and wrongly "new_failure" if the skew is // ignored. skew := 90 * time.Second rawActiveAt := from.Add(skew) @@ -813,7 +815,7 @@ func TestClassifyRule_SkewTranslatesActiveAtAcrossTheWindowBoundary(t *testing.T stillBad.LastEvaluation = to.Add(skew) outcome, badFor, _ := classifyRule(def, []Poll{poll, stillBad}, from, to, defaultBad, PreexistingFailUnlessRecovered) - require.Equal(t, OutcomePersistentlyBad, outcome, + require.Equal(t, OutcomeStillFailing, outcome, "a +90s skew must translate ActiveAt back to exactly `from`") require.Equal(t, to.Sub(from), badFor) } @@ -893,7 +895,7 @@ func TestClassifyRule_ClearedEventPastWindowEndClampsToWindowEnd(t *testing.T) { // reading of the upper boundary: an instance whose runner-domain onset lands // only slightly past windowEnd (to + transitionGrace) is reachable at all only // because inWindowPolls widens the boundary outward by the skew bound, so the -// gate cannot PROVE it belongs to the next window. It is charged as newly_bad — +// gate cannot PROVE it belongs to the next window. It is charged as new_failure — // with BadFor truncated to zero — rather than silently forgiven as clean. func TestClassifyRule_OnsetJustPastWindowEndIsNewlyBadNotClean(t *testing.T) { from := time.Date(2026, 1, 1, 0, 0, 0, 0, time.UTC) @@ -914,13 +916,13 @@ func TestClassifyRule_OnsetJustPastWindowEndIsNewlyBadNotClean(t *testing.T) { } outcome, badFor, viols := classifyRule(def, []Poll{poll}, from, windowEnd, defaultBad, PreexistingFailUnlessRecovered) - require.Equal(t, OutcomeNewlyBad, outcome, + require.Equal(t, OutcomeNewFailure, outcome, "an onset past windowEnd seen only via the skew bound must fail closed") require.Zero(t, badFor, "the zero-length episode must truncate to the window end") require.Len(t, viols, 1) } -// A clear after `to` gives persistently_bad. classifyRule filters +// A clear after `to` gives still_failing. classifyRule filters // its input to [from, windowEnd] itself (inWindowPolls), so a Cleared event // GENUINELY past windowEnd — well beyond any skew bound, unlike the clamp // case above — never reaches the timeline at all: the instance is still bad @@ -936,10 +938,10 @@ func TestClassifyRule_ClearAfterWindowEndIsPersistentlyBad(t *testing.T) { clearedPoll("r1", to.Add(time.Hour), key), // far past `to`, not a boundary case } outcome, badFor, viols := classifyRule(def, polls, from, to, defaultBad, PreexistingFailUnlessRecovered) - require.Equal(t, OutcomePersistentlyBad, outcome, "a clear outside the window must not read as a recovery") + require.Equal(t, OutcomeStillFailing, outcome, "a clear outside the window must not read as a recovery") require.Equal(t, to.Sub(from), badFor) require.Len(t, viols, 1) - require.Equal(t, OutcomePersistentlyBad, viols[0].Outcome) + require.Equal(t, OutcomeStillFailing, viols[0].Outcome) } // TestClassifyRule_CloseBeforeOpenClampsToZeroNotNegative pins the @@ -966,7 +968,7 @@ func TestClassifyRule_CloseBeforeOpenClampsToZeroNotNegative(t *testing.T) { quietPoll("r1", to), } outcome, badFor, viols := classifyRule(def, polls, from, to, defaultBad, PreexistingFailUnlessRecovered) - require.Equal(t, OutcomeNewlyBad, outcome) + require.Equal(t, OutcomeNewFailure, outcome) require.GreaterOrEqual(t, badFor, time.Duration(0), "a non-negative duration even though the closing poll's translated time landed before the opening poll's") require.Zero(t, badFor, "the clamp collapses the inverted span to a zero-length episode") @@ -990,3 +992,30 @@ func TestMergeDurations_OverlappingEpisodesCountOnce(t *testing.T) { func TestMergeDurations_Empty(t *testing.T) { require.Zero(t, mergeDurations(nil)) } + +// --- published vocabulary --- + +// Outcome strings are published JSON, but every other test compares against +// the constants, so a typo in a constant's literal would pass them all. Pin +// the literals themselves. +func TestOutcomeJSONVocabulary(t *testing.T) { + want := map[Outcome]string{ + OutcomeHealthy: "healthy", + OutcomeNewFailure: "new_failure", + OutcomeRecovered: "recovered", + OutcomeStillFailing: "still_failing", + OutcomeUnstable: "unstable", + OutcomePaused: "paused", + OutcomeNotVerified: "not_verified", + OutcomeNotCounted: "not_counted", + } + require.Len(t, want, 8, "every Outcome constant must be pinned here") + + for outcome, literal := range want { + t.Run(string(outcome), func(t *testing.T) { + raw, err := json.Marshal(outcome) + require.NoError(t, err) + require.Equal(t, `"`+literal+`"`, string(raw)) + }) + } +} diff --git a/grafana-alertcheck/internal/gate/coverage_test.go b/grafana-alertcheck/internal/gate/coverage_test.go index 7126372ed..226b46243 100644 --- a/grafana-alertcheck/internal/gate/coverage_test.go +++ b/grafana-alertcheck/internal/gate/coverage_test.go @@ -79,7 +79,7 @@ func TestProveCoverage_SentinelBeforeGraceIsUnobservable(t *testing.T) { dres, err := decide(Header{StartedAt: from.Add(-time.Hour)}, nil, &sentinel, defs, drt, gt, pol) require.Error(t, err, "a sentinel short of to+grace must fail the run") require.Len(t, dres.Verdicts, 1) - require.Equal(t, OutcomeUnobservable, dres.Verdicts[0].Outcome) + require.Equal(t, OutcomeNotVerified, dres.Verdicts[0].Outcome) } func TestProveCoverage_SentinelExactlyAtGraceIsFine(t *testing.T) { @@ -123,7 +123,7 @@ func TestProveCoverage_FromBeforeRecordIsUnobservable(t *testing.T) { dres, err := decide(Header{StartedAt: started}, nil, &sentinel, defs, drt, gt, pol) require.Error(t, err, "`from` before the recording started must fail the run") require.Len(t, dres.Verdicts, 1) - require.Equal(t, OutcomeUnobservable, dres.Verdicts[0].Outcome) + require.Equal(t, OutcomeNotVerified, dres.Verdicts[0].Outcome) } // The from-bounds check compares at whole-second granularity: a whole-second @@ -688,15 +688,15 @@ func TestProveCoverage_MultipleFailuresReasonIsFirstButAllNoted(t *testing.T) { "a later failure must still be recorded, not swallowed once Reason is already set") } -// --- Skipped rules --- +// --- Paused rules --- // A known limit of this function's contract, not a bug in it: a rule paused // BEFORE the window opened is never scheduled or polled (watch.go), so it // reaches proveCoverage with zero polls at all. proveCoverage has no notion of -// "skipped" — that classification belongs to the definitions +// "paused" — that classification belongs to the definitions // (LoggedRule.IsPaused / Definition.IsPaused), never to the polls — so it // reports the whole window as one big heartbeat_gap instead. decide is what -// reads skipped status from the header and never calls this function for such +// reads paused status from the header and never calls this function for such // a rule; this pins the behavior it relies on not reaching. func TestProveCoverage_SkippedRuleWithZeroPollsPinnedAsHeartbeatGap(t *testing.T) { from := time.Date(2026, 1, 1, 0, 0, 0, 0, time.UTC) diff --git a/grafana-alertcheck/internal/gate/terminal.go b/grafana-alertcheck/internal/gate/terminal.go index e41b9c923..c8b4197bb 100644 --- a/grafana-alertcheck/internal/gate/terminal.go +++ b/grafana-alertcheck/internal/gate/terminal.go @@ -9,10 +9,10 @@ import ( type TerminationKind string const ( - // TerminationViolation is a post-`from` bad onset (newly_bad or flapping). + // TerminationViolation is a post-`from` bad onset (new_failure or unstable). TerminationViolation TerminationKind = "violation" - // TerminationUnobservable is an inability that has already happened. - TerminationUnobservable TerminationKind = "unobservable" + // TerminationNotVerified is an inability that has already happened. + TerminationNotVerified TerminationKind = "not_verified" ) // Termination is why a fail-fast run stopped before the window closed. At is @@ -30,10 +30,10 @@ type Termination struct { // observed so far, treating at as the provisional end of the window. PURE. // // Only two conditions qualify, because only they can never become a pass: an -// inability that already happened, and a post-`from` bad onset (newly_bad or -// flapping). `recovered` forgives an observed bad state and is reserved for +// inability that already happened, and a post-`from` bad onset (new_failure or +// unstable). `recovered` forgives an observed bad state and is reserved for // bad-at-`from`, so a preexisting condition is deliberately not terminal. -// unobservable beats violation, as it does at the end of a full run. +// not_verified beats violation, as it does at the end of a full run. func terminalVerdict(h Header, polls []Poll, defs []Definition, rt map[string]RuleTimings, pol Policy, from, at time.Time) (Termination, bool) { @@ -55,17 +55,17 @@ func terminalVerdict(h Header, polls []Poll, defs []Definition, rt map[string]Ru } if cov.Unobservable { return Termination{ - Kind: TerminationUnobservable, + Kind: TerminationNotVerified, Alert: def.Title, RuleUID: def.UID, - Outcome: OutcomeUnobservable, + Outcome: OutcomeNotVerified, Reason: cov.Reason, At: at, }, true } outcome, _, _ := classifyRule(def, polls, from, at, badStates, pol.Preexisting) - if outcome == OutcomeNewlyBad || outcome == OutcomeFlapping { + if outcome == OutcomeNewFailure || outcome == OutcomeUnstable { if violation == nil { v := Termination{ Kind: TerminationViolation, diff --git a/grafana-alertcheck/internal/gate/terminal_test.go b/grafana-alertcheck/internal/gate/terminal_test.go index 7f46bfd46..a89b71437 100644 --- a/grafana-alertcheck/internal/gate/terminal_test.go +++ b/grafana-alertcheck/internal/gate/terminal_test.go @@ -1,6 +1,7 @@ package gate import ( + "encoding/json" "testing" "time" @@ -57,7 +58,7 @@ func TestTerminalVerdict(t *testing.T) { polls: func() []Poll { return badFromPolls(checkUID, from, at, from.Add(time.Minute)) }, want: true, wantKind: TerminationViolation, - wantOut: OutcomeNewlyBad, + wantOut: OutcomeNewFailure, }, { // Bad before `from` can still become `recovered` (a pass). @@ -72,17 +73,17 @@ func TestTerminalVerdict(t *testing.T) { want: false, }, { - name: "heartbeat gap is terminal unobservable", + name: "heartbeat gap is terminal not_verified", polls: func() []Poll { return append(denseHealthyPolls(checkUID, from, from.Add(time.Minute), checkPollEvery), quietPoll(checkUID, at)) }, want: true, - wantKind: TerminationUnobservable, + wantKind: TerminationNotVerified, wantReason: ReasonHeartbeatGap, }, { - name: "sustained health=error is terminal unobservable", + name: "sustained health=error is terminal not_verified", polls: func() []Poll { var out []Poll for ts := from; !ts.After(at); ts = ts.Add(checkPollEvery) { @@ -93,29 +94,29 @@ func TestTerminalVerdict(t *testing.T) { return out }, want: true, - wantKind: TerminationUnobservable, + wantKind: TerminationNotVerified, wantReason: ReasonHealthError, }, { - name: "in-window pause is terminal unobservable", + name: "in-window pause is terminal not_verified", polls: func() []Poll { out := denseHealthyPolls(checkUID, from, at, checkPollEvery) out = append(out, Poll{RuleUID: checkUID, GrafanaNow: from.Add(time.Minute), Found: true, Health: "ok", IsPaused: true}) return out }, want: true, - wantKind: TerminationUnobservable, + wantKind: TerminationNotVerified, wantReason: ReasonPausedInWindow, }, { - name: "absent rule is terminal unobservable", + name: "absent rule is terminal not_verified", polls: func() []Poll { out := denseHealthyPolls(checkUID, from, at, checkPollEvery) out = append(out, Poll{RuleUID: checkUID, GrafanaNow: from.Add(time.Minute)}) return out }, want: true, - wantKind: TerminationUnobservable, + wantKind: TerminationNotVerified, wantReason: ReasonRuleAbsent, }, } @@ -158,7 +159,7 @@ func TestTerminalVerdictNodataIsTerminalOnlyWhenConfigured(t *testing.T) { term, ok := terminalVerdict(terminalHeader(from), polls, []Definition{def}, rt, Policy{From: from, To: at, NodataIsUnobservable: true}, from, at) require.True(t, ok) - require.Equal(t, TerminationUnobservable, term.Kind) + require.Equal(t, TerminationNotVerified, term.Kind) require.Equal(t, ReasonNodata, term.Reason) } @@ -180,7 +181,25 @@ func TestTerminalVerdictUnobservableBeatsViolation(t *testing.T) { term, ok := terminalVerdict(terminalHeader(from), polls, []Definition{violating, gapped}, rt, Policy{From: from, To: at}, from, at) require.True(t, ok) - require.Equal(t, TerminationUnobservable, term.Kind) + require.Equal(t, TerminationNotVerified, term.Kind) require.Equal(t, "rule-two", term.RuleUID) require.Equal(t, ReasonHeartbeatGap, term.Reason) } + +// Termination kinds are published JSON too, and the assertions above compare +// against the constants. Pin the literals. +func TestTerminationKindJSONVocabulary(t *testing.T) { + want := map[TerminationKind]string{ + TerminationViolation: "violation", + TerminationNotVerified: "not_verified", + } + require.Len(t, want, 2, "every TerminationKind constant must be pinned here") + + for kind, literal := range want { + t.Run(string(kind), func(t *testing.T) { + raw, err := json.Marshal(kind) + require.NoError(t, err) + require.Equal(t, `"`+literal+`"`, string(raw)) + }) + } +}