From 2baacbb6daf625405bead7bfa0639aff4c28f051 Mon Sep 17 00:00:00 2001 From: Bartek Tofel Date: Tue, 1 Sep 2026 08:45:57 +0200 Subject: [PATCH 1/5] chore: more concise comments --- .../cmd/grafana-alertcheck/check.go | 35 +- .../cmd/grafana-alertcheck/check_test.go | 33 +- .../cmd/grafana-alertcheck/common.go | 27 +- .../cmd/grafana-alertcheck/env.go | 3 +- .../cmd/grafana-alertcheck/list.go | 11 +- .../cmd/grafana-alertcheck/main.go | 10 +- .../cmd/grafana-alertcheck/table.go | 30 +- .../cmd/grafana-alertcheck/table_test.go | 25 +- .../cmd/grafana-alertcheck/watch.go | 21 +- .../cmd/grafana-alertcheck/watch_test.go | 10 +- grafana-alertcheck/internal/gate/check.go | 405 ++++++++---------- .../internal/gate/check_process.go | 2 +- .../internal/gate/check_test.go | 203 +++++---- grafana-alertcheck/internal/gate/classify.go | 233 +++++----- .../internal/gate/classify_test.go | 114 +++-- grafana-alertcheck/internal/gate/coverage.go | 202 +++++---- .../internal/gate/coverage_test.go | 112 +++-- grafana-alertcheck/internal/gate/duration.go | 2 +- grafana-alertcheck/internal/gate/flock.go | 4 +- grafana-alertcheck/internal/gate/jsonreq.go | 2 +- grafana-alertcheck/internal/gate/log.go | 191 ++++----- grafana-alertcheck/internal/gate/log_test.go | 47 +- .../internal/gate/parse_ruler.go | 24 +- .../internal/gate/parse_ruler_test.go | 5 +- .../internal/gate/parse_state.go | 39 +- .../internal/gate/parse_state_test.go | 30 +- grafana-alertcheck/internal/gate/resolve.go | 47 +- .../internal/gate/resolve_test.go | 14 +- grafana-alertcheck/internal/gate/schedule.go | 181 ++++---- .../internal/gate/schedule_test.go | 36 +- grafana-alertcheck/internal/gate/source.go | 93 ++-- .../internal/gate/source_fake_test.go | 27 +- .../internal/gate/source_test.go | 17 +- .../internal/gate/testdata/README.md | 45 +- grafana-alertcheck/internal/gate/watch.go | 150 +++---- .../internal/gate/watch_daemon_test.go | 22 +- .../internal/gate/watch_process.go | 10 +- .../internal/gate/watch_test.go | 69 ++- 38 files changed, 1214 insertions(+), 1317 deletions(-) diff --git a/grafana-alertcheck/cmd/grafana-alertcheck/check.go b/grafana-alertcheck/cmd/grafana-alertcheck/check.go index f4e9dab26..3f2d295d6 100644 --- a/grafana-alertcheck/cmd/grafana-alertcheck/check.go +++ b/grafana-alertcheck/cmd/grafana-alertcheck/check.go @@ -17,28 +17,27 @@ const checkUsage = "usage: grafana-alertcheck check [--in ] [--pidfile F] "[--alerts ...] [--folder F] [--states ...] [--preexisting ...] [--min-observed N] [--allow-paused] " + "[--nodata-is-unobservable] [--concurrency N] [--output json]" -// runCheck is the classify step's CLI surface: parse flags into a -// gate.Config, run gate.Check, and translate its (Result, error) into -// §20.2/§20.3's output and exit code. All of the correctness lives in -// gate.Check (P9) and decide (P8) — this file's only job is presentation and -// the H6/H7 exit-code mapping, which exitCode below keeps as one pure -// function so it can be tested without a network. +// runCheck is the classify step's CLI surface: parse flags into a gate.Config, +// run gate.Check, and translate its (Result, error) into output and an exit +// code. All of the correctness lives in the gate package — this file's only job +// is presentation and the exit-code mapping, which exitCode below keeps as one +// pure function so it can be tested without a network. func runCheck(args []string, stdin io.Reader, stdout, stderr io.Writer) int { fs := flag.NewFlagSet("check", flag.ContinueOnError) fs.SetOutput(stderr) fs.Usage = func() { fmt.Fprintln(stderr, checkUsage) } common := registerCommon(fs) - in := fs.String("in", "", "path of a log recorded by watch; empty selects single-step mode (§9)") + in := fs.String("in", "", "path of a log recorded by watch; empty selects single-step mode") pidfile := fs.String("pidfile", "", "pidfile of the recorder to stop before reading --in (default .pid)") - from := fs.String("from", "", "the moment the deploy finished, RFC3339 (required in recorder mode, §7)") + from := fs.String("from", "", "the moment the deploy finished, RFC3339 (required with --in)") to := fs.String("to", "", "the end of the window to classify, RFC3339 (required)") - states := fs.String("states", "", "comma-separated bad states to classify against (default: firing, §13)") - preexisting := fs.String("preexisting", "", "how to judge an instance already bad at `from` (default: fail-unless-recovered, §11.7)") - minObserved := fs.Int("min-observed", 0, "minimum rules that must be observed (default: every resolved rule, §12)") - allowPaused := fs.Bool("allow-paused", false, "do not count a rule paused before the window against --min-observed (§12.1)") - nodataIsUnobservable := fs.Bool("nodata-is-unobservable", false, "treat a sustained health=nodata as unobservable rather than a note (§10.2)") - output := fs.String("output", "", `"json" writes the machine-readable Result to stdout in addition to the table (§20.2); default is the table alone`) + states := fs.String("states", "", "comma-separated bad states to classify against (default: firing)") + preexisting := fs.String("preexisting", "", "how to judge an instance already bad at `from` (default: fail-unless-recovered)") + minObserved := fs.Int("min-observed", 0, "minimum rules that must be observed (default: every resolved rule)") + allowPaused := fs.Bool("allow-paused", false, "do not count a rule paused before the window against --min-observed") + nodataIsUnobservable := fs.Bool("nodata-is-unobservable", false, "treat a sustained health=nodata as unobservable rather than a note") + output := fs.String("output", "", `"json" writes the machine-readable Result to stdout in addition to the table; default is the table alone`) if err := fs.Parse(args); err != nil { return 2 @@ -86,7 +85,7 @@ func runCheck(args []string, stdin io.Reader, stdout, stderr io.Writer) int { Notes: stderr, } if *to == "" { - fmt.Fprintln(stderr, "check: --to is required (§7)") + fmt.Fprintln(stderr, "check: --to is required") return 2 } t, err := time.Parse(time.RFC3339, *to) @@ -131,11 +130,11 @@ func runCheck(args []string, stdin io.Reader, stdout, stderr io.Writer) int { return exitCode(result, checkErr) } -// exitCode is §20.3/H6/H7's whole mapping, kept as one pure function of +// exitCode is the whole exit-code mapping, kept as one pure function of // exactly what Check returns so it is testable without a network: err != nil // is exit 2 UNCONDITIONALLY — never 0 and never 1, even alongside real -// violations, because inability beats violation (H6) and an error is never a -// pass (H7). Violations without an error is exit 1. Neither is exit 0. +// violations, because an inability to check beats a violation and an error is +// never a pass. Violations without an error is exit 1. Neither is exit 0. func exitCode(res gate.Result, err error) int { switch { case err != nil: diff --git a/grafana-alertcheck/cmd/grafana-alertcheck/check_test.go b/grafana-alertcheck/cmd/grafana-alertcheck/check_test.go index dbf81d3a9..99e8ebc3a 100644 --- a/grafana-alertcheck/cmd/grafana-alertcheck/check_test.go +++ b/grafana-alertcheck/cmd/grafana-alertcheck/check_test.go @@ -10,9 +10,9 @@ import ( "github.com/smartcontractkit/chainlink-testing-framework/grafana-alertcheck/internal/gate" ) -// TestExitCode pins §20.3/H6/H7's mapping directly against exitCode, with no -// network involved: err != nil is exit 2 even alongside violations (H6 — -// inability beats violation), violations alone are exit 1, and neither is 0. +// The exit-code mapping, pinned directly against exitCode with no network +// involved: err != nil is exit 2 even alongside violations (an inability to +// check beats a violation), violations alone are exit 1, and neither is 0. func TestExitCode(t *testing.T) { tests := []struct { name string @@ -43,10 +43,10 @@ func writeTempAlerts(t *testing.T) string { return path } -// TestRunCheck_FlagValidation is the flag-validation matrix: every one of -// these must fail before any network call, because Config.validate() (P9) -// runs first — an unreachable GRAFANA_URL succeeding or timing out is a -// different test than these, which check pure input validation. +// The flag-validation matrix: every one of these must fail before any network +// call, because gate.Config.validate() runs first — an unreachable GRAFANA_URL +// succeeding or timing out is a different test than these, which check pure +// input validation. func TestRunCheck_FlagValidation(t *testing.T) { tests := []struct { name string @@ -73,9 +73,9 @@ func TestRunCheck_FlagValidation(t *testing.T) { return []string{"--to", "2026-01-01T00:00:00Z", "--states", "bogus", "--alerts", writeTempAlerts(t)} }, "--states"}, {"states normal is rejected", true, func(t *testing.T) []string { - // normal is the good state, never a state to classify AS bad - // (R1): accepting it would make --states normal fail every - // healthy instance, the fail-open shape H7 exists to prevent. + // normal is the good state, never a state to classify AS bad: + // accepting it would make --states normal fail every healthy + // instance. return []string{"--to", "2026-01-01T00:00:00Z", "--states", "normal", "--alerts", writeTempAlerts(t)} }, "--states"}, {"bad preexisting", true, func(t *testing.T) []string { @@ -110,9 +110,8 @@ func TestRunCheck_FlagValidation(t *testing.T) { } } -// TestRunCheck_ToInPastNoLog pins §4.2's refusal: a `to` already in the past -// with no recorded log cannot be classified from anything, because nothing -// ever observed the window. +// A `to` already in the past with no recorded log cannot be classified from +// anything, because nothing ever observed the window. func TestRunCheck_ToInPastNoLog(t *testing.T) { t.Setenv("GRAFANA_URL", "http://example.invalid") t.Setenv("GRAFANA_TOKEN", "test-token") @@ -126,13 +125,13 @@ func TestRunCheck_ToInPastNoLog(t *testing.T) { t.Fatalf("code = %d, want 2; stderr = %q", code, stderr.String()) } if !strings.Contains(stderr.String(), "already passed") { - t.Fatalf("stderr = %q, want the §4.2 refusal", stderr.String()) + t.Fatalf("stderr = %q, want the past-`to` refusal", stderr.String()) } } -// TestRunCheck_NoResultOnConfigError pins §20.2: --output json never writes -// to stdout when Check was never reached, because there is no Result to -// encode — only the table (on stderr) can report a configuration failure. +// --output json never writes to stdout when Check was never reached, because +// there is no Result to encode — only the table (on stderr) can report a +// configuration failure. func TestRunCheck_NoResultOnConfigError(t *testing.T) { t.Setenv("GRAFANA_URL", "") t.Setenv("GRAFANA_TOKEN", "") diff --git a/grafana-alertcheck/cmd/grafana-alertcheck/common.go b/grafana-alertcheck/cmd/grafana-alertcheck/common.go index e605c029b..3c56cd63f 100644 --- a/grafana-alertcheck/cmd/grafana-alertcheck/common.go +++ b/grafana-alertcheck/cmd/grafana-alertcheck/common.go @@ -12,11 +12,11 @@ import ( ) // commonFlags is registerCommon's result: the exactly three flags watch and -// check share (§20). Connection details are never flags (§20.2) and states / -// poll-interval are deliberately NOT here — states is check-only because -// recording is unfiltered (P6), and poll-interval is watch-only because check -// reads the cadence from the log header (P5). Putting either here would -// silently reinstate a knob this plan removed. +// check share. Connection details are never flags, and states / poll-interval +// are deliberately NOT here — states is check-only because recording is +// unfiltered, and poll-interval is watch-only because check reads the cadence +// from the log header. Putting either here would give both commands an opinion +// about a value only one of them may set. type commonFlags struct { folder *string concurrency *int @@ -25,13 +25,13 @@ type commonFlags struct { func registerCommon(fs *flag.FlagSet) *commonFlags { return &commonFlags{ - folder: fs.String("folder", "", "default folder to scope an unqualified alert name to (§17)"), + folder: fs.String("folder", "", "default folder to scope an unqualified alert name to"), concurrency: fs.Int("concurrency", 1, "maximum concurrent requests to Grafana"), alerts: fs.String("alerts", "", "path to a file of alert names, one per line, or - for stdin"), } } -// readAlerts reads §17's alert names, one per line, from a file or from +// readAlerts reads alert names, one per line, from a file or from // stdin when path is "-". An empty path is not an error here — watch and // check each decide for themselves whether an empty list is allowed // (log mode never wants one; single-step / record mode always does). @@ -62,16 +62,15 @@ func readAlerts(stdin io.Reader, path string) ([]string, error) { } // parseStates parses check's --states flag: a comma-separated list of the -// "bad" state vocabulary Config.States matches against (§13, classify.go's +// "bad" state vocabulary Config.States matches against (classify.go's // badStateSet). An empty string is not resolved here — it means "use the // library default of {firing}" — so this returns nil, nil for "" rather than // an error. // -// normal is deliberately NOT accepted: the v2 plan fixes this vocabulary to -// firing | pending | nodata | error (line 378) precisely because "normal" is -// the good state, never a bad one to classify against. Accepting it here -// would let --states normal turn every healthy instance into a violation and -// fail every healthy fleet — the exact fail-open shape H7 exists to prevent. +// normal is deliberately NOT accepted. The vocabulary is fixed to +// firing | pending | nodata | error precisely because "normal" is the good +// state, never a bad one to classify against: --states normal would turn every +// healthy instance into a violation and fail every healthy fleet. func parseStates(s string) ([]gate.State, error) { if strings.TrimSpace(s) == "" { return nil, nil @@ -95,7 +94,7 @@ func parseStates(s string) ([]gate.State, error) { return out, nil } -// parsePreexisting parses check's --preexisting flag (§11.7). +// parsePreexisting parses check's --preexisting flag. func parsePreexisting(s string) (gate.PreexistingPolicy, error) { switch gate.PreexistingPolicy(s) { case "": diff --git a/grafana-alertcheck/cmd/grafana-alertcheck/env.go b/grafana-alertcheck/cmd/grafana-alertcheck/env.go index e5702a6d1..02a125f1d 100644 --- a/grafana-alertcheck/cmd/grafana-alertcheck/env.go +++ b/grafana-alertcheck/cmd/grafana-alertcheck/env.go @@ -7,8 +7,7 @@ import ( // grafanaEnv reads the connection details from the environment only, never // from a flag — a flag value lands in the process argv and in CI logs, and -// the token must never be logged or otherwise surface in an error string -// (§20.2). +// the token must never be logged or otherwise surface in an error string. func grafanaEnv() (url, token string, err error) { url = os.Getenv("GRAFANA_URL") if url == "" { diff --git a/grafana-alertcheck/cmd/grafana-alertcheck/list.go b/grafana-alertcheck/cmd/grafana-alertcheck/list.go index 687e5dd9f..c1d0e8532 100644 --- a/grafana-alertcheck/cmd/grafana-alertcheck/list.go +++ b/grafana-alertcheck/cmd/grafana-alertcheck/list.go @@ -11,11 +11,10 @@ import ( ) // runList reads every rule definition from the ruler endpoint and prints one -// line per rule: its kind, its Folder/Group/Title, and its uid. This is what -// makes the gate runnable end to end before any coverage logic exists (§9 -// rule 4) — it validates auth, the ruler parse, and the shapes Resolve -// matches against, all against a real Grafana. It is also the "did you mean" -// surface §17.2's no-match error points operators at. +// line per rule: its kind, its Folder/Group/Title, and its uid. It validates +// auth, the ruler parse, and the shapes Resolve matches against, all against a +// real Grafana, and it is the surface Resolve's no-match error points operators +// at. func runList(args []string, stdout, stderr io.Writer) int { if len(args) != 0 { fmt.Fprintf(stderr, "list takes no arguments, got %v\n", args) @@ -34,7 +33,7 @@ func runList(args []string, stdout, stderr io.Writer) int { // always terminates. It can still take minutes end-to-end under repeated // transient failures (5 retries * up to 30s backoff each, per call) — an // acceptable wait for an interactive `list`, not for `watch`/`check`, - // which get their own deadlines from `--until`/`to` in P10. + // which get their own deadlines from `--until`/`--to`. src := gate.NewHTTPSource(url, token, gate.SystemClock{}) version, err := src.Version(context.Background()) if err != nil { diff --git a/grafana-alertcheck/cmd/grafana-alertcheck/main.go b/grafana-alertcheck/cmd/grafana-alertcheck/main.go index db5f93197..2672de1a3 100644 --- a/grafana-alertcheck/cmd/grafana-alertcheck/main.go +++ b/grafana-alertcheck/cmd/grafana-alertcheck/main.go @@ -1,5 +1,5 @@ -// Command grafana-alertcheck is the CLI entry point for the gate: `list` -// (P3), `watch` (record, P10) and `check` (classify, P10). +// Command grafana-alertcheck is the CLI entry point for the gate: `list`, +// `watch` (record) and `check` (classify). package main import ( @@ -16,9 +16,9 @@ const usage = "usage: grafana-alertcheck " // run is the whole of main's testable surface: parse the subcommand, dispatch, // return the process exit code. Exit codes below 2 (pass/violations) belong to -// `check` alone (§20.3, P10); every failure reachable from here — a missing -// subcommand, a bad flag, a transport or auth failure — is a could-not-check -// condition and maps to 2, never to 0 or 1 (H7). +// `check` alone; every failure reachable from here — a missing subcommand, a +// bad flag, a transport or auth failure — is a could-not-check condition and +// maps to 2, never to 0 or 1. // // Requested help (-h/--help) is not a failure — it is the one exception to // that rule. Convention (and every stdlib flag.FlagSet default) is exit 0 to diff --git a/grafana-alertcheck/cmd/grafana-alertcheck/table.go b/grafana-alertcheck/cmd/grafana-alertcheck/table.go index 0e4c93c67..11b72a074 100644 --- a/grafana-alertcheck/cmd/grafana-alertcheck/table.go +++ b/grafana-alertcheck/cmd/grafana-alertcheck/table.go @@ -10,27 +10,25 @@ import ( "github.com/smartcontractkit/chainlink-testing-framework/grafana-alertcheck/internal/gate" ) -// renderTable is §20.2's required human table. It always writes to the -// writer it is given, which the caller (runCheck) always points at -// stderr — the human table is not the machine output §20.2 reserves stdout -// for. +// renderTable is the human table. It always writes to the writer it is given, +// which the caller (runCheck) always points at stderr — stdout is reserved for +// the machine-readable --output json. // // Three sections, in order: // // 1. one line per rule: outcome, BadFor, pollEvery, proved-or-not with the // largest gap; -// 2. one line per Violation (R2): a rule's worst-of outcome does not carry -// the State/Health of the instance that actually caused it — Violation -// does — so this is also where those two columns appear, sorted after -// the rule table rather than folded into it, and it is the only place an -// operator running WITHOUT --output json sees the §12.1 --allow-paused -// hint that Violation.Note already carries (classify.go); -// 3. a footer with the per-rule thresholds and the run-wide numbers §20.2 -// says are the answer to "why" on exit 2: each non-skipped rule's -// maxGap/healthGrace/evalStaleAfter, the global transitionGrace and +// 2. one line per Violation: a rule's worst-of outcome does not carry the +// State/Health of the instance that actually caused it — Violation does — +// so this is also where those two columns appear, sorted after the rule +// table rather than folded into it, and it is the only place an operator +// running WITHOUT --output json sees the --allow-paused hint that +// Violation.Note already carries (classify.go); +// 3. a footer with the numbers that answer "why" on exit 2: each non-skipped +// rule's maxGap/healthGrace/evalStaleAfter, the global transitionGrace and // drainTimeout, and the largest measured clock skew alongside its own -// error bound (RTT/2) — SkewHardLimit is a separate, fixed input -// threshold and is reported next to it, never as if it were that bound. +// error bound (RTT/2) — SkewHardLimit is a separate, fixed input threshold +// and is reported next to it, never as if it were that bound. func renderTable(w io.Writer, res gate.Result) error { alertOf := make(map[string]string, len(res.Verdicts)) for _, v := range res.Verdicts { @@ -77,7 +75,7 @@ func renderTable(w io.Writer, res gate.Result) error { // provedLabel is the table's PROVED column: "yes" for a clean coverage // proof, "no" with the reason and largest gap for an unobservable rule, and // "-" for a rule decide never asked proveCoverage about at all (skipped — -// paused before the window opened, §12). +// paused before the window opened). func provedLabel(cov gate.CoverageResult) string { if cov.Reason == "" && !cov.Unobservable && !cov.Proved { return "-" diff --git a/grafana-alertcheck/cmd/grafana-alertcheck/table_test.go b/grafana-alertcheck/cmd/grafana-alertcheck/table_test.go index 106760fbb..e585e9b61 100644 --- a/grafana-alertcheck/cmd/grafana-alertcheck/table_test.go +++ b/grafana-alertcheck/cmd/grafana-alertcheck/table_test.go @@ -9,10 +9,9 @@ import ( "github.com/smartcontractkit/chainlink-testing-framework/grafana-alertcheck/internal/gate" ) -// TestRenderTable is the golden table test: a fixed Result renders a -// deterministic, ordered rule table, a violations section (R2) and a footer -// carrying the per-rule and global thresholds plus the skew and its bound -// (R3) — with no live Check involved. +// The golden table test: a fixed Result renders a deterministic, ordered rule +// table, a violations section and a footer carrying the per-rule and global +// thresholds plus the skew and its bound — with no live Check involved. func TestRenderTable(t *testing.T) { gapAt := time.Date(2026, 1, 1, 12, 0, 0, 0, time.UTC) res := gate.Result{ @@ -64,13 +63,13 @@ func TestRenderTable(t *testing.T) { t.Fatalf("out = %q, want Zebra's clean row", out) } - // Violations section (R2): must show up even without --output json, and - // must carry the §12.1 --allow-paused hint text verbatim. + // The violations section must show up even without --output json, and must + // carry the --allow-paused hint text verbatim. if !strings.Contains(out, "VIOLATIONS") { t.Fatalf("out = %q, want a VIOLATIONS section", out) } if !strings.Contains(out, "--allow-paused") { - t.Fatalf("out = %q, want the §12.1 --allow-paused hint in the human table", out) + t.Fatalf("out = %q, want the --allow-paused hint in the human table", out) } if !strings.Contains(out, "STATE") || !strings.Contains(out, "HEALTH") { t.Fatalf("out = %q, want the violations table to have STATE and HEALTH columns", out) @@ -79,8 +78,8 @@ func TestRenderTable(t *testing.T) { t.Fatalf("out = %q, want Ape's violation State/Health", out) } - // Footer (R3): per-rule thresholds, global thresholds, and skew with its - // own bound rather than the fixed hard limit. + // The footer: per-rule thresholds, global thresholds, and skew with its own + // bound rather than the fixed hard limit. if !strings.Contains(out, "Ape Alert: maxGap=1m0s healthGrace=2m0s evalStaleAfter=1m0s") { t.Fatalf("out = %q, want Ape's per-rule thresholds", out) } @@ -88,7 +87,7 @@ func TestRenderTable(t *testing.T) { t.Fatalf("out = %q, want Zebra's per-rule thresholds", out) } if strings.Contains(out, "Paused Alert: maxGap") { - t.Fatalf("out = %q, a skipped rule must not report thresholds it never had (§12)", out) + t.Fatalf("out = %q, a skipped rule must not report thresholds it never had", out) } if !strings.Contains(out, "global: transitionGrace=5m0s (source: Ape Alert (for=5m)) drainTimeout=2m0s") { t.Fatalf("out = %q, want the global thresholds line", out) @@ -104,9 +103,9 @@ func TestRenderTable(t *testing.T) { } } -// TestProvedLabel_Skipped pins the "-" case: a rule decide never asked -// proveCoverage about (paused before the window opened, §12) has an empty -// CoverageResult and must not be reported as either proved or unobservable. +// The "-" case: a rule decide never asked proveCoverage about (paused before +// the window opened) has an empty CoverageResult and must not be reported as +// either proved or unobservable. func TestProvedLabel_Skipped(t *testing.T) { if got := provedLabel(gate.CoverageResult{}); got != "-" { t.Fatalf("provedLabel(zero value) = %q, want \"-\"", got) diff --git a/grafana-alertcheck/cmd/grafana-alertcheck/watch.go b/grafana-alertcheck/cmd/grafana-alertcheck/watch.go index f3ad57d92..df2c50029 100644 --- a/grafana-alertcheck/cmd/grafana-alertcheck/watch.go +++ b/grafana-alertcheck/cmd/grafana-alertcheck/watch.go @@ -15,19 +15,17 @@ const watchUsage = "usage: grafana-alertcheck watch --out [--pidfile F] [ // runWatch is the record step's entire CLI surface, split in two by one flag // set — gate.DaemonChildFlag ("--daemon-child") and gate.ReadyFDFlag -// ("--ready-fd") select which side of P6's parent/child split this -// invocation is: +// ("--ready-fd") select which side of the parent/child split this invocation +// is: // // - without them: the record command an operator types. It parses --out, // --alerts and the rest, builds a gate.WatchConfig and calls gate.Watch, // which resolves, records the first observation of every rule, and -// detaches the recorder before returning (§4.3). +// detaches the recorder before returning. // - with them: the detached recorder itself. gate.Watch's own childArgs // (watch_unix.go) is the only thing that ever sets them — an operator -// never types "--daemon-child" and it does not appear in watchUsage — -// and this dispatches straight to gate.RunDaemonChild. P6's integration -// test already covers the spawn; this is the one new test P10 owns: that -// seeing the flag reaches RunDaemonChild. +// never types "--daemon-child" and it does not appear in watchUsage — and +// this dispatches straight to gate.RunDaemonChild. // // Both flags live in the SAME flag set as the operator-facing ones rather // than a second, hidden set: the child is started with childArgs' exact @@ -106,11 +104,10 @@ func runWatch(args []string, stdin io.Reader, stdout, stderr io.Writer) int { return 0 } -// runDaemonChild is the detached recorder's whole entry point (P6's -// obligation on this phase). Its stdout and stderr are already the daemon -// log file — spawnChild (watch_unix.go) redirects both before Start — so -// writing to stderr here lands exactly where waitForChildReady's failure -// path quotes from. +// runDaemonChild is the detached recorder's whole entry point. Its stdout and +// stderr are already the daemon log file — spawnChild (watch_unix.go) +// redirects both before Start — so writing to stderr here lands exactly where +// waitForChildReady's failure path quotes from. func runDaemonChild(out, until string, concurrency, readyFD int, stderr io.Writer) int { url, token, err := grafanaEnv() if err != nil { diff --git a/grafana-alertcheck/cmd/grafana-alertcheck/watch_test.go b/grafana-alertcheck/cmd/grafana-alertcheck/watch_test.go index 267a312a2..6d8c8c1a6 100644 --- a/grafana-alertcheck/cmd/grafana-alertcheck/watch_test.go +++ b/grafana-alertcheck/cmd/grafana-alertcheck/watch_test.go @@ -7,9 +7,8 @@ import ( "testing" ) -// TestRunWatch_FlagValidation is the record step's flag-validation matrix. -// Every case fails inside gate.WatchConfig.validate() (P6) or before it, so -// none needs a reachable Grafana. +// The record step's flag-validation matrix. Every case fails inside +// gate.WatchConfig.validate() or before it, so none needs a reachable Grafana. func TestRunWatch_FlagValidation(t *testing.T) { tests := []struct { name string @@ -58,9 +57,8 @@ func TestRunWatch_FlagValidation(t *testing.T) { } } -// TestRunWatch_DaemonChildDispatch pins P6's obligation on this phase: seeing -// gate.DaemonChildFlag must dispatch to gate.RunDaemonChild, and the flag -// must never appear in watchUsage (an operator never types it). +// Seeing gate.DaemonChildFlag must dispatch to gate.RunDaemonChild, and the +// flag must never appear in watchUsage (an operator never types it). func TestRunWatch_DaemonChildDispatch(t *testing.T) { t.Setenv("GRAFANA_URL", "http://example.invalid") t.Setenv("GRAFANA_TOKEN", "test-token") diff --git a/grafana-alertcheck/internal/gate/check.go b/grafana-alertcheck/internal/gate/check.go index 710bedef9..0f21b4d86 100644 --- a/grafana-alertcheck/internal/gate/check.go +++ b/grafana-alertcheck/internal/gate/check.go @@ -11,46 +11,29 @@ import ( "time" ) -// Obligations this phase leaves for P10, carried forward the way P6 and P7 -// carried theirs so a later review has something concrete to check against: +// Check returns (Result, error) and no exit code: the code is a presentation +// decision the CLI makes. err != nil is exit 2 unconditionally, even alongside +// real violations; violations with err == nil is exit 1; neither is exit 0. // -// - Exit codes are the CLI's (§20.3, §9.1, H6/H7). Check returns -// (Result, error) and nothing else: err != nil is exit 2 unconditionally, -// never 0 and never 1, even alongside real violations; len(Violations) > 0 -// with err == nil is exit 1; both empty is exit 0. Check deliberately does -// not return a code, because a code is a presentation decision and the -// library must not make it. -// - §12.1 wants the paused rule AND --allow-paused both named to the user. -// decide names both in the shortfall Violation's Note (classify.go), and -// P10's renderer prints every Violation, including Note, in the human -// table — not only in --output json. -// - Config.Notes carries the running commentary (§13.2's planned run time, -// the countdown, the blind-interval warning). §20.2 puts the human output -// on stderr and reserves stdout for --output json, so the CLI must pass -// stderr here. -// - Check never reads the environment. GRAFANA_URL and GRAFANA_TOKEN are -// read by the CLI and passed in as fields, and the token must never reach -// a *flag.FlagSet (§20.2). +// Check never reads the environment either. The URL and token are read by the +// CLI and passed in as fields, and the token must never reach a *flag.FlagSet. // countdownEvery is how often the collection loop reports what it is waiting -// for (§13.2: "then print a countdown at regular intervals"). A silent wait is -// indistinguishable from a hung process, and the wait after `to` is the -// longest silence in the whole run. +// for. A silent wait is indistinguishable from a hung process, and the wait +// after `to` is the longest silence in the whole run. const countdownEvery = 30 * time.Second -// recorderStopTimeout bounds §4.4 step 3, the wait for the recorder's exit. -// Not in the source plan's table of values — a judgment call, on the same -// reasoning as childReadyTimeout (P6): everything the recorder does after -// SIGTERM is local (finish the in-flight write, append the sentinel, fsync) -// and an in-flight poll aborts through the child's own context, so the real -// figure is milliseconds. Loose enough for an overloaded runner, and a -// timeout is a hard error rather than a longer wait — a log a writer may -// still hold cannot be read at all (§4.4 step 4). +// recorderStopTimeout bounds the wait for the recorder's exit. Everything the +// recorder does after SIGTERM is local (finish the in-flight write, append the +// sentinel, fsync) and an in-flight poll aborts through the child's own +// context, so the real figure is milliseconds; this is loose enough for an +// overloaded runner. The timeout is a hard error rather than a longer wait — a +// log a writer may still hold cannot be read at all. const recorderStopTimeout = 30 * time.Second // recorderStopPoll is how often that wait re-checks the pid. There is no // wait(2) available: the recorder is a detached session leader, not this -// process's child (P6), so its exit can only be observed by polling. +// process's child, so its exit can only be observed by polling. const recorderStopPoll = 100 * time.Millisecond // Config is check's whole input. It is the CLI's view of a run, and it is @@ -59,14 +42,13 @@ const recorderStopPoll = 100 * time.Millisecond // never cross that line. type Config struct { // URL and Token are the connection details, read from the environment by - // the CLI and never registered as flags (§20.2). Token never enters the - // pure layer, an error string, or a Result. + // the CLI and never registered as flags. Token never enters the pure layer, + // an error string, or a Result. URL, Token string // Alerts is REQUIRED in single-step mode and must be EMPTY in log mode: - // with a log, the header IS the alert set (§19.1 step 3), and there is - // nothing to compare a second list against. Both directions are encoded, - // resolving the source plan's §19.1 step 1 / step 3 contradiction. + // with a log, the header IS the alert set, and there is nothing to compare + // a second list against. Alerts []string Folder string @@ -76,17 +58,16 @@ type Config struct { AllowPaused bool NodataIsUnobservable bool - // From is the moment the deploy finished and To is the end of the work - // (§7). They are different moments and both come from the work. In - // recorder mode an absent From is a hard error; in single-step mode it - // falls back to the start of this step, with the blind-interval warning - // §4.2 requires. + // From is the moment the deploy finished and To is the end of the work. + // They are different moments and both come from the work. In recorder mode + // an absent From is a hard error; in single-step mode it falls back to the + // start of this step, with a blind-interval warning. From, To time.Time // Log is the path of a recording made by watch; "" selects single-step // mode. PidFile defaults to .pid, the convention watch's parent - // writes (P6) and the only way check can reach the recorder it must stop - // before it may read the log (§4.4 steps 1-4). + // writes and the only way check can reach the recorder it must stop before + // it may read the log. Log string PidFile string @@ -94,16 +75,15 @@ type Config struct { // --poll-interval flag. In log mode the cadence comes from the header — // the cadence the recording actually used — and a second authority would // let an operator silently widen maxGap over evidence that was recorded at - // a different rate (P5, "two authorities"); in single-step mode the same - // process records and classifies, so §5's default is the only cadence - // there is. + // a different rate; in single-step mode the same process records and + // classifies, so the default cadence is the only cadence there is. Concurrency int Clock Clock // Notes is where the shell prints what an operator has to see while the // run is in progress: the planned run time, the grace and its source, the // countdown, the blind-interval warning. nil discards them. The library - // renders no table — the CLI owns presentation (§20.2). + // renders no table — the CLI owns presentation. Notes io.Writer } @@ -123,7 +103,7 @@ func (cfg Config) withDefaults() Config { return cfg } -// namedAlerts returns the alert names that survive §17.3's trim-and-discard, +// namedAlerts returns the alert names that survive Resolve's trim-and-discard, // so validation counts what Resolve will actually see rather than what the // caller happened to pass (a file ending in a newline yields an empty line). func (cfg Config) namedAlerts() []string { @@ -139,12 +119,12 @@ func (cfg Config) namedAlerts() []string { // Check is the I/O shell: HTTP, signals, the pidfile, file reads, the // countdown print. Every correctness question it touches is answered // elsewhere — by proveCoverage and decide, which are pure — and that split is -// the most important seam in the project (§2). Check therefore needs two +// the most important seam in the project. Check therefore needs two // integration tests; decide carries the suite. // -// H7 governs the return: a pass is exactly len(Violations) == 0 && err == nil. -// Every error path below leaves err non-nil, and no path anywhere in this file -// converts an error into an empty Result with a nil error. +// A pass is exactly len(Violations) == 0 && err == nil. Every error path below +// leaves err non-nil, and no path anywhere in this file converts an error into +// an empty Result with a nil error. func Check(ctx context.Context, cfg Config) (Result, error) { cfg = cfg.withDefaults() if err := cfg.validate(); err != nil { @@ -152,32 +132,31 @@ func Check(ctx context.Context, cfg Config) (Result, error) { } // The Source is built here and injected into check() so every behaviour // below is testable against a scripted fake — the same seam prepareWatch - // uses (P6), and the reason this file needs no test-only setter. + // uses, and the reason this file needs no test-only setter. return check(ctx, cfg, NewHTTPSource(cfg.URL, cfg.Token, cfg.Clock)) } -// validate is §19.1 step 1. It runs before any network call, so a -// configuration mistake costs nothing and, more importantly, is never -// discovered after a ten-minute wait. +// validate runs before any network call, so a configuration mistake costs +// nothing and, more importantly, is never discovered after a ten-minute wait. func (cfg Config) validate() error { if cfg.URL == "" { return errors.New("check: no grafana url") } if cfg.To.IsZero() { - return errors.New("check: no `to`: the end of the window is required (§7)") + return errors.New("check: no `to`: the end of the window is required") } named := cfg.namedAlerts() if cfg.Log == "" { - // §19.1 step 1: an empty Alerts is an error — but only without a log. + // An empty Alerts is an error — but only without a log. if len(named) == 0 { return errors.New("check: no alert names given and no recorded log to take them from") } } else if len(named) > 0 { - // §19.1 step 3, the other direction: the alert set comes from the log. - // Accepting both would mean reconciling two sets, which is the subset - // arithmetic the source plan removes by making the log the one source. - return fmt.Errorf("check: --alerts is refused with a recorded log: %s already names the alert set it recorded (§19.1 step 3)", cfg.Log) + // The other direction: with a log, the alert set comes from the log. + // Accepting both would mean reconciling two sets, which the log being + // the one source removes entirely. + return fmt.Errorf("check: --alerts is refused with a recorded log: %s already names the alert set it recorded", cfg.Log) } now := cfg.Clock.Now() @@ -188,13 +167,13 @@ func (cfg Config) validate() error { from := cfg.From switch { case from.IsZero() && cfg.Log != "": - // §7, and never a warning-and-continue: falling back to the start of - // the check step reinstates exactly the blind interval the recorder - // exists to remove, which is the fail-open shape this design refuses. - return errors.New("check: no `from` in recorder mode: the deploy step must emit a completion timestamp (§7)") + // Never a warning-and-continue: falling back to the start of the check + // step reinstates exactly the blind interval the recorder exists to + // remove, which is the fail-open shape this design refuses. + return errors.New("check: no `from` in recorder mode: the deploy step must emit a completion timestamp") case from.IsZero(): // Single-step only. The caller sees the resulting blind interval named - // exactly, once the first observation has fixed its end (§4.2). + // exactly, once the first observation has fixed its end. from = now } @@ -202,40 +181,39 @@ func (cfg Config) validate() error { return fmt.Errorf("check: `to` %s is before `from` %s", cfg.To.Format(time.RFC3339), from.Format(time.RFC3339)) } if from.After(now.Add(fromFutureTolerance)) { - return fmt.Errorf("check: `from` %s is more than %s ahead of this runner's clock %s (§7)", + return fmt.Errorf("check: `from` %s is more than %s ahead of this runner's clock %s", from.Format(time.RFC3339), fromFutureTolerance, now.Format(time.RFC3339)) } - // A `to` already in the past is not a special mode WITH a log (§7, §24.3): - // the collection loop's condition is simply already true and the evidence - // is classified immediately. Without one it is a different thing entirely - // — a request to prove a window that nothing observed. Refusing it is not + // A `to` already in the past is not a special mode WITH a log: the + // collection loop's condition is simply already true and the evidence is + // classified immediately. Without one it is a different thing entirely — a + // request to prove a window that nothing observed. Refusing it is not // pedantry: the coverage window would end before the first observation, // every heartbeat gap inside it would measure negative, and the run would // report a proved window it never saw. if cfg.Log == "" && !cfg.To.After(now) { - return fmt.Errorf("check: `to` %s has already passed and there is no recorded log: a window that ended before check started can only be classified from a recording (§4.2)", + return fmt.Errorf("check: `to` %s has already passed and there is no recorded log: a window that ended before check started can only be classified from a recording", cfg.To.Format(time.RFC3339)) } return nil } -// check is Check with the Source injected. Its body is §19.1 steps 1-9, one -// commented block each and in that order, so a review can diff it against the -// source plan line by line. +// check is Check with the Source injected, and its body is one commented block +// per stage of a run, in the order a run performs them. func check(ctx context.Context, cfg Config, src Source) (Result, error) { - // ---- §19.1 step 1 — validate the configuration. ----------------------- + // ---- Validate the configuration. -------------------------------------- // Done by Check before this function is reached, except for the one part // that needs a clock reading kept for later: the single-step fallback for // an absent `from`. from := cfg.From if from.IsZero() { from = cfg.Clock.Now() - fmt.Fprintf(cfg.Notes, "note: no `from` given; the window starts at the start of this step, %s (§4.2)\n", + fmt.Fprintf(cfg.Notes, "note: no `from` given; the window starts at the start of this step, %s\n", from.Format(time.RFC3339)) } - // ---- §19.1 step 2 — resolve the definitions from the ruler API. ------- + // ---- Resolve the definitions from the ruler API. ---------------------- // Unconditional, in BOTH modes. A log's header supplies the alert set as // UIDs and the recording facts, never the rule facts: `for`, // intervalSeconds and Kind always come from a fresh ruler read, which is @@ -249,17 +227,16 @@ func check(ctx context.Context, cfg Config, src Source) (Result, error) { } allDefs, err := src.Definitions(ctx) if err != nil { - // §19.3 case 2: resolution of the definitions failed. return Result{}, fmt.Errorf("read rule definitions: %w", err) } - // ---- §19.1 step 3 — with a log, validate its identity. ---------------- + // ---- With a log, validate its identity. ------------------------------- // The header is read early — line 1 only, the one line a writer can never // change (ReadLogHeader) — so a wrong URL or a rule that no longer // resolves fails closed NOW rather than after the whole window has // elapsed. It is advisory: the authoritative header comes from the single - // full ReadLog in step 6, after the writer has exited, and the identity is - // validated again against that one. + // full ReadLog once collection is over and the writer has exited, and the + // identity is validated again against that one. var ( resolved []Definition notes []string @@ -272,7 +249,6 @@ func check(ctx context.Context, cfg Config, src Source) (Result, error) { if cfg.Log != "" { earlyHdr, err = ReadLogHeader(cfg.Log) if err != nil { - // §19.3 case 3. return Result{}, fmt.Errorf("log identity: %w", err) } logHasHdr = true @@ -293,12 +269,12 @@ func check(ctx context.Context, cfg Config, src Source) (Result, error) { fmt.Fprintf(cfg.Notes, "note: %s\n", n) } - // ---- §19.1 step 4 — derive the timings, print the plan, fit the budget. + // ---- Derive the timings, print them, fit the request budget. ---------- if logHasHdr { // The header is the authority for the cadence actually recorded at; // re-deriving it from defs would compare gaps recorded at an override // cadence against thresholds computed from the default — fail-open in - // the faster-override direction (P5). + // the faster-override direction. rt, gt, err = DeriveTimingsFromLog(earlyHdr, resolved) if err != nil { return Result{}, fmt.Errorf("log identity: %w", err) @@ -316,10 +292,10 @@ func check(ctx context.Context, cfg Config, src Source) (Result, error) { } // The measurement pass and the budget check belong to single-step mode - // alone (§5.2): in recorder mode watch already took one observation of - // every rule and checked the budget against those measured latencies - // before it detached, and repeating it here would spend a second poll of - // every rule to re-answer a question already answered. + // alone: in recorder mode watch already took one observation of every rule + // and checked the budget against those measured latencies before it + // detached, and repeating it here would spend a second poll of every rule + // to re-answer a question already answered. var ( header Header initial []Poll @@ -328,8 +304,8 @@ func check(ctx context.Context, cfg Config, src Source) (Result, error) { if !logHasHdr { // StartedAt is fixed before the pass rather than after it, so the // interval it claims to have observed can only be wider than the one - // it really saw — and the first heartbeat's own boundary gap (P7 check - // 3) is what proves that interval, not this timestamp. + // it really saw — and the first heartbeat's own boundary gap is what + // proves that interval, not this timestamp. startedAt := cfg.Clock.Now() active := activeRules(resolved) var measured map[string]time.Duration @@ -341,10 +317,10 @@ func check(ctx context.Context, cfg Config, src Source) (Result, error) { return Result{}, err } - // Single-step synthesis — how the pure layer stays unconditional (§2). - // The shell builds the Header and later stamps the sentinel itself, so - // P7 checks 1 and 2 run exactly as they do over a recording and no - // mode flag ever reaches proveCoverage or decide. + // Single-step synthesis — how the pure layer stays unconditional. The + // shell builds the Header and later stamps the sentinel itself, so the + // sentinel and from-bounds coverage checks run exactly as they do over + // a recording and no mode flag ever reaches proveCoverage or decide. header = Header{ SchemaVersion: LogSchemaVersion, URL: cfg.URL, @@ -353,31 +329,31 @@ func check(ctx context.Context, cfg Config, src Source) (Result, error) { Rules: loggedRules(resolved, rt), } if from.Before(startedAt) { - // §4.2/§22.4's declared blind interval: in single-step mode this - // is a warning and a pass, and ONLY here. Recorder mode keeps P7 - // check 2 strict (§22.9), because there the recorder was supposed - // to be watching and the gap means it was not. - fmt.Fprintf(cfg.Notes, "warning: cannot see [%s, %s) — %s before the first observation; the window is classified from %s (§4.2)\n", + // The declared blind interval: in single-step mode this is a + // warning and a pass, and ONLY here. Recorder mode keeps the + // from-bounds coverage check strict, because there the recorder + // was supposed to be watching and the gap means it was not. + fmt.Fprintf(cfg.Notes, "warning: cannot see [%s, %s) — %s before the first observation; the window is classified from %s\n", from.Format(time.RFC3339), startedAt.Format(time.RFC3339), startedAt.Sub(from).Round(time.Second), startedAt.Format(time.RFC3339)) from = startedAt } } - // ---- §19.1 step 5 — apply MinObserved. -------------------------------- - // Its default is the resolved rule count AFTER the collapse (§17.3), which - // is len(resolved) by construction. decide defaults it identically; it is - // resolved here as well so the value the run will judge against is printed - // before the wait rather than inferred from the verdict afterwards. + // ---- Apply MinObserved. ----------------------------------------------- + // Its default is the resolved rule count AFTER duplicate names collapse, + // which is len(resolved) by construction. decide defaults it identically; + // it is resolved here as well so the value the run will judge against is + // printed before the wait rather than inferred from the verdict afterwards. minObserved := cfg.MinObserved if minObserved == 0 { minObserved = len(resolved) } fmt.Fprintf(cfg.Notes, "min-observed: %d of %d resolved rule(s)\n", minObserved, len(resolved)) - // ---- §19.1 step 6 — collect the evidence. ----------------------------- + // ---- Collect the evidence. -------------------------------------------- // Collect ONLY. No classification happens here and there is no early exit, - // even once a violation is certain (H5, §19.2): the loop always runs to + // even once a violation is certain: the loop always runs to // to + transitionGrace, which is what makes "did the early exit lose the // coverage proof?" a question that cannot be asked. windowEnd := cfg.To.Add(gt.transitionGrace) @@ -388,11 +364,10 @@ func check(ctx context.Context, cfg Config, src Source) (Result, error) { } collected, err := collectUntil(ctx, cfg, windowEnd, poller) if err != nil { - // §19.3 case 1: the failure limit was exceeded (retryTransport already - // gave every transient failure its backoff), or the context ended. - // Nothing collected is classified — the count is there so an operator - // can tell a run that failed at once from one that failed at minute - // nine. + // The failure limit was exceeded (retryTransport already gave every + // transient failure its backoff), or the context ended. Nothing + // collected is classified — the count is there so an operator can tell + // a run that failed at once from one that failed at minute nine. return Result{}, fmt.Errorf("collect evidence after %d poll(s): %w", len(collected), err) } @@ -401,10 +376,10 @@ func check(ctx context.Context, cfg Config, src Source) (Result, error) { sentinel *time.Time ) if logHasHdr { - // §4.4 steps 2-4, in this order and no other: signal the writer, wait - // for its exit, and only THEN read the log once. A log read while a - // writer can still append can only yield a shorter window than the one - // that was actually recorded. + // In this order and no other: signal the writer, wait for its exit, + // and only THEN read the log once. A log read while a writer can still + // append can only yield a shorter window than the one that was + // actually recorded. heldLog, err := stopRecorder(ctx, cfg) if err != nil { return Result{}, err @@ -442,23 +417,23 @@ func check(ctx context.Context, cfg Config, src Source) (Result, error) { polls = append(polls, collected...) // The shell stamps the sentinel itself, when the collection loop // exits: by construction that is at or after to + transitionGrace, so - // P7 check 1 passes for the same reason a clean recorder stop does, - // and for no other. + // the sentinel check passes for the same reason a clean recorder stop + // does, and for no other. stoppedAt := cfg.Clock.Now() sentinel = &stoppedAt } - // ---- §19.1 step 7 — the drain wait. ----------------------------------- - // The last instance of the liveness check (§14.6): did this rule evaluate - // through the end of the window? It is I/O and it is deliberately NOT part - // of proveCoverage — adding it there would put HTTP inside the pure layer - // and destroy the seam §2 depends on. + // ---- The drain wait. -------------------------------------------------- + // The last instance of the liveness check: did this rule evaluate through + // the end of the window? It is I/O and it is deliberately NOT part of + // proveCoverage — adding it there would put HTTP inside the pure layer and + // destroy the seam this design depends on. drained, err := drainWait(ctx, cfg, src, resolved, header.pausedAtStart(), rt, polls, windowEnd, gt.drainTimeout) if err != nil { return Result{}, err } - // ---- §19.1 step 8 — classify. ----------------------------------------- + // ---- Classify. -------------------------------------------------------- pol := Policy{ States: cfg.States, Preexisting: cfg.Preexisting, @@ -471,33 +446,34 @@ func check(ctx context.Context, cfg Config, src Source) (Result, error) { result, decideErr := decide(header, polls, sentinel, resolved, rt, gt, pol) result, drainErr := mergeDrainTimeouts(result, drained) - // ---- §19.1 step 9 — return Result. ------------------------------------ + // ---- Return the Result. ----------------------------------------------- // Both errors are joined rather than one shadowing the other: each names // rules the other does not, and on exit 2 that list IS the answer to - // "why". H7 needs only that err be non-nil when either fired. + // "why". return result, errors.Join(decideErr, drainErr) } -// resolveFromLog turns a log header into the resolved definitions, and is -// §19.1 step 3's identity check in practice. Three things are verified: the -// URL matches, the schema version matches (ReadLog/ReadLogHeader own that), -// and every header UID still resolves against the fresh ruler read. The alert -// set is TAKEN from the log, never compared — with Alerts required empty in -// log mode there is nothing to compare it against, and §22.4's "different -// alert set" refusal is exactly this URL-and-UID failure. +// resolveFromLog turns a log header into the resolved definitions, and is the +// log's identity check in practice. Three things are verified: the URL +// matches, the schema version matches (ReadLog/ReadLogHeader own that), and +// every header UID still resolves against the fresh ruler read. The alert set +// is TAKEN from the log, never compared — with Alerts required empty in log +// mode there is nothing to compare it against, and refusing a log recorded +// against a different alert set is exactly this URL-and-UID failure. // // Resolving through Resolve, by uid:, rather than by a private lookup, keeps -// one implementation of §17: a header naming a recording or datasource-managed -// rule gets the same specific refusal an operator would, and a header naming -// the same UID twice collapses with a note (DeriveTimingsFromLog rejects that -// case outright, so the note is belt and braces). +// one implementation of the resolution rules: a header naming a recording or +// datasource-managed rule gets the same specific refusal an operator would, +// and a header naming the same UID twice collapses with a note +// (DeriveTimingsFromLog rejects that case outright, so the note is belt and +// braces). // // Only the header-to-defs direction needs checking. The opposite direction // cannot fail here: resolved is BUILT from the header, so no resolved // definition can be absent from it. func resolveFromLog(allDefs []Definition, h Header, cfg Config) ([]Definition, []string, error) { if h.URL != cfg.URL { - return nil, nil, fmt.Errorf("log identity: %s recorded url %q but this run is configured for %q (§19.1 step 3)", + return nil, nil, fmt.Errorf("log identity: %s recorded url %q but this run is configured for %q", cfg.Log, h.URL, cfg.URL) } names := make([]string, 0, len(h.Rules)) @@ -506,16 +482,15 @@ func resolveFromLog(allDefs []Definition, h Header, cfg Config) ([]Definition, [ } resolved, notes, err := Resolve(allDefs, names, "") if err != nil { - return nil, nil, fmt.Errorf("log identity: %s names a rule that no longer resolves: %w (§19.1 step 3)", cfg.Log, err) + return nil, nil, fmt.Errorf("log identity: %s names a rule that no longer resolves: %w", cfg.Log, err) } return resolved, notes, nil } -// activeRules drops the rules whose DEFINITION says paused. They are skipped -// (§12): never polled, never waited for, and reported from the definitions -// alone — a skipped rule has no poll records at all, so it has no heartbeats -// to prove and no IsPaused poll to detect (P6 deviation 4, and the obligation -// it left P7/P8). +// activeRules drops the rules whose DEFINITION says paused. They are skipped: +// never polled, never waited for, and reported from the definitions alone — a +// skipped rule has no poll records at all, so it has no heartbeats to prove +// and no IsPaused poll to detect. func activeRules(defs []Definition) []Definition { out := make([]Definition, 0, len(defs)) for _, d := range defs { @@ -527,8 +502,9 @@ func activeRules(defs []Definition) []Definition { } // activeTimingsOf narrows the timings map to the rules that will actually be -// polled, which is what §5.2's budget is spent on: a skipped rule consumes -// none of the capacity, so counting it would refuse schedules that fit. +// polled, which is what the request budget is spent on: a skipped rule +// consumes none of the capacity, so counting it would refuse schedules that +// fit. func activeTimingsOf(active []Definition, rt map[string]ruleTimings) map[string]ruleTimings { out := make(map[string]ruleTimings, len(active)) for _, d := range active { @@ -538,14 +514,14 @@ func activeTimingsOf(active []Definition, rt map[string]ruleTimings) map[string] } // livePoller is single-step mode's collection engine: the same per-rule -// scheduler and the same Reducer the recorder uses (P4/P6), writing into -// memory instead of a log. Log mode has none — the recorder is doing this -// work in another process — and collectUntil takes a nil poller for it. +// scheduler and the same Reducer the recorder uses, writing into memory +// instead of a log. Log mode has none — the recorder is doing this work in +// another process — and collectUntil takes a nil poller for it. type livePoller struct { src Source reducer *Reducer sched *Scheduler - titles map[string]string // uid -> title: poll by title, select by UID (§14.5) + titles map[string]string // uid -> title: poll by title, select by UID concurrency int } @@ -573,10 +549,10 @@ func newLivePoller(src Source, reducer *Reducer, active []Definition, rt map[str // The successes are NOT kept for the reason watchLoopConfig.pollBatch keeps // its own: those go into a durable log that a later check will read, so // dropping one would turn a single rule's transport failure into a coverage -// gap for the others. Here there is no later reader. A terminal failure -// during collection is exit 2 (§19.3 case 1) and check discards the whole -// collection, so these come back only to let the error say how far the run -// got before it stopped — which is the one part of it an operator can act on. +// gap for the others. Here there is no later reader. A terminal failure during +// collection is exit 2 and check discards the whole collection, so these come +// back only to let the error say how far the run got before it stopped — +// which is the one part of it an operator can act on. func (p *livePoller) poll(ctx context.Context, uids []string) ([]Poll, error) { observed, obsErr := observeAll(ctx, p.src, p.titles, uids, p.concurrency) out := make([]Poll, 0, len(uids)) @@ -590,13 +566,13 @@ func (p *livePoller) poll(ctx context.Context, uids []string) ([]Poll, error) { return out, obsErr } -// collectUntil is §19.1 step 6's loop, shared by both modes. With a poller it +// collectUntil is the collection loop, shared by both modes. With a poller it // polls each rule on its own cadence; with nil it only waits, because in // recorder mode the evidence is being written by another process. Both print // the same countdown, because both are the same silence to an operator -// watching a job (§13.2). +// watching a job. // -// It never classifies and never exits early (H5). +// It never classifies and never exits early. func collectUntil(ctx context.Context, cfg Config, deadline time.Time, p *livePoller) ([]Poll, error) { var ( polls []Poll @@ -643,9 +619,9 @@ func collectUntil(ctx context.Context, cfg Config, deadline time.Time, p *livePo } } -// stopRecorder is §4.4 steps 2 and 3. Nothing here is best-effort: the log may -// not be read until the writer has provably gone, so every failure to reach -// that state is a hard error. +// stopRecorder signals the recorder and waits for it to go. Nothing here is +// best-effort: the log may not be read until the writer has provably gone, so +// every failure to reach that state is a hard error. // // It returns the log held under an exclusive flock. The caller must keep that // file open across ReadLog and close it afterwards — the lock is the proof @@ -654,12 +630,11 @@ func collectUntil(ctx context.Context, cfg Config, deadline time.Time, p *livePo // // Two authorities, and only one of them is evidence: // -// - The PIDFILE says whether a recording was ever started. An absent or -// unparseable one is the load-bearing case, and P6's obligation on this -// phase: it must never read as "there was nothing to stop". The parent -// writes the pidfile only AFTER the child reports that it holds the log -// and is polling, and removes it on every failing path, so a missing one -// means watch failed and this run has no evidence at all. +// - The PIDFILE says whether a recording was ever started, and an absent or +// unparseable one must never read as "there was nothing to stop". The +// parent writes the pidfile only AFTER the child reports that it holds the +// log and is polling, and removes it on every failing path, so a missing +// one means watch failed and this run has no evidence at all. // - The FLOCK says whether a writer exists RIGHT NOW. Nothing removes the // pidfile when a recorder exits cleanly — the parent has long returned and // the child never learns the path — so after a --until run, a supported @@ -673,7 +648,7 @@ func collectUntil(ctx context.Context, cfg Config, deadline time.Time, p *livePo func stopRecorder(ctx context.Context, cfg Config) (*os.File, error) { pid, err := ReadPidFile(cfg.PidFile) if err != nil { - return nil, fmt.Errorf("cannot stop the recorder: %w; a pidfile is written only once a recorder reports that it is running, so an unreadable one means the recording never started (§4.4)", err) + return nil, fmt.Errorf("cannot stop the recorder: %w; a pidfile is written only once a recorder reports that it is running, so an unreadable one means the recording never started", err) } log, err := os.Open(cfg.Log) @@ -690,7 +665,7 @@ func stopRecorder(ctx context.Context, cfg Config) (*os.File, error) { // No writer. Send no signal, whatever the pidfile says — the pid may // belong to somebody else entirely by now. Which of --until, a clean // stop and a death ended the recording is the sentinel's question, - // answered by P7 check 1 over the log this unblocks. + // answered by the coverage proof over the log this unblocks. fmt.Fprintf(cfg.Notes, "note: no writer holds %s; the recorder has already finished\n", cfg.Log) return log, nil } @@ -732,7 +707,7 @@ func stopRecorder(ctx context.Context, cfg Config) (*os.File, error) { } if !cfg.Clock.Now().Before(deadline) { log.Close() - return nil, fmt.Errorf("recorder pid %d still holds %s %s after SIGTERM; refusing to read a log a writer can still append to (§4.4 step 4)", + return nil, fmt.Errorf("recorder pid %d still holds %s %s after SIGTERM; refusing to read a log a writer can still append to", pid, cfg.Log, recorderStopTimeout) } } @@ -743,34 +718,33 @@ func stopRecorder(ctx context.Context, cfg Config) (*os.File, error) { // are genuinely different faults: drain_timeout means the rule is still there // and still behind, rule_absent means it is gone. Collapsing both into // drain_timeout would name the wait instead of the fault, and Reason is a -// published vocabulary that reaches the action's JSON (§19.0). +// published vocabulary that reaches the JSON output. type drainVerdict struct { reason UnobservableReason note string } -// drainWait is §19.1 step 7 and §14.6: the final instance of the liveness -// check, asking each rule the last question — did you evaluate through the end -// of the window? A rule that cannot answer within drainTimeout is -// unobservable, never a pass. +// drainWait is the final instance of the liveness check, asking each rule the +// last question — did you evaluate through the end of the window? A rule that +// cannot answer within drainTimeout is unobservable, never a pass. // // It returns one verdict per rule it could not clear, keyed by UID, which the // caller folds into the Result. It returns an error only for a hard failure of -// the wait itself (§19.3 case 1); a rule that simply never catches up is -// reported, not raised. +// the wait itself; a rule that simply never catches up is reported, not +// raised. // -// Two rules are excluded from the wait before it starts, and both are -// exclusions of work that could not change a verdict: +// Two kinds of rule are excluded before the wait starts, both because draining +// them could not change a verdict: // -// - a rule the HEADER says was already paused when the recording opened -// (§12): it is skipped, it was not evaluating, and it never was — there is -// no evaluation to wait for. The header and not the definition, for -// decide's reason (Header.pausedAtStart): a rule the header says was -// active must be drained or faulted, because a pause somebody applied -// after the window is not evidence about the window; -// - a rule whose last poll says Found == false: P7 check 8 already makes it -// unobservable, so the only thing draining it could add is drainTimeout of -// waiting before the same answer. +// - a rule the HEADER says was already paused when the recording opened: it +// is skipped, it was not evaluating, and it never was — there is no +// evaluation to wait for. The header and not the definition, for decide's +// reason (Header.pausedAtStart): a rule the header says was active must be +// drained or faulted, because a pause somebody applied after the window is +// not evidence about the window; +// - a rule whose last poll says Found == false: the rule-absent coverage +// check already makes it unobservable, so the only thing draining it could +// add is drainTimeout of waiting before the same answer. func drainWait(ctx context.Context, cfg Config, src Source, defs []Definition, pausedAtStart map[string]bool, rt map[string]ruleTimings, polls []Poll, windowEnd time.Time, timeout time.Duration) (map[string]drainVerdict, error) { @@ -817,15 +791,16 @@ func drainWait(ctx context.Context, cfg Config, src Source, defs []Definition, p } rule := stateRuleByUID(obs.Rules, uid) if rule == nil { - // §14.5: a 2xx that parsed and carries no matching rule is an - // authoritative "the rule is gone" — P2 retried every transport - // failure long before this Observation existed. It is knowable - // on the FIRST poll, so waiting the rest of drainTimeout would - // spend two minutes to reach the same verdict under a name that - // describes the wait rather than the fault. + // A 2xx that parsed and carries no matching rule is an + // authoritative "the rule is gone" — the transport retried + // every transient failure long before this Observation + // existed. It is knowable on the FIRST poll, so waiting the + // rest of drainTimeout would spend two minutes to reach the + // same verdict under a name that describes the wait rather + // than the fault. verdicts[uid] = drainVerdict{ reason: ReasonRuleAbsent, - note: fmt.Sprintf("rule %q: absent from the state endpoint during the drain wait; there is no evaluation to wait for (§14.5)", + note: fmt.Sprintf("rule %q: absent from the state endpoint during the drain wait; there is no evaluation to wait for", pending[uid]), } delete(pending, uid) @@ -835,11 +810,11 @@ func drainWait(ctx context.Context, cfg Config, src Source, defs []Definition, p // A paused rule does not evaluate, so this one can never catch // up and the rest of drainTimeout would buy nothing. The reason // stays drain_timeout: UnobservableReason is a published - // vocabulary that reaches the action's JSON (§19.0), and the - // prose below is where the detail belongs. + // vocabulary that reaches the JSON output, and the prose below + // is where the detail belongs. verdicts[uid] = drainVerdict{ reason: ReasonDrainTimeout, - note: fmt.Sprintf("rule %q: paused before it evaluated through %s, so it never will (§14.8)", + note: fmt.Sprintf("rule %q: paused before it evaluated through %s, so it never will", pending[uid], windowEnd.Format(time.RFC3339)), } delete(pending, uid) @@ -858,7 +833,7 @@ func drainWait(ctx context.Context, cfg Config, src Source, defs []Definition, p for uid, title := range pending { verdicts[uid] = drainVerdict{ reason: ReasonDrainTimeout, - note: fmt.Sprintf("rule %q: did not evaluate through %s within the %s drain limit (§19.1 step 7)", + note: fmt.Sprintf("rule %q: did not evaluate through %s within the %s drain limit", title, windowEnd.Format(time.RFC3339), timeout), } } @@ -867,8 +842,8 @@ func drainWait(ctx context.Context, cfg Config, src Source, defs []Definition, p // Re-ask no faster than the tightest cadence among the rules still // pending: a rule evaluating every 60s cannot answer differently 200ms - // later, and hammering it would spend the request budget §5 accounts - // for on nothing. + // later, and hammering it would spend the run's request budget on + // nothing. wait := deadline.Sub(now) for uid := range pending { if every := rt[uid].pollEvery; every > 0 { @@ -897,16 +872,16 @@ func anyPollEvaluatedThrough(polls []Poll, windowEnd time.Time) bool { return false } -// evaluatedThrough is the drain wait's one comparison, and it is cross-domain -// (§16): lastEvaluation is a Grafana timestamp and windowEnd is runner-domain, -// so the Grafana value is translated by its own poll's skew. The skew BOUND is -// then subtracted rather than added — the pessimistic end of the uncertainty — -// so an evaluation that only might have reached the end of the window does not +// evaluatedThrough is the drain wait's one comparison, and it is cross-domain: +// lastEvaluation is a Grafana timestamp and windowEnd is runner-domain, so the +// Grafana value is translated by its own poll's skew. The skew BOUND is then +// subtracted rather than added — the pessimistic end of the uncertainty — so +// an evaluation that only might have reached the end of the window does not // count as one that did. Understating it costs a few more seconds of waiting; // overstating it would pass an unproven window. // // A zero lastEvaluation never satisfies the wait: only a paused rule may -// legitimately report it (§2.3), and a paused rule has nothing to drain. +// legitimately report it, and a paused rule has nothing to drain. func evaluatedThrough(lastEval time.Time, skew, bound time.Duration, windowEnd time.Time) bool { if lastEval.IsZero() { return false @@ -915,19 +890,15 @@ func evaluatedThrough(lastEval time.Time, skew, bound time.Duration, windowEnd t } // mergeDrainTimeouts folds the I/O drain wait's verdicts into the pure layer's -// Result. P7 places this merge "before decide runs"; it cannot be, because -// decide owns proveCoverage and therefore builds the Coverage map itself — so -// the merge happens immediately after, which is the same thing from every -// caller's point of view and keeps decide's signature a pure function of its -// arguments. +// Result. It runs immediately after decide rather than before it, because +// decide owns proveCoverage and therefore builds the Coverage map itself; that +// keeps decide a pure function of its arguments. // // It returns its own error rather than mutating decide's, so neither hides the // other: a run with one rule unobservable from the coverage proof and another -// from the drain wait must name both (H6 — inability beats violation, and it -// beats a second inability being dropped from the message too). The error says -// "at the drain wait" for that reason: the two are joined into one message, and -// two counts under one identical phrase read as a contradiction rather than as -// two findings. +// from the drain wait must name both. The error says "at the drain wait" for +// that reason — the two are joined into one message, and two counts under one +// identical phrase read as a contradiction rather than as two findings. func mergeDrainTimeouts(res Result, drained map[string]drainVerdict) (Result, error) { if len(drained) == 0 { return res, nil diff --git a/grafana-alertcheck/internal/gate/check_process.go b/grafana-alertcheck/internal/gate/check_process.go index 580a2a143..86dd73148 100644 --- a/grafana-alertcheck/internal/gate/check_process.go +++ b/grafana-alertcheck/internal/gate/check_process.go @@ -6,7 +6,7 @@ import ( "syscall" ) -// signalRecorder asks the recorder to stop (§4.4 step 2). +// signalRecorder asks the recorder to stop. // // The caller must have established that a writer is alive — by taking the // log's flock and being refused — before it calls this. Nothing removes the diff --git a/grafana-alertcheck/internal/gate/check_test.go b/grafana-alertcheck/internal/gate/check_test.go index 8cd94468f..d13914a8e 100644 --- a/grafana-alertcheck/internal/gate/check_test.go +++ b/grafana-alertcheck/internal/gate/check_test.go @@ -21,7 +21,7 @@ import ( // from those three numbers, and the tests assert against them by name rather // than by magic constant: // -// pollEvery 30s (§5: intervalSeconds/2) +// pollEvery 30s (intervalSeconds/2) // maxGap 60s (2 x pollEvery) // healthGrace 60s (max(maxGap, interval)) // evalStaleAfter 120s (2 x interval) @@ -99,9 +99,9 @@ var _ Source = (*checkSource)(nil) // checkStateRule builds one state-endpoint rule whose totals agree with the // instances it carries. That agreement is load-bearing: a totals map claiming -// normal instances that the instance list does not contain fails §3.2's -// verification (VerifyNormalInstancesVisible), which is a different failure -// from the one most of these tests are about. +// normal instances that the instance list does not contain fails +// VerifyNormalInstancesVisible, which is a different failure from the one most +// of these tests are about. func checkStateRule(lastEval time.Time, insts ...Instance) StateRule { totals := map[string]int{} for _, i := range insts { @@ -143,7 +143,7 @@ func baseConfig(t *testing.T, clock Clock) Config { func notesOf(cfg Config) string { return cfg.Notes.(*strings.Builder).String() } // --------------------------------------------------------------------------- -// §19.1 step 1 — configuration validation +// Configuration validation // --------------------------------------------------------------------------- func TestCheckValidateRejectsBadConfigurations(t *testing.T) { @@ -173,7 +173,7 @@ func TestCheckValidateRejectsBadConfigurations(t *testing.T) { wantErr: "no `to`", }, { - // §19.1 step 1: an empty Alerts is an error — but only without a log. + // An empty Alerts is an error — but only without a log. name: "single-step without alerts", mutate: func(c *Config) {}, wantErr: "no alert names given", @@ -185,13 +185,13 @@ func TestCheckValidateRejectsBadConfigurations(t *testing.T) { wantErr: "no alert names given", }, { - // §19.1 step 3, the other direction: the log names the alert set. + // The other direction: with a log, the log names the alert set. name: "log mode with alerts", mutate: func(c *Config) { c.Log = "log.jsonl"; c.Alerts = []string{"A"} }, wantErr: "--alerts is refused with a recorded log", }, { - // §7 — never a warning-and-continue. + // Never a warning-and-continue. name: "log mode without from", mutate: func(c *Config) { c.Log = "log.jsonl"; c.From = time.Time{} }, wantErr: "the deploy step must emit a completion timestamp", @@ -238,9 +238,9 @@ func TestCheckValidateRejectsBadConfigurations(t *testing.T) { } } -// A past `to` WITH a log is explicitly not a special mode (§7, §24.3): the -// collection loop's condition is already true and the evidence classifies -// immediately. No branch, and no refusal. +// A past `to` WITH a log is not a special mode: the collection loop's condition +// is already true and the evidence classifies immediately. No branch, and no +// refusal. func TestCheckValidateAcceptsAPastToWithALog(t *testing.T) { cfg := Config{ URL: "https://grafana.example.com", @@ -273,7 +273,7 @@ func TestCheckSingleStepCleanWindowPasses(t *testing.T) { if err != nil { t.Fatalf("check() = %v, want nil\nnotes:\n%s", err, notesOf(cfg)) } - // H7: a pass is exactly this shape. + // A pass is exactly this shape. if len(res.Violations) != 0 { t.Fatalf("Violations = %+v, want none", res.Violations) } @@ -284,7 +284,7 @@ func TestCheckSingleStepCleanWindowPasses(t *testing.T) { t.Fatalf("Coverage = %+v, want proved", cov) } - // The collection loop ran to to+transitionGrace and no further (H5). + // The collection loop ran to to+transitionGrace and no further. windowEnd := cfg.To.Add(checkGrace) if clock.Now().Before(windowEnd) { t.Errorf("stopped collecting at %s, before to+grace %s", clock.Now(), windowEnd) @@ -297,15 +297,13 @@ func TestCheckSingleStepCleanWindowPasses(t *testing.T) { t.Errorf("polled %d times, want at least the ~13 a full 6-minute window at 30s implies", got) } if notes := notesOf(cfg); !strings.Contains(notes, "planned run time") { - t.Errorf("§13.2 requires the planned run time at start; notes were:\n%s", notes) + t.Errorf("the planned run time must be printed at start; notes were:\n%s", notes) } } -// §22.2: the collapse-note-plus-satisfied-MinObserved path (resolve_test.go's -// TestResolve_CollapseByUIDGivesNoteNotError and -// TestResolve_MinObservedCountIsPostCollapse) is proven only at Resolve() -// directly; this drives the same shape through check() end to end — the two -// input names must collapse to one verdict, the run must pass, and the +// resolve_test.go proves the collapse-note-plus-satisfied-MinObserved path at +// Resolve() directly; this drives the same shape through check() end to end — +// the two input names must collapse to one verdict, the run must pass, and the // collapse note must reach the run's own notes, not just Resolve()'s return // value. func TestCheckSingleStepDuplicateAlertNamesCollapseWithNote(t *testing.T) { @@ -331,11 +329,11 @@ func TestCheckSingleStepDuplicateAlertNamesCollapseWithNote(t *testing.T) { } } -// §22.1's highest-priority regression: a rule with health=error for the -// whole window is unobservable, exit 2 — using the real "[JD] No Job -// Proposals" capture (testdata/README.md), not a synthetic Poll table, so a -// change in how the real payload shapes health/lastError cannot slip past a -// hand-built fixture that happens to still look right. +// A rule with health=error for the whole window is unobservable, exit 2 — +// driven from the real "[JD] No Job Proposals" capture (testdata/README.md), +// not a synthetic Poll table, so a change in how the real payload shapes +// health/lastError cannot slip past a hand-built fixture that happens to still +// look right. func TestCheckSingleStepContinuousHealthErrorIsUnobservable(t *testing.T) { body := readFixture(t, "state_health_error.json") rules, err := ParseState(body) @@ -358,8 +356,8 @@ func TestCheckSingleStepContinuousHealthErrorIsUnobservable(t *testing.T) { src := newCheckSource(func(_ string, _ int) (Observation, error) { // Every field but LastEvaluation stays exactly as the real capture // shaped it (health=error, the real lastError text, the real Error - // instance); LastEvaluation tracks the poll so staleness (a - // different coverage check, §14) never becomes the actual cause. + // instance); LastEvaluation tracks the poll so staleness — a + // different coverage check — never becomes the actual cause. r := base r.LastEvaluation = clock.Now() return Observation{Rules: []StateRule{r}, GrafanaNow: clock.Now(), Latency: 200 * time.Millisecond}, nil @@ -368,7 +366,7 @@ func TestCheckSingleStepContinuousHealthErrorIsUnobservable(t *testing.T) { res, err := check(context.Background(), cfg, src) if err == nil { - t.Fatalf("check() = nil, want an error: continuous health=error must be unobservable (§22.1, H6/H7)\nnotes:\n%s", notesOf(cfg)) + t.Fatalf("check() = nil, want an error: continuous health=error must be unobservable\nnotes:\n%s", notesOf(cfg)) } if len(res.Verdicts) != 1 || res.Verdicts[0].Outcome != OutcomeUnobservable { t.Fatalf("Verdicts = %+v, want one unobservable verdict", res.Verdicts) @@ -378,8 +376,8 @@ func TestCheckSingleStepContinuousHealthErrorIsUnobservable(t *testing.T) { } } -// H5: a certain violation does not release the runner early, and it does not -// stop the gate reporting exit-1 shape — violations with a nil error. +// A certain violation does not release the runner early, and it does not stop +// the gate reporting exit-1 shape — violations with a nil error. func TestCheckSingleStepFiringInstanceReportsWithoutExitingEarly(t *testing.T) { clock := newVirtualClock(testNow) cfg := baseConfig(t, clock) @@ -403,15 +401,14 @@ func TestCheckSingleStepFiringInstanceReportsWithoutExitingEarly(t *testing.T) { t.Errorf("Outcome = %q, want %q", got, OutcomePersistentlyBad) } if windowEnd := cfg.To.Add(checkGrace); clock.Now().Before(windowEnd) { - t.Errorf("exited early at %s; H5 requires collecting to %s", clock.Now(), windowEnd) + t.Errorf("exited early at %s; collection must run to %s", clock.Now(), windowEnd) } } -// §22.8: "newly_bad at from+30s gives exit 1, but ONLY after -// to+transition_grace." The test above pins H5 for a rule already bad -// before the window opened (persistently_bad); this pins the anti-fail-fast -// case the plan names explicitly — a fresh onset just inside the window -// must not release the runner the instant it is first observed. +// A newly_bad instance at from+30s gives exit 1, but ONLY after +// to+transitionGrace. The test above covers a rule already bad before the +// window opened (persistently_bad); this covers a fresh onset just inside the +// window, which must not release the runner the instant it is first observed. func TestCheckSingleStepNewOnsetDoesNotExitEarly(t *testing.T) { clock := newVirtualClock(testNow) cfg := baseConfig(t, clock) @@ -437,13 +434,13 @@ func TestCheckSingleStepNewOnsetDoesNotExitEarly(t *testing.T) { t.Fatalf("Violations = %+v, want exactly one newly_bad", res.Violations) } if windowEnd := cfg.To.Add(checkGrace); clock.Now().Before(windowEnd) { - t.Errorf("exited early at %s; H5 requires collecting to %s even for a fresh onset at from+30s", clock.Now(), windowEnd) + t.Errorf("exited early at %s; collection must run to %s even for a fresh onset at from+30s", clock.Now(), windowEnd) } } -// §22.9: an ABSENT `from` in single-step mode (as opposed to recorder mode, -// which hard-errors — TestCheckValidateRejectsBadConfigurations's "log mode -// without from") falls back to the start of this check step, with the same +// An ABSENT `from` in single-step mode (as opposed to recorder mode, which +// hard-errors — TestCheckValidateRejectsBadConfigurations's "log mode without +// from") falls back to the start of this check step, with the same // declared-blind-interval warning as an explicit early `from`. func TestCheckSingleStepAbsentFromFallsBackToStepStart(t *testing.T) { clock := newVirtualClock(testNow) @@ -459,16 +456,16 @@ func TestCheckSingleStepAbsentFromFallsBackToStepStart(t *testing.T) { } notes := notesOf(cfg) if !strings.Contains(notes, "no `from` given") { - t.Errorf("want the §4.2 fallback note; notes were:\n%s", notes) + t.Errorf("want the step-start fallback note; notes were:\n%s", notes) } if !res.From.Equal(testNow) { t.Errorf("Result.From = %s, want the step-start fallback %s", res.From, testNow) } } -// §4.2/§22.4: in single-step mode an explicit `from` earlier than the first -// observation is a DECLARED blind interval — a warning and a pass, naming the -// exact interval it cannot see. Recorder mode keeps P7 check 2 strict. +// In single-step mode an explicit `from` earlier than the first observation is +// a DECLARED blind interval — a warning and a pass, naming the exact interval +// it cannot see. Recorder mode keeps the from-bounds coverage check strict. func TestCheckSingleStepFromBeforeFirstObservationWarnsAndPasses(t *testing.T) { clock := newVirtualClock(testNow) cfg := baseConfig(t, clock) @@ -492,9 +489,9 @@ func TestCheckSingleStepFromBeforeFirstObservationWarnsAndPasses(t *testing.T) { } } -// §19.3 case 1: the failure limit was exceeded. The measurement pass succeeds -// and the collection loop then hits a terminal failure, so this exercises the -// path a live run really takes. +// The failure limit was exceeded. The measurement pass succeeds and the +// collection loop then hits a terminal failure, so this exercises the path a +// live run really takes. func TestCheckFailClosedOnExhaustedRetries(t *testing.T) { clock := newVirtualClock(testNow) cfg := baseConfig(t, clock) @@ -517,8 +514,8 @@ func TestCheckFailClosedOnExhaustedRetries(t *testing.T) { } } -// §19.3 case 2: the resolution of the definitions failed. Both shapes — the -// ruler read itself failing, and a name that resolves to nothing. +// The resolution of the definitions failed. Both shapes — the ruler read +// itself failing, and a name that resolves to nothing. func TestCheckFailClosedOnDefinitionResolution(t *testing.T) { t.Run("ruler read fails", func(t *testing.T) { clock := newVirtualClock(testNow) @@ -545,8 +542,8 @@ func TestCheckFailClosedOnDefinitionResolution(t *testing.T) { }) } -// The version gate (§2.7 control 2): an unsupported Grafana is exit 2 before -// anything else is attempted. +// The version gate: an unsupported Grafana is exit 2 before anything else is +// attempted. func TestCheckRefusesUnsupportedGrafanaVersion(t *testing.T) { clock := newVirtualClock(testNow) cfg := baseConfig(t, clock) @@ -559,9 +556,9 @@ func TestCheckRefusesUnsupportedGrafanaVersion(t *testing.T) { } } -// §5.2: the budget is checked against the latencies the measurement pass -// actually measured, and a schedule that cannot fit errors at START rather -// than producing a gap-riddled recording nobody can classify. +// The budget is checked against the latencies the measurement pass actually +// measured, and a schedule that cannot fit errors at START rather than +// producing a gap-riddled recording nobody can classify. func TestCheckSingleStepRefusesAScheduleThatDoesNotFit(t *testing.T) { clock := newVirtualClock(testNow) cfg := baseConfig(t, clock) @@ -577,7 +574,7 @@ func TestCheckSingleStepRefusesAScheduleThatDoesNotFit(t *testing.T) { } for _, want := range []string{"raising concurrency", "raising poll-interval", "watching fewer alerts"} { if !strings.Contains(err.Error(), want) { - t.Errorf("err = %q, want it to name the control %q (§5.1)", err, want) + t.Errorf("err = %q, want it to name the control %q", err, want) } } } @@ -691,16 +688,15 @@ func TestCheckRecorderModeCleanWindowPasses(t *testing.T) { if res.GrafanaVersion != "13.1.0" { t.Errorf("GrafanaVersion = %q, want the recorded one", res.GrafanaVersion) } - // The collection loop still waited out to+transitionGrace (H5) even though - // the recorder had already finished. + // The collection loop still waited out to+transitionGrace even though the + // recorder had already finished. if clock.Now().Before(windowEnd) { t.Errorf("returned at %s, before to+grace %s", clock.Now(), windowEnd) } } -// §19.3 case 3: the identity of the log is not correct. The check runs against -// the header read EARLY, so it fails before the window's wait rather than -// after it. +// The identity of the log is not correct. The check runs against the header +// read EARLY, so it fails before the window's wait rather than after it. func TestCheckFailClosedOnWrongLogIdentity(t *testing.T) { t.Run("different url", func(t *testing.T) { dir := t.TempDir() @@ -736,8 +732,8 @@ func TestCheckFailClosedOnWrongLogIdentity(t *testing.T) { }) } -// §19.3 case 4: the coverage proof failed. A hole in the middle of the -// recording is not saved by healthy data at both ends (§22.4). +// The coverage proof failed: a hole in the middle of the recording is not +// saved by healthy data at both ends. func TestCheckFailClosedOnCoverageGap(t *testing.T) { dir := t.TempDir() windowEnd := testNow.Add(5*time.Minute + checkGrace) @@ -782,13 +778,12 @@ func TestCheckFailClosedOnCoverageGap(t *testing.T) { } } -// §22.4: "an episode fully between the deploy and the start of the check" — -// recorder mode must find this at the LEADING edge of the window too, right -// after `from` (the deploy's completion), not only in the middle -// (TestCheckFailClosedOnCoverageGap above). No poll exists for -// [from, from+3m): whatever happened there is invisible to every per-poll -// check, so only the coverage gap itself can catch it — the reason this -// two-phase recorder model exists at all (§4.2). +// An episode fully between the deploy and the start of the check: recorder +// mode must find this at the LEADING edge of the window too, right after `from` +// (the deploy's completion), not only in the middle +// (TestCheckFailClosedOnCoverageGap above). No poll exists for [from, from+3m), +// so whatever happened there is invisible to every per-poll check and only the +// coverage gap itself can catch it — the reason the recorder exists at all. func TestCheckRecorderModeFindsAGapImmediatelyAfterTheDeploy(t *testing.T) { dir := t.TempDir() windowEnd := testNow.Add(5*time.Minute + checkGrace) @@ -827,14 +822,14 @@ func TestCheckRecorderModeFindsAGapImmediatelyAfterTheDeploy(t *testing.T) { } } -// §19.3 case 5: the drain limit passed. The recording itself is clean, so this -// isolates the drain wait — the rule simply never evaluates through the end of -// the window, and a rule that cannot answer that question is unobservable. +// The drain limit passed. The recording itself is clean, so this isolates the +// drain wait — the rule simply never evaluates through the end of the window, +// and a rule that cannot answer that question is unobservable. func TestCheckFailClosedOnDrainTimeout(t *testing.T) { dir := t.TempDir() windowEnd := testNow.Add(5*time.Minute + checkGrace) - // A 45s lag keeps every poll inside evalStaleAfter (120s), so P7 check 6 - // is silent and only the drain wait can fail. + // A 45s lag keeps every poll inside evalStaleAfter (120s), so the liveness + // coverage check is silent and only the drain wait can fail. logPath := recordedLog(t, dir, "https://grafana.example.com", testNow.Add(-time.Minute), testNow.Add(-time.Minute), windowEnd, windowEnd.Add(30*time.Second), 45*time.Second) writePid(t, logPath+".pid", fmt.Sprintf("%d\n", deadPid(t))) @@ -868,8 +863,8 @@ func TestCheckFailClosedOnDrainTimeout(t *testing.T) { } } -// §14.5: a rule the state endpoint no longer serves is knowable on the FIRST -// drain poll, and the answer is rule_absent — the fault — rather than +// A rule the state endpoint no longer serves is knowable on the FIRST drain +// poll, and the answer is rule_absent — the fault — rather than // drain_timeout, which would only name the wait. It must not spend the whole // drain limit to reach it. func TestCheckDrainWaitNamesADeletedRuleAtOnce(t *testing.T) { @@ -881,9 +876,9 @@ func TestCheckDrainWaitNamesADeletedRuleAtOnce(t *testing.T) { clock := newVirtualClock(testNow) cfg := recorderConfig(t, clock, logPath) - // An authoritative 2xx that parsed and carries no matching rule. P2 - // retried every transport failure long before an Observation exists, so - // this is a deletion, not a hiccup. + // An authoritative 2xx that parsed and carries no matching rule. The + // transport retried every transient failure long before an Observation + // exists, so this is a deletion, not a hiccup. src := newCheckSource(func(_ string, _ int) (Observation, error) { return Observation{GrafanaNow: clock.Now()}, nil }) @@ -1016,8 +1011,7 @@ func TestCheckHeaderPausedRuleStaysSkipped(t *testing.T) { t.Fatalf("NewWriter: %v", err) } // Named in the header, is_paused true, and no poll records at all — the - // shape watch writes for a rule paused before the window opened (P6 - // deviation 4). + // shape watch writes for a rule paused before the window opened. if err := w.WriteHeader(Header{ URL: "https://grafana.example.com", GrafanaVersion: "13.1.0", StartedAt: testNow.Add(-time.Minute), Rules: []LoggedRule{{ @@ -1054,7 +1048,7 @@ func TestCheckHeaderPausedRuleStaysSkipped(t *testing.T) { t.Errorf("Coverage[%s] present, want absent: a skipped rule has no coverage to prove", checkUID) } if len(res.Violations) != 1 { - t.Errorf("Violations = %+v, want the MinObserved shortfall (§12.1)", res.Violations) + t.Errorf("Violations = %+v, want the MinObserved shortfall", res.Violations) } res, err = run(true) @@ -1089,7 +1083,7 @@ func TestCheckDrainWaitConcludesAtOnceOnAPausedRule(t *testing.T) { t.Errorf("polled %d times, want exactly 1: a paused rule can never catch up", got) } if got := res.Coverage[checkUID].Reason; got != ReasonDrainTimeout { - t.Errorf("Reason = %q, want %q — the vocabulary is published (§19.0), so the detail goes in the note", got, ReasonDrainTimeout) + t.Errorf("Reason = %q, want %q — the vocabulary is published, so the detail goes in the note", got, ReasonDrainTimeout) } if !strings.Contains(res.Verdicts[0].Note, "paused before it evaluated through") { t.Errorf("Note = %q, want it to say the rule was paused", res.Verdicts[0].Note) @@ -1099,10 +1093,10 @@ func TestCheckDrainWaitConcludesAtOnceOnAPausedRule(t *testing.T) { } } -// P6's obligation on this phase: an absent or unparseable pidfile is never -// "there was nothing to stop". The parent writes the pidfile only once the -// child reports that it is recording, so a missing one means the recording -// never started — and the log must not be read at all. +// An absent or unparseable pidfile is never "there was nothing to stop". The +// parent writes the pidfile only once the child reports that it is recording, +// so a missing one means the recording never started — and the log must not be +// read at all. func TestCheckRefusesToReadALogItCannotStop(t *testing.T) { windowEnd := testNow.Add(5*time.Minute + checkGrace) @@ -1159,8 +1153,8 @@ func startLockHolder(t *testing.T, logPath string) int { return cmd.Process.Pid } -// §4.4 step 4: a recorder that will not let go of the log means the log may -// still be appended to, and a log a writer can change cannot be read at all. +// A recorder that will not let go of the log means the log may still be +// appended to, and a log a writer can change cannot be read at all. func TestCheckFailsWhenTheRecorderWillNotExit(t *testing.T) { dir := t.TempDir() windowEnd := testNow.Add(5*time.Minute + checkGrace) @@ -1209,9 +1203,9 @@ func TestCheckDoesNotSignalABystanderHoldingAReusedPid(t *testing.T) { } } -// §22.5: a dead pidfile (the recorder process has already exited, holding no -// flock) with NO sentinel in the log — the shape a killed `watch` leaves -// behind — must not hang the stop wait: the flock is free immediately, so +// A dead pidfile (the recorder process has already exited, holding no flock) +// with NO sentinel in the log — the shape a killed `watch` leaves behind — +// must not hang the stop wait: the flock is free immediately, so // check reads the log at once, finds no sentinel, and fails closed. func TestCheckDeadPidWithNoSentinelIsUnobservable(t *testing.T) { dir := t.TempDir() @@ -1254,12 +1248,11 @@ func TestCheckDeadPidWithNoSentinelIsUnobservable(t *testing.T) { } } -// §22.5: "an incomplete last line gives exit 2" is otherwise proven only -// indirectly — log_test.go's TestReadLogRejectsBadLogs pins ReadLog's own -// error, and TestExitCode pins that any non-nil error maps to exit 2 — but -// nothing feeds a genuinely truncated log through check() itself. This closes -// that seam: a raw file with a valid header and poll, then a torn JSON tail, -// exactly what a recorder killed mid-write leaves behind. +// An incomplete last line gives exit 2. log_test.go's TestReadLogRejectsBadLogs +// pins ReadLog's own error and TestExitCode pins that any non-nil error maps to +// exit 2, but only this feeds a genuinely truncated log through check() itself: +// a raw file with a valid header and poll, then a torn JSON tail, exactly what +// a recorder killed mid-write leaves behind. func TestCheckRecorderModeTruncatedLogFailsClosed(t *testing.T) { dir := t.TempDir() path := filepath.Join(dir, "log.jsonl") @@ -1291,10 +1284,10 @@ func TestCheckRecorderModeTruncatedLogFailsClosed(t *testing.T) { } } -// P5's "two authorities", from check's side: maxGap comes from the cadence the -// header records, never from a re-derivation off intervalSeconds. The -// fail-open direction is the one asserted — a log recorded at 5s on a 60s rule -// must still fail on a hole a re-derived 30s maxGap would have forgiven. +// One authority for the cadence, from check's side: maxGap comes from the +// cadence the header records, never from a re-derivation off intervalSeconds. +// The fail-open direction is the one asserted — a log recorded at 5s on a 60s +// rule must still fail on a hole a re-derived 30s maxGap would have forgiven. func TestCheckDerivesMaxGapFromTheRecordedCadence(t *testing.T) { dir := t.TempDir() windowEnd := testNow.Add(5*time.Minute + checkGrace) @@ -1341,8 +1334,8 @@ func TestCheckDerivesMaxGapFromTheRecordedCadence(t *testing.T) { // The pieces, in isolation // --------------------------------------------------------------------------- -// The drain wait's one comparison is cross-domain (§16), and its uncertainty -// is spent in the fail-closed direction: an evaluation that only MIGHT have +// The drain wait's one comparison is cross-domain, and its uncertainty is +// spent in the fail-closed direction: an evaluation that only MIGHT have // reached the end of the window does not count as one that did. func TestEvaluatedThroughSpendsItsUncertaintyFailingClosed(t *testing.T) { end := testNow @@ -1381,8 +1374,8 @@ func TestEvaluatedThroughSpendsItsUncertaintyFailingClosed(t *testing.T) { } } -// H6 through the merge: a drain timeout on one rule and a coverage failure on -// another must both reach the message. Neither error may shadow the other. +// A drain timeout on one rule and a coverage failure on another must both +// reach the message. Neither error may shadow the other. func TestMergeDrainTimeoutsNamesEveryUnobservableRule(t *testing.T) { res := Result{ Coverage: map[string]CoverageResult{ diff --git a/grafana-alertcheck/internal/gate/classify.go b/grafana-alertcheck/internal/gate/classify.go index 76f05edc0..5d28c16ce 100644 --- a/grafana-alertcheck/internal/gate/classify.go +++ b/grafana-alertcheck/internal/gate/classify.go @@ -7,19 +7,19 @@ import ( "time" ) -// ReasonNodata is decide's own unobservable reason (§10.1/§10.2): proveCoverage -// (P7) deliberately never sets it — health=nodata is a note there, never fatal, -// because escalating it needs Policy.NodataIsUnobservable, and the pure -// coverage layer has no Policy to consult (coverage.go, check 5). decide is -// the seam that DOES have a Policy, so the escalation lives here. +// ReasonNodata is decide's own unobservable reason: proveCoverage deliberately +// never sets it — health=nodata is a note there, never fatal, because +// escalating it needs Policy.NodataIsUnobservable, and the pure coverage layer +// has no Policy to consult (coverage.go, check 5). decide is the seam that DOES +// have a Policy, so the escalation lives here. const ReasonNodata UnobservableReason = "nodata" // Outcome is the verdict of one instance's timeline, and — after decide takes -// the worst across a rule's instances — of the rule itself (§9). It is a -// published JSON output (§19.0): the three fail values stay distinct even -// though v1 maps all three to exit 1, because a later reason string cannot -// recover the information a single "fail" value would have thrown away, and -// because splitting them later would break a published interface for no gain. +// the worst across a rule's instances — of the rule itself. It is a published +// JSON output: the three fail values stay distinct even though v1 maps all +// three to exit 1, because a later reason string cannot recover the +// information a single "fail" value would have thrown away, and because +// splitting them later would break a published interface for no gain. type Outcome string const ( @@ -34,16 +34,15 @@ const ( // PreexistingPolicy governs only the ONE ambiguous case in the outcome table: // an instance that was already bad when the window opened. A newly_bad or -// flapping instance is a fail under every policy (§11.3) — the plan lists -// them among the outcomes that "do not change" — so this type only ever +// flapping instance is a fail under every policy, so this type only ever // changes how `recovered` and `persistently_bad` are judged (isViolation // below). type PreexistingPolicy string const ( - // PreexistingFailUnlessRecovered is the default (§11.7): a preexisting - // instance that clears and stays clear is a pass (`recovered`); one that - // never clears is still a fail (`persistently_bad`). + // PreexistingFailUnlessRecovered is the default: a preexisting instance + // that clears and stays clear is a pass (`recovered`); one that never + // clears is still a fail (`persistently_bad`). PreexistingFailUnlessRecovered PreexistingPolicy = "fail-unless-recovered" // PreexistingFail makes ANY preexisting instance a fail, even one that // recovers — for a user who wants no benefit of the doubt for a @@ -61,9 +60,9 @@ type Violation struct { Alert, RuleUID string Outcome Outcome State State - Health string // raw, reporting-only, like Poll.Health (P1.2a) + Health string // raw, reporting-only, like Poll.Health LastError string - // FirstSeen is the episode's onset, in the runner domain (§16): activeAt + // FirstSeen is the episode's onset, in the runner domain: activeAt // translated by its poll's own skew when the episode opened strictly // inside the window, or `from` itself when the instance was already bad // at window-open (preexisting) — never a raw, untranslated Grafana @@ -74,14 +73,14 @@ type Violation struct { ClearedAt time.Time InstanceLabels map[string]string // Note carries an explanation for a Violation that has no instance - // behind it — the synthetic MinObserved shortfall entry decide emits - // when the deficit exceeds what any named paused rule explains (§12). - // LastError is reporting-only rule state from a real poll and must not - // double as a message field for a Violation that never touched one. + // behind it — the synthetic MinObserved shortfall entry decide emits when + // the deficit exceeds what any named paused rule explains. LastError is + // reporting-only rule state from a real poll and must not double as a + // message field for a Violation that never touched one. Note string } -// RuleVerdict is one rule's worst-of outcome (§9), always present for every +// RuleVerdict is one rule's worst-of outcome, always present for every // resolved rule — Verdicts includes the passes, not only the failures — so a // human reading the table sees every alert that was asked for, not only the // ones that misbehaved. @@ -93,9 +92,9 @@ type RuleVerdict struct { Note string } -// Policy is decide's narrowed, pure-layer view of Config/Cfg (§9's P9 -// comment): the classification knobs and the window, nothing else. No URL, -// no token, no I/O handles — those never reach the pure layer. +// Policy is decide's narrowed, pure-layer view of a Config: the classification +// knobs and the window, nothing else. No URL, no token, no I/O handles — those +// never reach the pure layer. type Policy struct { States []State Preexisting PreexistingPolicy @@ -104,37 +103,35 @@ type Policy struct { From, To time.Time } -// RuleThresholds is one non-skipped rule's resolved coverage thresholds -// (§5/§10.1/§14.1), carried on Result so the CLI's table (P10, §20.2) can -// print the numbers that answer "why" on exit 2 without decide exposing the -// unexported ruleTimings type itself. +// RuleThresholds is one non-skipped rule's resolved coverage thresholds, +// carried on Result so the CLI's table can print the numbers that answer "why" +// on exit 2 without decide exposing the unexported ruleTimings type itself. type RuleThresholds struct { MaxGap time.Duration HealthGrace time.Duration EvalStaleAfter time.Duration } -// GlobalThresholds is the run-wide half of the same information (§13.1, -// §19): transitionGrace and drainTimeout apply once, across every -// non-skipped watched rule, not per rule (globalTimings). +// GlobalThresholds is the run-wide half of the same information: +// transitionGrace and drainTimeout apply once, across every non-skipped +// watched rule, not per rule (globalTimings). type GlobalThresholds struct { TransitionGrace time.Duration - // GraceSource names, and already carries the `for` value of, the rule - // that set TransitionGrace (§13.2 requires printing both). "none" when no - // rule contributed (TransitionGrace is then 0). + // GraceSource names, and already carries the `for` value of, the rule that + // set TransitionGrace — an operator has to see both. "none" when no rule + // contributed (TransitionGrace is then 0). GraceSource string DrainTimeout time.Duration } -// Result is decide's whole answer: everything §20.2's table and the action's -// JSON outputs need. Coverage carries one CoverageResult per non-skipped -// rule — no separate Interval type anywhere in the project (§2's -// simplification table). +// Result is decide's whole answer: everything the human table and the JSON +// output need. Coverage carries one CoverageResult per non-skipped rule — +// there is deliberately no separate Interval type anywhere in the project. type Result struct { From, To time.Time GrafanaVersion string ClockSkew time.Duration // the largest |skew| across every poll decide was given, not only the ones a rule's window actually used - // ClockSkewBound is the skew BOUND (RTT/2, §16) of that SAME poll — not + // ClockSkewBound is the skew BOUND (RTT/2) of that SAME poll — not // the largest bound seen overall, which would pair a wide bound from an // unrelated slow request with the worst skew and misstate how tightly // that skew is actually known. SkewHardLimit is a separate, fixed input @@ -143,9 +140,9 @@ type Result struct { ClockSkewBound time.Duration Coverage map[string]CoverageResult // Thresholds carries one RuleThresholds per rule Coverage also covers — - // every non-skipped rule, keyed by UID. A skipped rule has neither: it - // was never scheduled, so it has no maxGap/healthGrace/evalStaleAfter to - // report (§12). + // every non-skipped rule, keyed by UID. A skipped rule has neither: it was + // never scheduled, so it has no maxGap/healthGrace/evalStaleAfter to + // report. Thresholds map[string]RuleThresholds Global GlobalThresholds Verdicts []RuleVerdict @@ -154,19 +151,19 @@ type Result struct { // episode is one contiguous, policy-bad span of one instance's timeline, // already resolved to the runner domain and clamped to [from, windowEnd]. It -// never crosses a genuine Cleared event (H2): a Vanished marker freezes the -// state instead of closing the episode, which is what keeps a vanish from -// ever reading as a recovery. +// never crosses a genuine Cleared event: a Vanished marker freezes the state +// instead of closing the episode, which is what keeps a vanish from ever +// reading as a recovery. type episode struct { start, end time.Time closedByRealClear bool } // instanceTimeline accumulates one instance's walk across a rule's in-window -// polls. preexisting is decided once, the first time this key is seen bad: -// by the translated ActiveAt against `from` (§16), never by which poll -// happened to report it first — a poll's own cadence is not evidence of when -// the condition actually began (F1/F2). +// polls. preexisting is decided once, the first time this key is seen bad: by +// the translated ActiveAt against `from`, never by which poll happened to +// report it first — a poll's own cadence is not evidence of when the condition +// actually began. type instanceTimeline struct { labels map[string]string preexisting bool @@ -179,27 +176,27 @@ type instanceTimeline struct { episodes []episode } -// runnerTime translates a Grafana-domain timestamp recorded on poll p into -// the runner domain, undoing that poll's own measured skew (§16). GrafanaNow -// and ActiveAt come from the same response, so the same poll's skew applies -// to both. This is the single implementation of that translation for the -// package (same drift argument as pollsForRule, F5): coverage.go's window -// membership test and heartbeat boundary segments call it too, rather than -// each keeping its own copy of `p.GrafanaNow.Add(-p.Skew())` that could -// silently diverge from this one. +// runnerTime translates a Grafana-domain timestamp recorded on poll p into the +// runner domain, undoing that poll's own measured skew. GrafanaNow and +// ActiveAt come from the same response, so the same poll's skew applies to +// both. This is the single implementation of that translation for the package +// (same drift argument as pollsForRule): coverage.go's window membership test +// and heartbeat boundary segments call it too, rather than each keeping its own +// copy of `p.GrafanaNow.Add(-p.Skew())` that could silently diverge from this +// one. func runnerTime(p Poll, grafanaDomain time.Time) time.Time { return grafanaDomain.Add(-p.Skew()) } // classifyRule builds every instance timeline for one rule across -// [from, windowEnd] and reduces them to the rule's worst outcome (§9), its -// merged BadFor, and the Violations the preexisting policy actually charges -// against the run. It is PURE: no I/O, no clock reads (§2) — decide supplies -// windowEnd (to + transitionGrace) rather than this function deriving it, so -// a test can pin the boundary directly. +// [from, windowEnd] and reduces them to the rule's worst outcome, its merged +// BadFor, and the Violations the preexisting policy actually charges against +// the run. It is PURE: no I/O, no clock reads — decide supplies windowEnd +// (to + transitionGrace) rather than this function deriving it, so a test can +// pin the boundary directly. // // polls need not be pre-filtered to this rule, matching proveCoverage's own -// contract (§14.5): selection is by def.UID. +// contract: selection is by def.UID. func classifyRule(def Definition, polls []Poll, from, windowEnd time.Time, badStates map[State]bool, pol PreexistingPolicy) (Outcome, time.Duration, []Violation) { rulePolls := pollsForRule(polls, def.UID) inWindow := inWindowPolls(rulePolls, from, windowEnd) @@ -207,8 +204,8 @@ func classifyRule(def Definition, polls []Poll, from, windowEnd time.Time, badSt timelines := make(map[string]*instanceTimeline) order := make([]string, 0) - // get backfills labels the first time a real Instance is seen (F4): a key - // can be created earlier by a bare Cleared/Vanished marker, which carries + // get backfills labels the first time a real Instance is seen: a key can + // be created earlier by a bare Cleared/Vanished marker, which carries // no labels of its own, and the instance later re-firing must not report // an empty InstanceLabels just because of which event happened to create // the timeline first. @@ -233,7 +230,7 @@ func classifyRule(def Definition, polls []Poll, from, windowEnd time.Time, badSt closeEpisode := func(tl *instanceTimeline, end time.Time, real bool) { // inWindowPolls admits a poll whose translated time is up to its own // skew bound PAST windowEnd (the membership test widens the boundary - // outward, §16). Without this clamp a genuine Cleared event on such a + // outward). Without this clamp a genuine Cleared event on such a // poll would produce an episode.end slightly beyond windowEnd, // contradicting the episode type's own "clamped to // [from, windowEnd]" contract. @@ -279,12 +276,12 @@ func classifyRule(def Definition, polls []Poll, from, windowEnd time.Time, badSt case !tl.seen: tl.seen = true if bad { - // Fail-closed (§16): only call an onset "preexisting" - // when even the worst-case skew error still puts it at - // or before `from`. An onset that might really have - // landed just inside the window must classify as a new - // episode, never earn the `recovered` benefit of the - // doubt it would get if it later clears (F1/F2). + // Fail-closed: only call an onset "preexisting" when + // even the worst-case skew error still puts it at or + // before `from`. An onset that might really have landed + // just inside the window must classify as a new episode, + // never earn the `recovered` benefit of the doubt it + // would get if it later clears. activeAtRunner := runnerTime(p, inst.ActiveAt) tl.preexisting = !activeAtRunner.Add(p.SkewBound()).After(from) if tl.preexisting { @@ -318,7 +315,7 @@ func classifyRule(def Definition, polls []Poll, from, windowEnd time.Time, badSt tl.lastHealth, tl.lastError = p.Health, p.LastError } - // Vanished is a deliberate no-op (H2): freeze whatever badOpen/preexisting + // Vanished is a deliberate no-op: freeze whatever badOpen/preexisting // already holds. An instance that vanishes while bad must stay bad, and // one that vanishes while never having been bad must stay uninteresting. for _, key := range p.Vanished { @@ -361,8 +358,8 @@ func classifyRule(def Definition, polls []Poll, from, windowEnd time.Time, badSt } default: // A genuinely new onset always fails, whether or not it later - // clears within the window (§11.4 point 3): only a PREEXISTING - // condition earns the benefit of `recovered`. + // clears within the window: only a PREEXISTING condition earns + // the benefit of `recovered`. instOutcome = OutcomeNewlyBad } @@ -395,9 +392,9 @@ func classifyRule(def Definition, polls []Poll, from, windowEnd time.Time, badSt } // isViolation decides whether one instance's outcome counts against the run, -// once the preexisting policy is applied. newly_bad and flapping always do -// (§11.3): both contain a genuinely new bad episode, so no policy forgives -// them. recovered and persistently_bad are, by classifyRule's construction, +// once the preexisting policy is applied. newly_bad and flapping always do: +// both contain a genuinely new bad episode, so no policy forgives them. +// recovered and persistently_bad are, by classifyRule's construction, // ALWAYS preexisting (a non-preexisting single episode is newly_bad instead, // regardless of whether it clears) — so these are the only two policy can // change, and isViolation needs no separate preexisting flag to know that. @@ -415,15 +412,16 @@ func isViolation(o Outcome, pol PreexistingPolicy) bool { } // outcomeRank orders outcomes for classifyRule's worst-of reduction across a -// rule's instances (§9). The three fail values, and recovered above clean, -// give it exactly the ordering the table requires — -// "unobservable > {flapping, persistently_bad, newly_bad} > recovered > -// skipped > clean" — with unobservable and skipped applied outside this -// function (decide owns both: unobservable from CoverageResult, skipped from -// Definition.IsPaused). The table does not distinguish among the three fail -// values, so their relative order here (flapping above persistently_bad -// above newly_bad) is an arbitrary but fixed and documented tie-break, not a -// claim that one is worse than another. +// rule's instances: +// +// unobservable > {flapping, persistently_bad, newly_bad} > recovered > +// skipped > clean +// +// with unobservable and skipped applied outside this function (decide owns +// both: unobservable from CoverageResult, skipped from the log header). The +// three fail values are not ranked against each other by anything that reads +// this, so their relative order here is an arbitrary but fixed tie-break, not +// a claim that one is worse than another. func outcomeRank(o Outcome) int { switch o { case OutcomeFlapping: @@ -466,12 +464,11 @@ func mergeDurations(eps []episode) time.Duration { } // pollsForRule filters polls to one rule and sorts them by GrafanaNow, the -// same selection proveCoverage uses (§14.5: selection is by UID, never by -// title) — stable, because two polls sharing a coarse Date header must not -// reorder nondeterministically in a pure function. This is the single -// filter+sort implementation for the package (F5): proveCoverage calls it -// too, rather than keeping its own copy that could silently drift from this -// one's membership test. +// same selection proveCoverage uses (by UID, never by title) — stable, because +// two polls sharing a coarse Date header must not reorder nondeterministically +// in a pure function. This is the single filter+sort implementation for the +// package: proveCoverage calls it too, rather than keeping its own copy that +// could silently drift from this one's membership test. func pollsForRule(polls []Poll, uid string) []Poll { var out []Poll for _, p := range polls { @@ -484,9 +481,9 @@ func pollsForRule(polls []Poll, uid string) []Poll { } // badStateSet turns Policy.States into a lookup set, defaulting to {firing} -// (§13) when the caller leaves States empty — decide applies the default -// itself so a test can pass a zero-value Policy and get v1's real default, -// rather than relying on a CLI layer that does not exist yet. +// when the caller leaves States empty — decide applies the default itself so a +// test can pass a zero-value Policy and get the real default, rather than +// depending on the CLI to have filled it in. func badStateSet(states []State) map[State]bool { if len(states) == 0 { states = []State{StateFiring} @@ -499,17 +496,16 @@ func badStateSet(states []State) map[State]bool { } // decide is the pure seam between the collected evidence and the CLI's exit -// code: nearly every §22 test targets this function, not Check (P9). It -// combines proveCoverage's nine checks with classifyRule's timelines under -// one Policy, and OWNS the H6 mapping: any unobservable rule makes decide -// return a non-nil error, which P10's CLI maps to exit 2 unconditionally -// (H7) — never to 0 or 1, and never suppressed by a real violation found -// alongside it. +// code, and carries nearly the whole test suite because of it. It combines +// proveCoverage's nine checks with classifyRule's timelines under one Policy, +// and owns the inability-beats-violation rule: any unobservable rule makes +// decide return a non-nil error, which the CLI maps to exit 2 unconditionally +// — never to 0 or 1, and never suppressed by a real violation found alongside +// it. // -// Result is fully populated even when the returned error is non-nil: H7's -// "err != nil, the violation list is irrelevant" means the CALLER must not -// use Violations to second-guess the error, not that Result stops being -// useful for the human table on exit 2. +// Result is fully populated even when the returned error is non-nil. A caller +// must not use Violations to second-guess the error, but Result stays useful +// for the human table on exit 2. func decide(h Header, polls []Poll, sentinel *time.Time, defs []Definition, rt map[string]ruleTimings, gt globalTimings, pol Policy) (Result, error) { @@ -560,7 +556,7 @@ func decide(h Header, polls []Poll, sentinel *time.Time, defs []Definition, unobservableNames []string ) - // `skipped` is decided from the header, never from defs (§12). defs are + // `skipped` is decided from the header, never from defs. defs are // resolved after the window has closed, so Definition.IsPaused describes // the present; Header.pausedAtStart describes the moment the recording // opened, which is the only moment "paused before the window opened" can @@ -617,6 +613,7 @@ func decide(h Header, polls []Poll, sentinel *time.Time, defs []Definition, }) } +<<<<<<< HEAD // MinObserved (§12): default len(defs) after the collapse (already done // by Resolve before decide ever sees defs). skipped rules count against // it unless AllowPaused says otherwise. A shortfall counts toward exit 1 @@ -627,6 +624,18 @@ func decide(h Header, polls []Poll, sentinel *time.Time, defs []Definition, // what could ever be resolved). counted := watchedCount var attributable []Definition +======= + // MinObserved defaults to len(defs) after duplicate names collapse + // (already done by Resolve before decide ever sees defs). Skipped rules + // count against it unless AllowPaused says otherwise. A shortfall counts + // toward exit 1, never exit 2 — decide never returns an error for this — + // and it has to surface through Violations like any other fail reason, so + // a shortfall always produces at least one, even when no rule is paused at + // all (an operator-supplied MinObserved that simply exceeds what could ever + // be resolved). + counted := observedCount + var chargeable []Definition +>>>>>>> 641701bb (chore: more concise comments) if pol.AllowPaused { counted += len(skippedRules) } else { @@ -638,10 +647,10 @@ func decide(h Header, polls []Poll, sentinel *time.Time, defs []Definition, if attributed >= shortfall { break } - // §12.1 requires the paused rule and --allow-paused both be - // named to the user; both live in this one Violation, in Note — - // P10's renderer prints Note verbatim rather than re-deriving - // the hint, so the exact wording here is what an operator reads. + // The paused rule and --allow-paused must both be named to the + // user; both live in this one Violation, in Note — the renderer + // prints Note verbatim rather than re-deriving the hint, so the + // exact wording here is what an operator reads. result.Violations = append(result.Violations, Violation{ Alert: def.Title, RuleUID: def.UID, Outcome: OutcomeSkipped, Note: "paused before the window opened; counts against --min-observed unless --allow-paused is set", diff --git a/grafana-alertcheck/internal/gate/classify_test.go b/grafana-alertcheck/internal/gate/classify_test.go index 6522f6e38..017d7ae6b 100644 --- a/grafana-alertcheck/internal/gate/classify_test.go +++ b/grafana-alertcheck/internal/gate/classify_test.go @@ -83,8 +83,7 @@ func TestClassifyRule_NewOnsetInsideWindowIsNewlyBad(t *testing.T) { } } -// TestClassifyRule_NewOnsetThatClearsStillFails pins §11.4 point 3: a -// genuinely new bad episode fails even if it clears again before the window +// A genuinely new bad episode fails even if it clears again before the window // ends — only a PREEXISTING condition earns the benefit of `recovered`. func TestClassifyRule_NewOnsetThatClearsStillFails(t *testing.T) { from := time.Date(2026, 1, 1, 0, 0, 0, 0, time.UTC) @@ -134,9 +133,9 @@ func TestClassifyRule_PreexistingThatRecoversIsRecoveredAndNotAViolation(t *test } } -// §22.2's "late condition": bad for 58 of a 60-minute window, clear at -// minute 58, still passes with a large BadFor — never a fail against some -// derived deadline (e.g. "must clear before 90% of the window"). +// The late condition: bad for 58 of a 60-minute window, clear at minute 58, +// still passes with a large BadFor — never a fail against some derived +// deadline (e.g. "must clear before 90% of the window"). func TestClassifyRule_LateRecoveryPassesRegardlessOfHowLateItIs(t *testing.T) { from := time.Date(2026, 1, 1, 0, 0, 0, 0, time.UTC) to := from.Add(60 * time.Minute) @@ -205,9 +204,9 @@ func TestClassifyRule_ClearThenBadAgainIsFlapping(t *testing.T) { } } -// §22.2: "a clear and then a second bad state gives flapping, at each -// possible time of the second bad state." A table over where the second -// onset lands — immediately after the clear, mid-window, and right at the +// A clear and then a second bad state gives flapping, wherever the second bad +// state lands. A table over where the second onset falls — immediately after +// the clear, mid-window, and right at the // last instant before windowEnd — closes the boundary this single fixed // timing above cannot. func TestClassifyRule_FlappingAtEveryTimingOfTheSecondOnset(t *testing.T) { @@ -244,7 +243,7 @@ func TestClassifyRule_FlappingAtEveryTimingOfTheSecondOnset(t *testing.T) { } } -// --- H2: vanished is a discontinuity, never a clear --- +// --- vanished is a discontinuity, never a clear --- func TestClassifyRule_VanishedWhileBadStaysPersistentlyBad(t *testing.T) { from := time.Date(2026, 1, 1, 0, 0, 0, 0, time.UTC) @@ -259,7 +258,7 @@ func TestClassifyRule_VanishedWhileBadStaysPersistentlyBad(t *testing.T) { } outcome, badFor, viols := classifyRule(def, polls, from, to, defaultBad, PreexistingFailUnlessRecovered) if outcome != OutcomePersistentlyBad { - t.Fatalf("outcome = %v, want persistently_bad: a vanish must never read as a recovery (H2)", outcome) + t.Fatalf("outcome = %v, want persistently_bad: a vanish must never read as a recovery", outcome) } if badFor != to.Sub(from) { t.Fatalf("badFor = %v, want the full window %v: the freeze must hold the episode open to windowEnd", badFor, to.Sub(from)) @@ -375,7 +374,7 @@ func TestClassifyRule_WorstOfMultipleInstancesWins(t *testing.T) { } } -// --- decide(): skipped rules, unobservable (H6), MinObserved, exit mapping (H7) --- +// --- decide(): skipped rules, unobservable, MinObserved, exit mapping --- func TestDecide_SkippedRuleNeverReachesProveCoverage(t *testing.T) { from := time.Date(2026, 1, 1, 0, 0, 0, 0, time.UTC) @@ -387,7 +386,7 @@ func TestDecide_SkippedRuleNeverReachesProveCoverage(t *testing.T) { pol := Policy{From: from, To: to, AllowPaused: true} // No polls, no sentinel at all: a heartbeat_gap/no_sentinel misclassification - // here would mean proveCoverage ran for a skipped rule (§4.3's obligation). + // here would mean proveCoverage ran for a skipped rule. // The HEADER is what says paused — decide reads skipped from there, not // from def.IsPaused, which is a post-window reading (Header.pausedAtStart). res, err := decide(pausedHeader(from.Add(-time.Hour), "r1"), nil, nil, defs, rt, gt, pol) @@ -415,16 +414,15 @@ func TestDecide_UnobservableRuleAlwaysReturnsAnError(t *testing.T) { // regardless of anything else. res, err := decide(Header{StartedAt: from.Add(-time.Hour)}, nil, nil, defs, rt, gt, pol) if err == nil { - t.Fatalf("err = nil, want non-nil: H6/H7 require an unobservable rule to always fail the run") + t.Fatalf("err = nil, want non-nil: an unobservable rule must always fail the run") } if len(res.Verdicts) != 1 || res.Verdicts[0].Outcome != OutcomeUnobservable { t.Fatalf("Verdicts = %+v, want exactly one unobservable verdict", res.Verdicts) } } -// TestDecide_UnobservableWinsEvenAlongsideARealViolation pins H6 exactly: -// "Any unobservable rule -> exit 2, no exception, even alongside a real -// newly_bad." +// Any unobservable rule means exit 2, with no exception — even alongside a +// real newly_bad. func TestDecide_UnobservableWinsEvenAlongsideARealViolation(t *testing.T) { from := time.Date(2026, 1, 1, 0, 0, 0, 0, time.UTC) to := from.Add(10 * time.Minute) @@ -468,18 +466,17 @@ func TestDecide_UnobservableWinsEvenAlongsideARealViolation(t *testing.T) { t.Fatalf("broken.Outcome = %v, want unobservable", gotBroken) } if gotBad != OutcomeNewlyBad { - t.Fatalf("bad.Outcome = %v, want newly_bad: classification still runs and is still visible in Verdicts (H5)", gotBad) + t.Fatalf("bad.Outcome = %v, want newly_bad: classification still runs and is still visible in Verdicts", gotBad) } if len(res.Violations) == 0 { t.Fatalf("Violations empty, want the newly_bad instance still reported even though the run fails on the unobservable rule") } } -// §22.10: "a clean verdict with a coverage gap ... must never give exit 0", -// and "a recovered verdict and a skipped verdict also need proved coverage -// of the full window." One genuinely unobservable rule ("broken", zero -// polls) alongside a rule with each of the three favorable outcomes — none -// of them may waive the run. +// A clean verdict with a coverage gap must never give exit 0, and recovered +// and skipped verdicts need proved coverage of the full window just as much. +// One genuinely unobservable rule ("broken", zero polls) alongside a rule with +// each of the three favorable outcomes — none of them may waive the run. func TestDecide_UnobservableRuleWinsOverEveryFavorableOutcome(t *testing.T) { from := time.Date(2026, 1, 1, 0, 0, 0, 0, time.UTC) to := from.Add(10 * time.Minute) @@ -575,8 +572,8 @@ func TestDecide_UnobservableRuleWinsOverEveryFavorableOutcome(t *testing.T) { } } -// §22.10: the table above puts the coverage gap on a DIFFERENT rule from the -// one with the favorable outcome. This pins the tighter claim: a rule that +// The table above puts the coverage gap on a DIFFERENT rule from the one with +// the favorable outcome. This pins the tighter claim: a rule that // itself recovers, but ALSO itself has a coverage gap, is still overridden to // unobservable — the favorable classification of a rule is never a reason to // skip that same rule's own coverage check. @@ -637,21 +634,20 @@ func TestDecide_CleanWindowIsAPass(t *testing.T) { t.Fatalf("err = %v, want nil", err) } if len(res.Violations) != 0 { - t.Fatalf("Violations = %+v, want none: H7 says a pass is exactly len(Violations)==0 && err==nil", res.Violations) + t.Fatalf("Violations = %+v, want none: a pass is exactly len(Violations)==0 && err==nil", res.Violations) } if res.Verdicts[0].Outcome != OutcomeClean { t.Fatalf("Outcome = %v, want clean", res.Verdicts[0].Outcome) } } -// §22.7's second of the plan's "if only three tests could exist" cases: a -// pause and then an unpause inside the window, with an episode that would +// A pause and then an unpause inside the window, with an episode that would // fire and resolve entirely inside the blind interval. A drain wait alone — // "did the rule eventually evaluate through windowEnd?" — would see // lastEvaluation catch up after the unpause and answer yes, a pass. decide() -// never runs a drain wait (that is check.go's I/O concern, §14.6); this pins -// that proveCoverage's own per-poll checks already refuse the window without -// one, so a live drain wait is not what is saving this case. +// never runs a drain wait (that is check.go's I/O concern); this pins that +// proveCoverage's own per-poll checks already refuse the window without one, +// so a live drain wait is not what is saving this case. func TestDecide_PauseThenUnpauseWithHiddenEpisodeGivesUnobservableNotClean(t *testing.T) { from := time.Date(2026, 1, 1, 0, 0, 0, 0, time.UTC) to := from.Add(20 * time.Minute) @@ -671,7 +667,7 @@ func TestDecide_PauseThenUnpauseWithHiddenEpisodeGivesUnobservableNotClean(t *te for ts := pauseStart; !ts.After(pauseEnd); ts = ts.Add(30 * time.Second) { // No fire/resolve is ever observed here: the rule was not // evaluating, so any real episode inside this stretch is invisible - // to every poll (§14.7). + // to every poll. polls = append(polls, Poll{RuleUID: "r1", GrafanaNow: ts, Found: true, Health: "ok", IsPaused: true, LastEvaluation: pauseStart}) } for ts := pauseEnd.Add(30 * time.Second); !ts.After(to); ts = ts.Add(30 * time.Second) { @@ -693,7 +689,7 @@ func TestDecide_PauseThenUnpauseWithHiddenEpisodeGivesUnobservableNotClean(t *te } } -// --- MinObserved shortfall (§12) --- +// --- MinObserved shortfall --- func TestDecide_SkippedOnlyShortfallProducesAViolationWithoutAnError(t *testing.T) { from := time.Date(2026, 1, 1, 0, 0, 0, 0, time.UTC) @@ -718,13 +714,13 @@ func TestDecide_SkippedOnlyShortfallProducesAViolationWithoutAnError(t *testing. res, err := decide(pausedHeader(from.Add(-time.Hour), "paused"), polls, &sentinel, defs, rt, gt, pol) if err != nil { - t.Fatalf("err = %v, want nil: a shortfall caused only by a skipped rule is exit 1, not exit 2 (§9.1)", err) + t.Fatalf("err = %v, want nil: a shortfall caused only by a skipped rule is exit 1, not exit 2", err) } if len(res.Violations) != 1 { - t.Fatalf("Violations = %+v, want exactly one: H7 needs the shortfall visible through Violations to keep its equivalence", res.Violations) + t.Fatalf("Violations = %+v, want exactly one: a shortfall must be visible through Violations like any other fail reason", res.Violations) } if v := res.Violations[0]; v.Outcome != OutcomeSkipped || v.RuleUID != "paused" || v.Alert != "Paused" { - t.Fatalf("Violations[0] = %+v, want Outcome=skipped naming the paused rule (§12.1: the message names the paused rule)", v) + t.Fatalf("Violations[0] = %+v, want Outcome=skipped naming the paused rule", v) } if res.Violations[0].Note == "" { t.Fatalf("Violations[0].Note is empty, want an explanation: the shortfall reason must not be smuggled into LastError, " + @@ -732,10 +728,9 @@ func TestDecide_SkippedOnlyShortfallProducesAViolationWithoutAnError(t *testing. } } -// TestDecide_ExplicitMinObservedShortfallWithNoPausedRuleStillProducesAViolation -// pins F3: an operator-supplied MinObserved that exceeds what could ever be -// resolved is still a shortfall, even with zero paused rules to blame it on -// — H7 must not let this silently read as a pass. +// An operator-supplied MinObserved that exceeds what could ever be resolved is +// still a shortfall, even with zero paused rules to blame it on — it must not +// silently read as a pass. func TestDecide_ExplicitMinObservedShortfallWithNoPausedRuleStillProducesAViolation(t *testing.T) { from := time.Date(2026, 1, 1, 0, 0, 0, 0, time.UTC) to := from.Add(10 * time.Minute) @@ -758,7 +753,7 @@ func TestDecide_ExplicitMinObservedShortfallWithNoPausedRuleStillProducesAViolat } if len(res.Violations) != 2 { t.Fatalf("Violations = %+v, want two: the shortfall (3-1=2) is not explained by any paused rule, "+ - "so H7 requires it to surface directly rather than pass silently", res.Violations) + "so it must surface directly rather than pass silently", res.Violations) } for _, v := range res.Violations { if v.Outcome != OutcomeSkipped { @@ -846,13 +841,12 @@ func TestDecide_NodataIsANoteByDefault(t *testing.T) { } } -// --- F1/F2 regressions: preexisting is decided by ActiveAt, not poll timing --- +// --- preexisting is decided by ActiveAt, not poll timing --- -// TestClassifyRule_OnsetBetweenFromAndFirstPollIsNewlyBadNotRecovered pins -// F1: an instance whose true onset (ActiveAt) falls strictly inside the -// window — even though the first poll that happens to observe it already -// shows it bad — must never be treated as preexisting. If it then clears, -// the plan requires newly_bad (exit 1), not recovered (exit 0). +// An instance whose true onset (ActiveAt) falls strictly inside the window — +// even though the first poll that happens to observe it already shows it bad — +// must never be treated as preexisting. If it then clears, that is newly_bad +// (exit 1), not recovered (exit 0). func TestClassifyRule_OnsetBetweenFromAndFirstPollIsNewlyBadNotRecovered(t *testing.T) { from := time.Date(2026, 1, 1, 0, 0, 0, 0, time.UTC) to := from.Add(10 * time.Minute) @@ -870,13 +864,13 @@ func TestClassifyRule_OnsetBetweenFromAndFirstPollIsNewlyBadNotRecovered(t *test outcome, badFor, viols := classifyRule(def, polls, from, to, defaultBad, PreexistingFailUnlessRecovered) if outcome != OutcomeNewlyBad { t.Fatalf("outcome = %v, want newly_bad: the onset is after `from`, so it is not preexisting even though "+ - "the FIRST in-window poll already observes it bad (F1)", outcome) + "the FIRST in-window poll already observes it bad", outcome) } if len(viols) != 1 || viols[0].Outcome != OutcomeNewlyBad { t.Fatalf("viols = %+v, want one newly_bad violation: a policy=fail-unless-recovered default must still fail this", viols) } if want := clearAt.Sub(onset); badFor != want { - t.Fatalf("badFor = %v, want %v: BadFor must count from the true onset, not from `from` (F1's overcount bug)", badFor, want) + t.Fatalf("badFor = %v, want %v: BadFor must count from the true onset, not from `from`", badFor, want) } } @@ -909,11 +903,9 @@ func TestClassifyRule_OnsetJustBeforeFromIsPreexisting(t *testing.T) { } } -// TestClassifyRule_SkewTranslatesActiveAtAcrossTheWindowBoundary pins F2: a -// poll carrying a nonzero skew must have its ActiveAt (and GrafanaNow) +// A poll carrying a nonzero skew must have its ActiveAt (and GrafanaNow) // translated to the runner domain before comparing against `from` — a raw, -// untranslated comparison would land on the wrong side of the F1 boundary -// check. +// untranslated comparison would land on the wrong side of that boundary. func TestClassifyRule_SkewTranslatesActiveAtAcrossTheWindowBoundary(t *testing.T) { from := time.Date(2026, 1, 1, 0, 0, 0, 0, time.UTC) to := from.Add(10 * time.Minute) @@ -937,14 +929,14 @@ func TestClassifyRule_SkewTranslatesActiveAtAcrossTheWindowBoundary(t *testing.T outcome, badFor, _ := classifyRule(def, []Poll{poll, stillBad}, from, to, defaultBad, PreexistingFailUnlessRecovered) if outcome != OutcomePersistentlyBad { - t.Fatalf("outcome = %v, want persistently_bad: a +90s skew must translate ActiveAt back to exactly `from` (F2)", outcome) + t.Fatalf("outcome = %v, want persistently_bad: a +90s skew must translate ActiveAt back to exactly `from`", outcome) } if badFor != to.Sub(from) { t.Fatalf("badFor = %v, want the full window %v", badFor, to.Sub(from)) } } -// --- F4: InstanceLabels must survive a timeline first created by a bare marker --- +// --- InstanceLabels must survive a timeline first created by a bare marker --- func TestClassifyRule_LabelsSurviveWhenTimelineStartsFromAClearedMarker(t *testing.T) { from := time.Date(2026, 1, 1, 0, 0, 0, 0, time.UTC) @@ -967,12 +959,11 @@ func TestClassifyRule_LabelsSurviveWhenTimelineStartsFromAClearedMarker(t *testi } if viols[0].InstanceLabels == nil || viols[0].InstanceLabels["instance"] != "a" { t.Fatalf("InstanceLabels = %+v, want {instance: a}: labels must backfill even though the "+ - "timeline was first created by a label-less Cleared marker (F4)", viols[0].InstanceLabels) + "timeline was first created by a label-less Cleared marker", viols[0].InstanceLabels) } } -// TestClassifyRule_ViolationFieldsArePrecise pins FirstSeen/ClearedAt exactly, -// not just that a violation exists (F7). +// FirstSeen/ClearedAt are pinned exactly, not just that a violation exists. func TestClassifyRule_ViolationFieldsArePrecise(t *testing.T) { from := time.Date(2026, 1, 1, 0, 0, 0, 0, time.UTC) to := from.Add(10 * time.Minute) @@ -1002,10 +993,9 @@ func TestClassifyRule_ViolationFieldsArePrecise(t *testing.T) { } } -// TestClassifyRule_ClearedEventPastWindowEndClampsToWindowEnd pins the -// episode.end clamp: inWindowPolls admits a poll up to its own skew bound -// past windowEnd (§16's widened membership test), so a genuine Cleared event -// on such a poll must not leave the episode extending beyond windowEnd. +// The episode.end clamp: inWindowPolls admits a poll up to its own skew bound +// past windowEnd, so a genuine Cleared event on such a poll must not leave the +// episode extending beyond windowEnd. func TestClassifyRule_ClearedEventPastWindowEndClampsToWindowEnd(t *testing.T) { from := time.Date(2026, 1, 1, 0, 0, 0, 0, time.UTC) to := from.Add(10 * time.Minute) @@ -1069,7 +1059,7 @@ func TestClassifyRule_OnsetJustPastWindowEndIsNewlyBadNotClean(t *testing.T) { } } -// §22.2: "a clear after `to` gives persistently_bad." classifyRule filters +// A clear after `to` gives persistently_bad. classifyRule filters // its input to [from, windowEnd] itself (inWindowPolls), so a Cleared event // GENUINELY past windowEnd — well beyond any skew bound, unlike the clamp // case above — never reaches the timeline at all: the instance is still bad diff --git a/grafana-alertcheck/internal/gate/coverage.go b/grafana-alertcheck/internal/gate/coverage.go index ec13ac91c..5c82458ee 100644 --- a/grafana-alertcheck/internal/gate/coverage.go +++ b/grafana-alertcheck/internal/gate/coverage.go @@ -5,32 +5,28 @@ import ( "time" ) -// keepLastReason is the instance Reason that check 9 watches for (§10.2). +// keepLastReason is the instance Reason that check 9 watches for. const keepLastReason = "KeepLast" -// Two things this file deliberately does not do, and where they are done -// instead — both were open obligations when P7 was written, and both are now -// discharged: +// Two things this file deliberately leaves to its callers: // -// - §7's second clause, "from more than fromFutureTolerance ahead is a hard -// error", is once-per-run input validation rather than a per-rule -// coverage check, and this function has no error return. Discharged by -// P9: the constant is fromFutureTolerance (schedule.go) and Config.validate -// (check.go) applies it. Check 2 below still owns the first clause, -// "from < StartedAt". -// - A rule paused before the window opened is never scheduled or polled -// (§4.3), so it reaches this function with zero polls and reads as one -// large heartbeat_gap, not as skipped (pinned by -// TestProveCoverage_SkippedRuleWithZeroPollsPinnedAsHeartbeatGap). -// Discharged by P8: decide returns before it ever calls proveCoverage for -// such a rule (classify.go). It reads skipped from the log header -// (Header.pausedAtStart), NOT from Definition.IsPaused — the definitions -// are re-resolved after the window closed, so they cannot answer what was -// paused when it opened. +// - "from more than fromFutureTolerance ahead of the runner's clock" is a +// hard error, but it is once-per-run input validation rather than a +// per-rule coverage check, and this function has no error return. +// Config.validate (check.go) applies it; check 2 below owns only the +// "from < StartedAt" half. +// - A rule paused before the window opened is never scheduled or polled, so +// it would reach this function with zero polls and read as one large +// heartbeat_gap rather than as skipped (pinned by +// TestProveCoverage_SkippedRuleWithZeroPollsPinnedAsHeartbeatGap). decide +// returns before it ever calls proveCoverage for such a rule, reading +// skipped from the log header (Header.pausedAtStart) and NOT from +// Definition.IsPaused — the definitions are re-resolved after the window +// closed, so they cannot answer what was paused when it opened. // UnobservableReason names why proveCoverage could not prove a rule's window. -// It is machine-readable — this reaches the action's JSON outputs, so it is a -// published vocabulary like Outcome (§19.0); prose belongs in Notes. +// It is machine-readable — this reaches the JSON output, so it is a published +// vocabulary like Outcome; prose belongs in Notes. type UnobservableReason string const ( @@ -43,16 +39,16 @@ const ( ReasonFutureEvaluation UnobservableReason = "future_evaluation" ReasonPausedInWindow UnobservableReason = "paused_in_window" ReasonRuleAbsent UnobservableReason = "rule_absent" - // ReasonDrainTimeout is set by check.go's drain wait (a later phase), - // never by proveCoverage: the wait is I/O and must not be added to this - // pure function — that would put HTTP inside the pure layer and destroy - // the seam §2's architecture depends on. + // ReasonDrainTimeout is set by check.go's drain wait, never by + // proveCoverage: the wait is I/O and must not be added to this pure + // function — that would put HTTP inside the pure layer and destroy the + // seam this design depends on. ReasonDrainTimeout UnobservableReason = "drain_timeout" ) // CoverageResult is proveCoverage's whole answer for one rule. No interval // list: proved-or-not plus the largest gap and where is everything a human -// reads on exit 2, and everything §20.2's table needs. +// reads on exit 2, and everything the rendered table needs. type CoverageResult struct { Proved bool LargestGap time.Duration @@ -65,21 +61,21 @@ type CoverageResult struct { BlindFor time.Duration } -// proveCoverage applies the nine coverage checks (§6, §10, §14) to one rule's -// polls and is PURE: no HTTP, no files, no clock reads — everything it needs -// arrives as an argument, which is what lets §22's tests build []Poll literals -// instead of a fixture server (§2). +// proveCoverage applies the nine coverage checks to one rule's polls and is +// PURE: no HTTP, no files, no clock reads — everything it needs arrives as an +// argument, which is what lets its tests build []Poll literals instead of a +// fixture server. // // polls need not be pre-filtered to this rule: proveCoverage selects by -// def.UID itself, exactly as Reduce selects by UID rather than by title -// (§14.5) — a caller handing it a whole log's polls must not have to -// pre-filter to get a correct answer. +// def.UID itself, exactly as Reduce selects by UID rather than by title — a +// caller handing it a whole log's polls must not have to pre-filter to get a +// correct answer. // // Every check always runs, even once an earlier one has already set // Unobservable: LargestGap and the notes are diagnostics an operator reads on -// exit 2 regardless of which check actually failed (§20.2). Reason names the -// FIRST check, in the order below, that failed; a later failure still adds -// its own Note. +// exit 2 regardless of which check actually failed. Reason names the FIRST +// check, in the order below, that failed; a later failure still adds its own +// Note. func proveCoverage(h Header, polls []Poll, sentinel *time.Time, t ruleTimings, def Definition, from, to time.Time, grace time.Duration) CoverageResult { @@ -101,9 +97,9 @@ func proveCoverage(h Header, polls []Poll, sentinel *time.Time, t ruleTimings, d res.Notes = append(res.Notes, fmt.Sprintf("rule %q: %s", def.Title, note)) } - // Check 1 — sentinel (§4.5). Present and At >= to+grace -> coverage - // provable; absent, or short of it, is never a pass. A recorder that died - // early must look exactly like a coverage gap, because it is one. + // Check 1 — sentinel. Present and At >= to+grace -> coverage provable; + // absent, or short of it, is never a pass. A recorder that died early must + // look exactly like a coverage gap, because it is one. switch { case sentinel == nil: fail(ReasonNoSentinel, "no stopped sentinel: the recorder never reported finishing") @@ -112,13 +108,12 @@ func proveCoverage(h Header, polls []Poll, sentinel *time.Time, t ruleTimings, d sentinel.Format(time.RFC3339), windowEnd.Format(time.RFC3339))) } - // Check 2 — from bounds (§7), first sentence only: from < StartedAt makes - // coverage unprovable, no matter how healthy the polls that DO exist look. - // Both are runner-domain clock reads (the recorder's own Clock.Now()), so - // no cross-domain translation applies here. The second sentence — from - // more than fromFutureTolerance ahead is a hard error — is Check's input - // validation, once per run rather than per rule, and belongs to a later - // phase: this function has no error return, only a per-rule verdict. + // Check 2 — from bounds: from < StartedAt makes coverage unprovable, no + // matter how healthy the polls that DO exist look. Both are runner-domain + // clock reads (the recorder's own Clock.Now()), so no cross-domain + // translation applies here. The other half of the bound — from too far + // ahead of the runner's clock — is Check's input validation, once per run + // rather than per rule. if from.Before(h.StartedAt) { fail(ReasonFromBeforeRecord, fmt.Sprintf( "requested from %s is before recording started at %s", from.Format(time.RFC3339), h.StartedAt.Format(time.RFC3339))) @@ -130,18 +125,18 @@ func proveCoverage(h Header, polls []Poll, sentinel *time.Time, t ruleTimings, d // drifting from the other's membership test. inWindow := inWindowPolls(rulePolls, from, windowEnd) - // Check 3 — heartbeat continuity (§6). Data at both ends with a hole in - // between is not enough (§22.4): this scans every gap inside the window, - // not just its edges. + // Check 3 — heartbeat continuity. Data at both ends with a hole in between + // is not enough: this scans every gap inside the window, not just its + // edges. res.LargestGap, res.LargestGapAt = ruleHeartbeatGap(inWindow, from, windowEnd) if res.LargestGap > t.maxGap { fail(ReasonHeartbeatGap, fmt.Sprintf( "gap of %s starting at %s exceeds maxGap %s", res.LargestGap, res.LargestGapAt.Format(time.RFC3339), t.maxGap)) } - // Check 4 — health=="error" (§10.1). A short blip is a note only (§22.1: - // one failed evaluation must not exit 2 over an otherwise clean window); - // only a run longer than healthGrace consumes coverage. + // Check 4 — health=="error". A short blip is a note only — one failed + // evaluation must not exit 2 over an otherwise clean window; only a run + // longer than healthGrace consumes coverage. if runLen, sawAny := longestHealthRun(inWindow, "error"); sawAny { res.Notes = append(res.Notes, fmt.Sprintf("rule %q: health=error observed (longest run %s)", def.Title, runLen)) if runLen > t.healthGrace { @@ -149,25 +144,25 @@ func proveCoverage(h Header, polls []Poll, sentinel *time.Time, t ruleTimings, d } } - // Check 5 — health=="nodata" (§10.1/§10.2). Never fatal here: 96% of the - // fleet runs no_data_state:OK, so treating this as fatal by default would - // block nearly every healthy deploy in an idle environment. Escalating it - // under Policy.NodataIsUnobservable is decide's job (a later phase), - // applied directly against the raw polls — this pure function has no - // Policy to consult and must not invent one. + // Check 5 — health=="nodata". Never fatal here: 96% of the fleet runs + // no_data_state:OK, so treating this as fatal by default would block + // nearly every healthy deploy in an idle environment. Escalating it under + // Policy.NodataIsUnobservable is decide's job, applied directly against + // the raw polls — this pure function has no Policy to consult and must not + // invent one. if _, sawAny := longestHealthRun(inWindow, "nodata"); sawAny { res.Notes = append(res.Notes, fmt.Sprintf("rule %q: health=nodata observed (not fatal; see --nodata-is-unobservable)", def.Title)) } - // Check 6 — liveness (H3). Absolute only, per poll: GrafanaNow and + // Check 6 — liveness. Absolute only, per poll: GrafanaNow and // LastEvaluation are both Grafana-domain reads off the SAME response, so // this is a same-domain comparison and uses raw values — never a delta // against a previous poll, which reports stale on ~half the polls of a // perfectly healthy rule (polling runs at intervalSeconds/2). // // Skipped only for a poll whose own flags SAY there is nothing to check: - // IsPaused (a zero LastEvaluation is legal only while paused, §2.3; check - // 7 is its detector) or !Found (no rule, no evaluation; check 8 is its + // IsPaused (a zero LastEvaluation is legal only while paused; check 7 is + // its detector) or !Found (no rule, no evaluation; check 8 is its // detector). Deliberately NOT skipped merely because LastEvaluation is // zero: ReadLog does no field validation, so a corrupted or hand-edited // log line can claim found:true, is_paused:false and still carry a zero @@ -204,11 +199,10 @@ func proveCoverage(h Header, polls []Poll, sentinel *time.Time, t ruleTimings, d staleCount, worstStale, t.evalStaleAfter, worstStaleAt.Format(time.RFC3339))) } - // Check 7 — isPaused in-window (§12.2, §14.8). The PRIMARY pause - // detector: liveness (check 6) is only the backup for what IsPaused - // cannot show (a deleted rule, a stopped scheduler, a blocked - // evaluation). This is what catches pause-then-unpause, which the drain - // wait alone passes (§14.7). + // Check 7 — isPaused in-window. The PRIMARY pause detector: liveness + // (check 6) is only the backup for what IsPaused cannot show (a deleted + // rule, a stopped scheduler, a blocked evaluation). This is what catches + // pause-then-unpause, which the drain wait alone passes. var pausedCount int var pausedAt time.Time for _, p := range inWindow { @@ -223,8 +217,8 @@ func proveCoverage(h Header, polls []Poll, sentinel *time.Time, t ruleTimings, d fail(ReasonPausedInWindow, fmt.Sprintf("observed paused on %d poll(s), first at %s", pausedCount, pausedAt.Format(time.RFC3339))) } - // Check 8 — rule absent (§14.5). Found==false is authoritative (P2 - // already retried every transport failure before a Poll record ever + // Check 8 — rule absent. Found==false is authoritative (the transport + // already retried every transient failure before a Poll record ever // exists): the rule resolved at resolve time but the state endpoint // stopped serving it. Never drop a watched rule from the verdict set // silently. @@ -242,12 +236,12 @@ func proveCoverage(h Header, polls []Poll, sentinel *time.Time, t ruleTimings, d fail(ReasonRuleAbsent, fmt.Sprintf("state endpoint returned no rule on %d poll(s), first at %s", absentCount, absentAt.Format(time.RFC3339))) } - // Check 9 — KeepLast (§10.2). Two distinct notes, both non-fatal: + // Check 9 — KeepLast. Two distinct notes, both non-fatal: // - // DECLARED: the rule's no_data_state/exec_err_state is configured as - // KeepLast — a standing blind spot (§10.2). Prefer the header's - // LoggedRule snapshot: in log mode def is re-resolved after the window - // and can drift (see pausedAtStart, log.go). Reads config, fires once. + // DECLARED: the rule's own no_data_state/exec_err_state is configured as + // KeepLast — a standing blind spot whether or not it is ever exercised + // during this particular window. This reads def, not polls, so it fires + // exactly once regardless of poll content. nds, ees := def.NoDataState, def.ExecErrState for _, lr := range h.Rules { if lr.UID == def.UID { @@ -257,13 +251,12 @@ func proveCoverage(h Header, polls []Poll, sentinel *time.Time, t ruleTimings, d } if nds == keepLastReason || ees == keepLastReason { res.Notes = append(res.Notes, fmt.Sprintf( - "rule %q: configured with no_data_state/exec_err_state=KeepLast — a stale state can continue past a real fault (§10.2)", def.Title)) + "rule %q: configured with no_data_state/exec_err_state=KeepLast — a stale state can continue past a real fault", def.Title)) } // OBSERVED: an instance actually reported the KeepLast reason during the - // window. It surfaces only as an instance Reason after P1.2a's parsing, - // and Reasons keys can be comma-joined composites, so membership - // (reasonsContain) is required — indexing "KeepLast" directly would miss - // "KeepLast, MissingSeries". + // window. It surfaces only as an instance Reason, and Reasons keys can be + // comma-joined composites, so membership (reasonsContain) is required — + // indexing "KeepLast" directly would miss "KeepLast, MissingSeries". for _, p := range inWindow { if reasonsContain(p.Reasons, keepLastReason) { res.Notes = append(res.Notes, fmt.Sprintf( @@ -277,16 +270,16 @@ func proveCoverage(h Header, polls []Poll, sentinel *time.Time, t ruleTimings, d } // inWindowPolls filters polls to those inside [from, windowEnd] using the -// CROSS-DOMAIN membership test (§16): each poll's Grafana-domain GrafanaNow -// is translated to the runner domain by its OWN skew, and its own skew bound -// is the membership tolerance, so a poll that is genuinely inside the window -// is never excluded by ordinary clock imprecision. +// CROSS-DOMAIN membership test: each poll's Grafana-domain GrafanaNow is +// translated to the runner domain by its OWN skew, and its own skew bound is +// the membership tolerance, so a poll that is genuinely inside the window is +// never excluded by ordinary clock imprecision. // -// Everything downstream of this filter (health runs, liveness, pause, -// absence) reads the poll's raw fields: GrafanaNow paired with -// LastEvaluation on the SAME response, or one poll's GrafanaNow against the -// next's, are same-domain comparisons and need no translation (§16, "Clock -// domains" — only window membership and check 3's two boundary segments do). +// Everything downstream of this filter (health runs, liveness, pause, absence) +// reads the poll's raw fields: GrafanaNow paired with LastEvaluation on the +// SAME response, or one poll's GrafanaNow against the next's, are same-domain +// comparisons and need no translation. Only window membership and check 3's +// two boundary segments cross domains. func inWindowPolls(polls []Poll, from, windowEnd time.Time) []Poll { var out []Poll for _, p := range polls { @@ -300,22 +293,22 @@ func inWindowPolls(polls []Poll, from, windowEnd time.Time) []Poll { return out } -// ruleHeartbeatGap finds the largest unobserved span inside [from, windowEnd] -// (§6), including the two boundary segments — which is why "data at both -// ends with a hole in the middle" still fails (§22.4): the segment between -// the polls just inside each edge is exactly what this measures. in must -// already be filtered to this window (inWindowPolls) and sorted by -// GrafanaNow — proveCoverage computes that filter once and threads it through -// every check, this one included, rather than each check re-filtering. +// ruleHeartbeatGap finds the largest unobserved span inside [from, windowEnd], +// including the two boundary segments — which is why "data at both ends with a +// hole in the middle" still fails: the segment between the polls just inside +// each edge is exactly what this measures. in must already be filtered to this +// window (inWindowPolls) and sorted by GrafanaNow — proveCoverage computes that +// filter once and threads it through every check, this one included, rather +// than each check re-filtering. // // The two boundary segments compare a Grafana-domain poll time against the // runner-domain from/windowEnd, so each is translated by its own poll's skew -// AND widened by that same poll's skew bound (§16: "with that poll's bound as -// the tolerance") — on the side that makes the segment larger, never smaller, -// so an uncertain boundary reads as at least as big a gap as it might really -// be. Understating it by up to the bound would be fail-open. The spacing -// BETWEEN consecutive polls compares two Grafana-domain reads to each other — -// same domain — and uses the raw GrafanaNow difference, no bound needed. +// AND widened by that same poll's skew bound — on the side that makes the +// segment larger, never smaller, so an uncertain boundary reads as at least as +// big a gap as it might really be. Understating it by up to the bound would be +// fail-open. The spacing BETWEEN consecutive polls compares two Grafana-domain +// reads to each other — same domain — and uses the raw GrafanaNow difference, +// no bound needed. func ruleHeartbeatGap(in []Poll, from, windowEnd time.Time) (largestGap time.Duration, largestGapAt time.Time) { if len(in) == 0 { return windowEnd.Sub(from), from @@ -339,16 +332,15 @@ func ruleHeartbeatGap(in []Poll, from, windowEnd time.Time) (largestGap time.Dur return largestGap, largestGapAt } -// longestHealthRun returns the longest contiguous wall-clock span (§10.1) -// during which polls — already sorted by GrafanaNow, same-domain spacing -// (§16) — read the given rule-level Health, and whether any poll matched it -// at all. +// longestHealthRun returns the longest contiguous wall-clock span during which +// polls — already sorted by GrafanaNow, same-domain spacing — read the given +// rule-level Health, and whether any poll matched it at all. // // It detects the span as it accumulates rather than waiting for the run to // end, so an open-ended run that is still failing at the last poll in the // window is measured correctly without needing data past the window: waiting // for the run to "end" would have to assume the best case about what happens -// next, which is exactly what this gate must not do (§1). +// next, which is exactly what this gate must not do. func longestHealthRun(polls []Poll, health string) (longest time.Duration, sawAny bool) { var runStart time.Time for _, p := range polls { diff --git a/grafana-alertcheck/internal/gate/coverage_test.go b/grafana-alertcheck/internal/gate/coverage_test.go index 3e6829e5a..720386d65 100644 --- a/grafana-alertcheck/internal/gate/coverage_test.go +++ b/grafana-alertcheck/internal/gate/coverage_test.go @@ -45,7 +45,7 @@ func TestProveCoverage_FiltersPollsByUID(t *testing.T) { } } -// --- Check 1: sentinel (§4.5) --- +// --- Check 1: sentinel --- func TestProveCoverage_NoSentinelIsUnobservable(t *testing.T) { from := time.Date(2026, 1, 1, 0, 0, 0, 0, time.UTC) @@ -109,7 +109,7 @@ func TestProveCoverage_SentinelExactlyAtGraceIsFine(t *testing.T) { } } -// --- Check 2: from bounds (§7) --- +// --- Check 2: from bounds --- func TestProveCoverage_FromBeforeRecordIsUnobservable(t *testing.T) { started := time.Date(2026, 1, 1, 1, 0, 0, 0, time.UTC) @@ -141,10 +141,10 @@ func TestProveCoverage_FromBeforeRecordIsUnobservable(t *testing.T) { } } -// --- Check 3: heartbeat continuity (§6) --- +// --- Check 3: heartbeat continuity --- -// TestProveCoverage_HeartbeatGapBetweenBoundariesIsUnobservable is §22.4's -// core regression: data at both ends with a hole between is not enough. +// The core heartbeat regression: data at both ends with a hole between is not +// enough. func TestProveCoverage_HeartbeatGapBetweenBoundariesIsUnobservable(t *testing.T) { from := time.Date(2026, 1, 1, 0, 0, 0, 0, time.UTC) to := from.Add(10 * time.Minute) @@ -159,7 +159,7 @@ func TestProveCoverage_HeartbeatGapBetweenBoundariesIsUnobservable(t *testing.T) res := proveCoverage(Header{StartedAt: from.Add(-time.Hour)}, polls, &sentinel, rt, def, from, to, 0) if res.Reason != ReasonHeartbeatGap { - t.Fatalf("Reason = %q, want heartbeat_gap: healthy edges with a hole in the middle must still fail (§22.4)", res.Reason) + t.Fatalf("Reason = %q, want heartbeat_gap: healthy edges with a hole in the middle must still fail", res.Reason) } // The gap is the SPACING between the two polls (598s), not either // boundary segment (1s each) — pin the actual values, not just the verdict. @@ -172,7 +172,7 @@ func TestProveCoverage_HeartbeatGapBetweenBoundariesIsUnobservable(t *testing.T) } } -// --- Check 4/5: health (§10.1/§10.2) --- +// --- Check 4/5: health --- func TestProveCoverage_HealthErrorShortBlipPassesWithNote(t *testing.T) { from := time.Date(2026, 1, 1, 0, 0, 0, 0, time.UTC) @@ -193,7 +193,7 @@ func TestProveCoverage_HealthErrorShortBlipPassesWithNote(t *testing.T) { res := proveCoverage(Header{StartedAt: from.Add(-time.Hour)}, polls, &sentinel, rt, def, from, to, 0) if !res.Proved { - t.Fatalf("Proved = false, want true: one failed evaluation must not fail an otherwise clean window (§22.1): %+v", res) + t.Fatalf("Proved = false, want true: one failed evaluation must not fail an otherwise clean window: %+v", res) } if !anyContains(res.Notes, "health=error") { t.Fatalf("Notes = %v, want a health=error note even though it did not fail the window", res.Notes) @@ -238,20 +238,19 @@ func TestProveCoverage_HealthNodataNeverFatalHere(t *testing.T) { res := proveCoverage(Header{StartedAt: from.Add(-time.Hour)}, polls, &sentinel, rt, def, from, to, 0) if !res.Proved { t.Fatalf("Proved = false, want true: health=nodata for the WHOLE window must still not be fatal by itself "+ - "(escalating it is Policy.NodataIsUnobservable's job, applied by decide in a later phase): %+v", res) + "(escalating it is Policy.NodataIsUnobservable's job, applied by decide): %+v", res) } if !anyContains(res.Notes, "health=nodata") { t.Fatalf("Notes = %v, want a health=nodata note", res.Notes) } } -// --- Check 6: liveness / H3 --- +// --- Check 6: liveness --- -// TestProveCoverage_LivenessAbsoluteNeverFalseStale is §22.7's disproportionate -// test: a healthy rule polled at intervalSeconds/2, across the full window, -// must show zero staleness violations. lastEvaluation only advances once per -// full evaluation interval here — the realistic shape a delta check -// misreads as stale on roughly half of all polls (H3). +// A healthy rule polled at intervalSeconds/2, across the full window, must +// show zero staleness violations. lastEvaluation only advances once per full +// evaluation interval here — the realistic shape a delta check misreads as +// stale on roughly half of all polls. func TestProveCoverage_LivenessAbsoluteNeverFalseStale(t *testing.T) { from := time.Date(2026, 1, 1, 0, 0, 0, 0, time.UTC) pollEvery := 30 * time.Second @@ -272,7 +271,7 @@ func TestProveCoverage_LivenessAbsoluteNeverFalseStale(t *testing.T) { res := proveCoverage(Header{StartedAt: from.Add(-time.Hour)}, polls, &sentinel, rt, def, from, windowEnd, 0) if res.Reason == ReasonStaleEvaluation || res.BlindFor != 0 { - t.Fatalf("proveCoverage flagged staleness on a healthy rule polled at intervalSeconds/2 — H3 must be absolute, "+ + t.Fatalf("proveCoverage flagged staleness on a healthy rule polled at intervalSeconds/2 — liveness must be absolute, "+ "never a delta against a previous poll: %+v", res) } if !res.Proved { @@ -312,9 +311,8 @@ func TestProveCoverage_ZeroLastEvaluationNeverFalseStale(t *testing.T) { rt := newRuleTimings(30*time.Second, 60) def := Definition{UID: "r1", Title: "R1"} - // A paused rule legitimately reports the zero time (§2.3); check 6 must - // not read that as an enormous staleness violation. Check 7 is its - // detector. + // A paused rule legitimately reports the zero time; check 6 must not read + // that as an enormous staleness violation. Check 7 is its detector. polls := []Poll{ {RuleUID: "r1", GrafanaNow: from.Add(time.Minute), Found: true, IsPaused: true}, } @@ -326,7 +324,7 @@ func TestProveCoverage_ZeroLastEvaluationNeverFalseStale(t *testing.T) { } } -// --- Check 7: isPaused in-window (§12.2, §14.8) --- +// --- Check 7: isPaused in-window --- func TestProveCoverage_PausedInWindowIsUnobservable(t *testing.T) { from := time.Date(2026, 1, 1, 0, 0, 0, 0, time.UTC) @@ -341,7 +339,7 @@ func TestProveCoverage_PausedInWindowIsUnobservable(t *testing.T) { for i := range polls { if polls[i].GrafanaNow.Equal(pausedAt) { polls[i].IsPaused = true - polls[i].LastEvaluation = time.Time{} // legal only while paused, §2.3 + polls[i].LastEvaluation = time.Time{} // legal only while paused } } sentinel := to @@ -379,7 +377,7 @@ func TestProveCoverage_PausedAfterWindowIsFine(t *testing.T) { } } -// --- Check 8: rule absent (§14.5) --- +// --- Check 8: rule absent --- func TestProveCoverage_RuleAbsentIsUnobservable(t *testing.T) { from := time.Date(2026, 1, 1, 0, 0, 0, 0, time.UTC) @@ -418,7 +416,7 @@ func denseHealthyPolls(uid string, from, to time.Time, every time.Duration) []Po return out } -// --- Check 9: KeepLast (§10.2) --- +// --- Check 9: KeepLast --- func TestProveCoverage_KeepLastObservedIsNoteOnly(t *testing.T) { from := time.Date(2026, 1, 1, 0, 0, 0, 0, time.UTC) @@ -431,7 +429,7 @@ func TestProveCoverage_KeepLastObservedIsNoteOnly(t *testing.T) { polls = append(polls, Poll{ RuleUID: "r1", GrafanaNow: ts, Found: true, Health: "ok", LastEvaluation: ts, // A comma-joined composite — reasonsContain must match by - // membership, never by an exact key, per P5's markers. + // membership, never by an exact key. Reasons: map[string]int{"KeepLast, MissingSeries": 1}, }) } @@ -446,8 +444,8 @@ func TestProveCoverage_KeepLastObservedIsNoteOnly(t *testing.T) { } } -// §22.2/§10.2: "KeepLast in the configuration gives a note" — a DIFFERENT -// claim from the observed-reason test above. A rule DECLARED with +// KeepLast in the CONFIGURATION gives a note — a different claim from the +// observed-reason test above. A rule DECLARED with // no_data_state or exec_err_state = KeepLast is a standing blind spot // whether or not any poll ever actually reports the reason, so the note // must fire off the definition alone, over an otherwise perfectly healthy @@ -480,12 +478,11 @@ func TestProveCoverage_KeepLastConfiguredIsNoteOnly(t *testing.T) { } } -// --- Clock domains (§16) --- +// --- Clock domains --- -// TestProveCoverage_SkewTranslationAtWindowBoundary pins §16's "Clock -// domains" rule: a constant clock skew on every poll must not itself read as -// a coverage gap or a from-before-record violation, because every -// cross-domain comparison translates by that poll's own skew first. +// A constant clock skew on every poll must not itself read as a coverage gap +// or a from-before-record violation, because every cross-domain comparison +// translates by that poll's own skew first. func TestProveCoverage_SkewTranslationAtWindowBoundary(t *testing.T) { from := time.Date(2026, 1, 1, 0, 0, 0, 0, time.UTC) to := from.Add(10 * time.Minute) @@ -507,15 +504,14 @@ func TestProveCoverage_SkewTranslationAtWindowBoundary(t *testing.T) { res := proveCoverage(Header{StartedAt: from.Add(-time.Hour)}, polls, &sentinel, rt, def, from, to, 0) if !res.Proved { - t.Fatalf("res = %+v, want proved: a constant clock skew must not itself read as a coverage gap (§16)", res) + t.Fatalf("res = %+v, want proved: a constant clock skew must not itself read as a coverage gap", res) } } -// --- Override round-trip (P5's "two authorities") --- +// --- Override round-trip: one authority for the cadence --- -// TestProveCoverage_OverrideRoundTrip is P7's other disproportionate done-gate -// test: it exercises DeriveTimingsFromLog and proveCoverage together, exactly -// as check will, to prove maxGap tracks the RECORDED cadence, never a +// This exercises DeriveTimingsFromLog and proveCoverage together, exactly as +// check does, to prove maxGap tracks the RECORDED cadence and never a // re-derivation from the rule's own evaluation interval. func TestProveCoverage_OverrideRoundTrip(t *testing.T) { from := time.Date(2026, 1, 1, 0, 0, 0, 0, time.UTC) @@ -577,7 +573,7 @@ func TestProveCoverage_OverrideRoundTrip(t *testing.T) { res := proveCoverage(h, polls, &sentinel, rt["r1"], defs[0], from, windowEnd, 0) if res.Reason != ReasonHeartbeatGap { t.Fatalf("Reason = %q, want heartbeat_gap: if maxGap had been re-derived from the 300s definition instead of "+ - "the recorded 5s cadence, this 250s gap would pass silently — the fail-open direction P5 warns about", res.Reason) + "the recorded 5s cadence, this 250s gap would pass silently — the fail-open direction", res.Reason) } }) } @@ -598,7 +594,7 @@ func anyContains(notes []string, substr string) bool { // found:true, is_paused:false and still carry a zero LastEvaluation (a // corrupted write, a hand-edited fixture, a future log format bug). That // combination must read as maximally stale, not be waved through the way a -// legitimately paused poll's zero time is (§2.3) — the skip must key off +// legitimately paused poll's zero time is — the skip must key off // IsPaused/Found, never off LastEvaluation being zero. func TestProveCoverage_ZeroLastEvaluationWithoutPauseIsStale(t *testing.T) { from := time.Date(2026, 1, 1, 0, 0, 0, 0, time.UTC) @@ -647,10 +643,9 @@ func TestProveCoverage_FutureLastEvaluationIsUnobservable(t *testing.T) { // --- Check 3, tightened: the boundary segments must widen by the skew bound --- -// TestProveCoverage_BoundaryGapWidensBySkewBound pins §16's "with that -// poll's bound as the tolerance" for the two boundary segments specifically: -// a boundary gap that lands EXACTLY at maxGap must still fail once the -// poll's own skew bound is added, because the translation is only a best +// The two boundary segments take their own poll's bound as the tolerance: a +// boundary gap that lands EXACTLY at maxGap must still fail once the poll's +// own skew bound is added, because the translation is only a best // estimate and understating the gap by up to the bound would be fail-open. func TestProveCoverage_BoundaryGapWidensBySkewBound(t *testing.T) { from := time.Date(2026, 1, 1, 0, 0, 0, 0, time.UTC) @@ -670,17 +665,16 @@ func TestProveCoverage_BoundaryGapWidensBySkewBound(t *testing.T) { res := proveCoverage(Header{StartedAt: from.Add(-time.Hour)}, polls, &sentinel, rt, def, from, to, 0) if res.Reason != ReasonHeartbeatGap { t.Fatalf("Reason = %q, want heartbeat_gap: the leading boundary segment sits at EXACTLY maxGap (60s) before "+ - "widening; the poll's own %s skew bound must push it past the threshold (§16), not just the skew translation", res.Reason, bound) + "widening; the poll's own %s skew bound must push it past the threshold, not just the skew translation", res.Reason, bound) } } // --- Multi-failure contract --- -// TestProveCoverage_MultipleFailuresReasonIsFirstButAllNoted exercises two -// checks failing in the same rule: check 7 (paused in-window) precedes check -// 8 (rule absent) in the §5 order, so Reason must name the pause even though -// the rule also goes absent later — and the later failure must still add its -// own Note rather than being swallowed once Reason is set. +// Two checks failing in the same rule: check 7 (paused in-window) runs before +// check 8 (rule absent), so Reason must name the pause even though the rule +// also goes absent later — and the later failure must still add its own Note +// rather than being swallowed once Reason is set. func TestProveCoverage_MultipleFailuresReasonIsFirstButAllNoted(t *testing.T) { from := time.Date(2026, 1, 1, 0, 0, 0, 0, time.UTC) to := from.Add(10 * time.Minute) @@ -705,7 +699,7 @@ func TestProveCoverage_MultipleFailuresReasonIsFirstButAllNoted(t *testing.T) { res := proveCoverage(Header{StartedAt: from.Add(-time.Hour)}, polls, &sentinel, rt, def, from, to, 0) if res.Reason != ReasonPausedInWindow { - t.Fatalf("Reason = %q, want paused_in_window (the FIRST check to fail, in §5's order)", res.Reason) + t.Fatalf("Reason = %q, want paused_in_window: the FIRST check to fail names the reason", res.Reason) } if !anyContains(res.Notes, "paused") { t.Fatalf("Notes = %v, want a note about the pause", res.Notes) @@ -716,18 +710,16 @@ func TestProveCoverage_MultipleFailuresReasonIsFirstButAllNoted(t *testing.T) { } } -// --- Skipped rules (P6/P8 obligation) --- +// --- Skipped rules --- -// TestProveCoverage_SkippedRuleWithZeroPollsPinnedAsHeartbeatGap pins a known -// gap in this function's contract, not a bug in it: a rule paused BEFORE the -// window opened is never scheduled or polled (watch.go, §4.3), so it reaches -// proveCoverage with zero polls at all. proveCoverage has no notion of +// A known limit of this function's contract, not a bug in it: a rule paused +// BEFORE the window opened is never scheduled or polled (watch.go), so it +// reaches proveCoverage with zero polls at all. proveCoverage has no notion of // "skipped" — that classification belongs to the definitions -// (LoggedRule.IsPaused / Definition.IsPaused), never to the polls — so today -// it reports the whole window as one big heartbeat_gap instead. decide (P8) -// MUST read skipped status from the definitions and either skip calling this -// function for that rule entirely, or override this result — this test pins -// today's behavior so that review has something concrete to check against. +// (LoggedRule.IsPaused / Definition.IsPaused), never to the polls — so it +// reports the whole window as one big heartbeat_gap instead. decide is what +// reads skipped status from the header and never calls this function for such +// a rule; this pins the behavior it relies on not reaching. func TestProveCoverage_SkippedRuleWithZeroPollsPinnedAsHeartbeatGap(t *testing.T) { from := time.Date(2026, 1, 1, 0, 0, 0, 0, time.UTC) to := from.Add(10 * time.Minute) @@ -738,7 +730,7 @@ func TestProveCoverage_SkippedRuleWithZeroPollsPinnedAsHeartbeatGap(t *testing.T res := proveCoverage(Header{StartedAt: from.Add(-time.Hour)}, nil, &sentinel, rt, def, from, to, 0) if res.Reason != ReasonHeartbeatGap { t.Fatalf("Reason = %q, want heartbeat_gap (pinned, not the desired end state): proveCoverage has no "+ - "'skipped' concept, so decide (P8) must handle a skipped rule's classification itself, before or "+ + "'skipped' concept, so decide must handle a skipped rule's classification itself, before or "+ "instead of calling this function", res.Reason) } } diff --git a/grafana-alertcheck/internal/gate/duration.go b/grafana-alertcheck/internal/gate/duration.go index 011c27f14..3690a2549 100644 --- a/grafana-alertcheck/internal/gate/duration.go +++ b/grafana-alertcheck/internal/gate/duration.go @@ -29,7 +29,7 @@ var promDurationUnits = []promDurationUnit{ } // ParsePromDuration parses a Grafana/Prometheus-style duration ("1h30m", "1d", "1w"). -// Unlike time.ParseDuration, it accepts "d" and "w" (§11.8). "" and "0" are 0. +// Unlike time.ParseDuration, it accepts "d" and "w". "" and "0" are 0. func ParsePromDuration(s string) (time.Duration, error) { if s == "" || s == "0" { return 0, nil diff --git a/grafana-alertcheck/internal/gate/flock.go b/grafana-alertcheck/internal/gate/flock.go index f2a3dbb24..ae21f943b 100644 --- a/grafana-alertcheck/internal/gate/flock.go +++ b/grafana-alertcheck/internal/gate/flock.go @@ -8,7 +8,7 @@ import ( ) // lockExclusive takes a non-blocking exclusive lock on f. Non-blocking is the -// point (§8): a second writer must fail immediately with an error the operator +// point: a second writer must fail immediately with an error the operator // sees, not queue behind the first and start appending to a log somebody else // already finished. func lockExclusive(f *os.File) error { @@ -30,7 +30,7 @@ func isLockContention(err error) bool { // // check needs that distinction where NewWriter does not. NewWriter is entitled // to treat any refusal as "another writer has it", because it wants the lock; -// check only wants to know whether a writer EXISTS (§4.4). The lock answers +// check only wants to know whether a writer EXISTS. The lock answers // that directly, where a pid can only infer it — the kernel releases a flock // when the holder exits, crash included, and pids get reused. func tryLockExclusive(f *os.File) (held bool, err error) { diff --git a/grafana-alertcheck/internal/gate/jsonreq.go b/grafana-alertcheck/internal/gate/jsonreq.go index bfddca382..5b0799a55 100644 --- a/grafana-alertcheck/internal/gate/jsonreq.go +++ b/grafana-alertcheck/internal/gate/jsonreq.go @@ -8,7 +8,7 @@ import ( // req decodes m[key] into *dst. It returns an error when key is absent from m // or explicitly JSON null, so a caller can never mistake absence for a zero -// value (H1) — json.Unmarshal treats "null" as a documented no-op for +// value — json.Unmarshal treats "null" as a documented no-op for // non-pointer targets (string, bool, int, ...), so without this check a // required field sent as null would silently pass through as its zero value. func req[T any](m map[string]json.RawMessage, key string, dst *T) error { diff --git a/grafana-alertcheck/internal/gate/log.go b/grafana-alertcheck/internal/gate/log.go index 243060535..2fdb71d5f 100644 --- a/grafana-alertcheck/internal/gate/log.go +++ b/grafana-alertcheck/internal/gate/log.go @@ -13,11 +13,11 @@ import ( // LogSchemaVersion is the version stamped into every log header. A log with // any other value is a read error, never a best-effort read: the log is the -// gate's only evidence, and misreading a stale shape is a fail-open (§5). +// gate's only evidence, and misreading a stale shape is a fail-open. const LogSchemaVersion = 1 // RecordType tags each JSONL line. There are exactly three, and a poll record -// IS the heartbeat — there is deliberately no separate heartbeat type (§4.6). +// IS the heartbeat — there is deliberately no separate heartbeat type. type RecordType string const ( @@ -28,13 +28,13 @@ const ( // missingSeriesReason is the reason Grafana parks a disappearing series at // ("Normal (MissingSeries)") for a couple of evaluations before deleting the -// instance. Reading that as a recovery is H2's named bug, so the markers below -// route it to Vanished (P1.2a). +// instance. Reading that as a recovery would turn a disappearing series into a +// fake recovery, so the markers below route it to Vanished. const missingSeriesReason = "MissingSeries" // LoggedRule is the per-rule identity written into the header. Together with -// the header URL it IS the log's identity, which check validates (§19.1 step -// 3), and it supplies the alert set in check mode. +// the header URL it IS the log's identity, which check validates, and it +// supplies the alert set in check mode. type LoggedRule struct { UID string `json:"uid"` Title string `json:"title"` @@ -42,17 +42,17 @@ type LoggedRule struct { Group string `json:"group"` // ForSeconds, IntervalSeconds, NoDataState and ExecErrState are purely // forensic: a resolve-time snapshot that makes the uploaded artifact - // self-describing to a human reading it after the runner is gone (§21.3). - // check never converts them back into a Definition — it always re-resolves - // definitions from the ruler API (§19.1 step 2). + // self-describing to a human reading it after the runner is gone. check + // never converts them back into a Definition — it always re-resolves + // definitions from the ruler API. ForSeconds float64 `json:"for_seconds"` IntervalSeconds int `json:"interval_seconds"` // IsPaused is NOT forensic, and is the second load-bearing field here // beside PollEverySeconds. It is the pause state at record start, which is - // the only moment `skipped` can honestly mean (§12), and decide reads it - // through Header.pausedAtStart rather than reading Definition.IsPaused off - // a ruler read taken after the window had already closed. See that method - // for what goes wrong the other way. + // the only moment `skipped` can honestly mean, and decide reads it through + // Header.pausedAtStart rather than reading Definition.IsPaused off a ruler + // read taken after the window had already closed. See that method for what + // goes wrong the other way. IsPaused bool `json:"is_paused"` NoDataState string `json:"no_data_state"` ExecErrState string `json:"exec_err_state"` @@ -60,34 +60,34 @@ type LoggedRule struct { // --poll-interval override. Load-bearing, not forensic: check derives // maxGap from it and never re-derives it from the definitions. Getting // that wrong is fail-open in the faster-override direction — a real - // recorder gap would pass silently (see "Two authorities", P5). + // recorder gap would pass silently. PollEverySeconds float64 `json:"poll_every_seconds"` } // Header is the log's first line: what was recorded, from where, and when the // recording started. It carries no States field — recording is deliberately // unfiltered, so the same log can be re-classified under different --states -// without re-recording (P6). +// without re-recording. type Header struct { SchemaVersion int `json:"schema_version"` - URL string `json:"url"` // the log's identity (§19.1 step 3) + URL string `json:"url"` // the log's identity GrafanaVersion string `json:"grafana_version"` - StartedAt time.Time `json:"started_at"` // the record start (§7 validation) - Rules []LoggedRule `json:"rules"` // THE alert set (§19.1 step 3) + StartedAt time.Time `json:"started_at"` // the record start + Rules []LoggedRule `json:"rules"` // THE alert set } // pausedAtStart reports, per rule UID, whether the rule was paused when the -// recording opened. That instant — and no other — is what `skipped` means -// (§12): a rule nobody was watching on purpose. +// recording opened. That instant — and no other — is what `skipped` means: a +// rule nobody was watching on purpose. // // It is the authority for `skipped` in BOTH modes, and the reason is that no // other source knows the right moment. `check` re-resolves the definitions -// AFTER the window closed (§19.1 step 2), so Definition.IsPaused there -// describes the present, not the window: a rule that fired and was then -// paused would read as skipped, its firing would never be classified, and -// under --allow-paused the run would pass. The header cannot drift that way, -// because watch stamps it before the deploy step runs and single-step check -// stamps it from definitions resolved at the start of its own step. +// AFTER the window closed, so Definition.IsPaused there describes the present, +// not the window: a rule that fired and was then paused would read as skipped, +// its firing would never be classified, and under --allow-paused the run would +// pass. The header cannot drift that way, because watch stamps it before the +// deploy step runs and single-step check stamps it from definitions resolved at +// the start of its own step. // // A UID the header does not name is reported NOT paused, which is the safe // direction: it then reaches proveCoverage with no polls and fails closed as @@ -104,7 +104,7 @@ func (h Header) pausedAtStart() map[string]bool { // only input the pure coverage and classification layers ever see. type Poll struct { RuleUID string `json:"rule_uid"` - GrafanaNow time.Time `json:"grafana_now"` // the Date header — H4 + GrafanaNow time.Time `json:"grafana_now"` // the response's Date header // SkewMS, SkewBoundMS and LatencyMS are milliseconds for JSONL // compactness ONLY. The pure layer never touches raw ms: it reads // Skew(), SkewBound() and Latency() below, which convert at the @@ -112,21 +112,21 @@ type Poll struct { SkewMS int64 `json:"skew_ms"` SkewBoundMS int64 `json:"skew_bound_ms"` LatencyMS int64 `json:"latency_ms"` - // Found false means an authoritative 2xx in which this rule was absent - // (§14.5) — never a transport failure, which P2 retried and never turns - // into a Poll. P7 check 8 turns it into unobservable. + // Found false means an authoritative 2xx in which this rule was absent — + // never a transport failure, which the transport retries and never turns + // into a Poll. The coverage proof turns it into unobservable. Found bool `json:"found"` // State, Health and LastError are the raw rule-level strings, reporting - // only and never classified (P1.2a). + // only and never classified. State string `json:"state,omitempty"` Health string `json:"health,omitempty"` LastError string `json:"last_error,omitempty"` - // omitzero, not omitempty: a not-found poll (and a paused rule, §2.3) has - // no evaluation time, and writing "0001-01-01T00:00:00Z" into an artifact - // humans and jq read (§21.3) invites reading it as a real timestamp. + // omitzero, not omitempty: a not-found poll (and a paused rule) has no + // evaluation time, and writing "0001-01-01T00:00:00Z" into an artifact + // humans and jq read invites reading it as a real timestamp. LastEvaluation time.Time `json:"last_evaluation,omitzero"` IsPaused bool `json:"is_paused"` - Histogram map[string]int `json:"histogram,omitempty"` // §4.9 — written, never analysed + Histogram map[string]int `json:"histogram,omitempty"` // written, never analysed // Reasons counts this poll's non-empty instance reasons, e.g. // {"NoData":1091,"Error":14}; nil when none. Reporting-only, and the ONLY // place composite states stay visible: they are canonical normal (so they @@ -134,37 +134,37 @@ type Poll struct { // // The KEYS are raw reason strings and can be comma-joined composites // ("KeepLast, MissingSeries") — newer Grafana versions join several - // reasons into one. So any consumer, P7 check 9's KeepLast note included, - // must test membership across the keys with reasonNames and must NEVER - // index a literal key: reasons["KeepLast"] misses every composite. + // reasons into one. So any consumer, the coverage proof's KeepLast note + // included, must test membership across the keys with reasonNames and must + // NEVER index a literal key: reasons["KeepLast"] misses every composite. Reasons map[string]int `json:"reasons,omitempty"` - // Abnormal holds the instances whose CANONICAL state is not normal - // (§4.6). "Normal (NoData)" and "Normal (Error)" are canonical normal and - // are deliberately not retained here (P1.2a). + // Abnormal holds the instances whose CANONICAL state is not normal. + // "Normal (NoData)" and "Normal (Error)" are canonical normal and are + // deliberately not retained here. Abnormal []Instance `json:"abnormal,omitempty"` - // Cleared and Vanished are instance keys (§4.7): keys that left the - // abnormal set, resolved against the SAME response — a clear and a - // discontinuity are not the same fact (H2). + // Cleared and Vanished are instance keys that left the abnormal set, + // resolved against the SAME response — a clear and a discontinuity are not + // the same fact. Cleared []string `json:"cleared,omitempty"` Vanished []string `json:"vanished,omitempty"` } -// Skew is the signed clock skew of this poll (§16). +// Skew is the signed clock skew of this poll. func (p Poll) Skew() time.Duration { return time.Duration(p.SkewMS) * time.Millisecond } // SkewBound is the uncertainty on Skew — the tolerance every cross-domain -// comparison in P7 applies alongside it. +// comparison applies alongside it. func (p Poll) SkewBound() time.Duration { return time.Duration(p.SkewBoundMS) * time.Millisecond } -// Latency is the wall time this poll's request took, feeding §5.2's budget check. +// Latency is the wall time this poll's request took, feeding the budget check. func (p Poll) Latency() time.Duration { return time.Duration(p.LatencyMS) * time.Millisecond } // Reducer turns each Observation into the single Poll record that goes into // the log. It holds the previous poll's abnormal instance keys per rule, which -// is all the state the transition markers need (§4.7). +// is all the state the transition markers need. // // A Reducer is safe for concurrent use: watch polls a fleet of rules -// concurrently (P6) and every one of those goroutines reduces through the same +// concurrently and every one of those goroutines reduces through the same // instance, because the per-rule marker state has to live in one place. The // lock is per-Reducer rather than per-rule — Reduce only touches maps and // slices, so it never blocks on I/O while holding it. @@ -179,12 +179,12 @@ func NewReducer() *Reducer { // Reduce selects the rule identified by uid out of obs and reduces it to a // Poll. Selection is BY UID, never by title: a filtered response can carry -// several rules sharing one title (the known 2-way collision, §14.5), and -// picking the first would silently watch the wrong rule. +// several rules sharing one title, and picking the first would silently watch +// the wrong rule. // -// The reduction (§4.6) keeps the rule-level fields, the raw totals histogram, -// the reason counts, and only the instances whose canonical state is not -// normal. That makes per-poll size independent of NORMAL cardinality — not of +// The reduction keeps the rule-level fields, the raw totals histogram, the +// reason counts, and only the instances whose canonical state is not normal. +// That makes per-poll size independent of NORMAL cardinality — not of // cardinality outright: a rule with 449 firing instances still stores all 449. func (r *Reducer) Reduce(uid string, obs Observation) Poll { r.mu.Lock() @@ -216,8 +216,8 @@ func (r *Reducer) Reduce(uid string, obs Observation) Poll { p.Histogram = rule.Totals // present indexes every instance in THIS response, normal ones included — - // the markers below must resolve a departed key against the same response - // (H2), which is impossible from the abnormal subset alone. + // the markers below must resolve a departed key against the same response, + // which is impossible from the abnormal subset alone. present := make(map[string]Instance, len(rule.Instances)) curAbnormal := make(map[string]struct{}) for _, inst := range rule.Instances { @@ -246,7 +246,7 @@ func (r *Reducer) Reduce(uid string, obs Observation) Poll { p.Vanished = append(p.Vanished, key) case reasonNames(inst.Reason, missingSeriesReason): // The vanish in disguise, caught one poll earlier than the fully - // absent case — H2's named bug. + // absent case. p.Vanished = append(p.Vanished, key) default: // Present as canonical normal without a MissingSeries reason. @@ -268,10 +268,10 @@ func (r *Reducer) Reduce(uid string, obs Observation) Poll { // // It exists for the one place a recording changes hands: watch's parent takes // the first observation of every rule and its detached child continues from -// there (P6). Without the seed, an instance that is abnormal in the parent's +// there. Without the seed, an instance that is abnormal in the parent's // observation and gone by the child's first poll produces no marker at all — -// it leaves the record as though it had never been bad, which is H2's -// fail-open reached through the handoff rather than through a reason string. +// it leaves the record as though it had never been bad, the same fail-open a +// misread MissingSeries causes, reached through the handoff instead. // // Not-found polls are skipped, mirroring Reduce: an absent rule leaves the // previous abnormal set untouched rather than emptying it. @@ -291,7 +291,7 @@ func (r *Reducer) seedFrom(polls []Poll) { } // stateRuleByUID picks one rule out of a state-endpoint response BY UID, and -// nil means the response is an authoritative "the rule is absent" (§14.5). +// nil means the response is an authoritative "the rule is absent". // // Never by title: the ?rule_name= filter is a title filter, and a filtered // response can carry several rules sharing one title (the known 2-way @@ -311,7 +311,7 @@ func stateRuleByUID(rules []StateRule, uid string) *StateRule { // reasonNames reports whether reason names want. Newer Grafana versions // comma-join several reasons into one string, so this tests membership rather -// than equality (P7 check 9 needs the same test for KeepLast). +// than equality. func reasonNames(reason, want string) bool { for part := range strings.SplitSeq(reason, ",") { if strings.TrimSpace(part) == want { @@ -321,12 +321,12 @@ func reasonNames(reason, want string) bool { return false } -// VerifyNormalInstancesVisible checks §3.2's assumption on a first -// observation: that the state endpoint really does return normal instances, -// not only the abnormal ones. If it ever stops doing so, the reduction's -// "keep the non-normal instances" becomes "keep everything the API happened to -// send" and the transition markers lose their ground truth — a silent -// fail-open. So this is verified at start, never assumed. +// VerifyNormalInstancesVisible checks, on a first observation, that the state +// endpoint really does return normal instances and not only the abnormal ones. +// If it ever stops doing so, the reduction's "keep the non-normal instances" +// becomes "keep everything the API happened to send" and the transition markers +// lose their ground truth — a silent fail-open. So this is verified at start, +// never assumed. // // The counts are summed over every totals key whose LOWERCASED name is // "normal" or "inactive". Never index one literal key: the captured @@ -351,7 +351,7 @@ func VerifyNormalInstancesVisible(rules []StateRule) error { } return fmt.Errorf( "rule %q (%s): totals claim %d normal instances but the response returned none — "+ - "the state endpoint no longer returns normal instances, which the §3.2 reduction depends on", + "the state endpoint no longer returns normal instances, which the reduction depends on", r.Title, r.UID, claimed) } return nil @@ -384,10 +384,10 @@ type stoppedRecord struct { At time.Time `json:"at"` } -// Writer appends records to the JSONL log. It is append-only by construction -// (§8): O_APPEND|O_CREATE|O_WRONLY, never O_TRUNC, so no writer can ever -// destroy evidence a previous one recorded. An exclusive non-blocking flock -// makes a second writer fail immediately rather than interleave. +// Writer appends records to the JSONL log. It is append-only by construction — +// O_APPEND|O_CREATE|O_WRONLY, never O_TRUNC — so no writer can ever destroy +// evidence a previous one recorded. An exclusive non-blocking flock makes a +// second writer fail immediately rather than interleave. type Writer struct { mu sync.Mutex f *os.File @@ -418,8 +418,8 @@ func NewWriter(path string, clock Clock) (*Writer, error) { // WriteHeader writes line 1 and stamps the current schema version, so no // caller can leave it at zero. It refuses a non-empty file: the log already // has a header, and a second one would make ReadLog's "header is line 1" -// contract a lie. In the P6 handoff the parent writes the header and the child -// only appends polls. +// contract a lie. In watch's handoff the parent writes the header and the +// detached child only appends polls. func (w *Writer) WriteHeader(h Header) error { w.mu.Lock() defer w.mu.Unlock() @@ -453,15 +453,14 @@ func (w *Writer) WritePoll(p Poll) error { return nil } -// Stop finishes recording in the fixed §4.4 order, which must not be -// reordered: let the in-flight write finish (the mutex), append the stopped -// sentinel, fsync, then release. Any other order can leave a log whose last -// durable byte is a sentinel that was never actually preceded by the polls it -// vouches for. +// Stop finishes recording in a fixed order that must not be rearranged: let +// the in-flight write finish (the mutex), append the stopped sentinel, fsync, +// then release. Any other order can leave a log whose last durable byte is a +// sentinel that was never actually preceded by the polls it vouches for. // // Stop writes the sentinel with the recorder's OWN stop time and makes no // comparison against `to` — watch never knows `to` or the transition grace. -// check does that comparison, after this writer has exited (§4.5). +// check does that comparison, after this writer has exited. // // Calling Stop twice is a no-op: watch reaches it from both a signal handler // and a defer, and a second sentinel would be indistinguishable from a second @@ -491,9 +490,8 @@ func (w *Writer) Stop() error { // Close releases the file and the lock WITHOUT writing a sentinel. It exists // for exactly one caller: watch's parent, which writes the header and then -// hands the log to the detached child that will finish it (P6). A sentinel -// here would tell check the recording ended before the child had even -// started. +// hands the log to the detached child that will finish it. A sentinel here +// would tell check the recording ended before the child had even started. func (w *Writer) Close() error { w.mu.Lock() defer w.mu.Unlock() @@ -510,15 +508,15 @@ func (w *Writer) Close() error { // ReadLogHeader reads ONLY line 1 and is the one read of a log that a writer // may still hold. That is safe for exactly one line and for no other: the // header is written once, by watch's parent, before any child appends a byte, -// the file is opened O_APPEND and never O_TRUNC (§8), so line 1 is complete -// and immutable for the whole life of the recording. +// and the file is opened O_APPEND and never O_TRUNC, so line 1 is complete and +// immutable for the whole life of the recording. // -// It exists so check can fail closed EARLY (§19.1 steps 3-4): the log's -// identity, the rule set and the cadences are all knowable at the start, and -// discovering a wrong URL or an unresolvable rule after a ten-minute wait -// helps nobody. It is advisory only — the authoritative read is still ReadLog, -// once, after the writer has exited (§4.4 step 4), and check re-validates the -// identity against that header rather than trusting this one. +// It exists so check can fail closed EARLY: the log's identity, the rule set +// and the cadences are all knowable at the start, and discovering a wrong URL +// or an unresolvable rule after a ten-minute wait helps nobody. It is advisory +// only — the authoritative read is still ReadLog, once, after the writer has +// exited, and check re-validates the identity against that header rather than +// trusting this one. func ReadLogHeader(path string) (Header, error) { f, err := os.Open(path) if err != nil { @@ -553,12 +551,11 @@ func ReadLogHeader(path string) (Header, error) { // recording never finished — check turns that into unobservable, never a // pass). // -// Call this only after the writer has exited (§4.4 step 4). Reading a log a -// writer can still append to can only produce a shorter window than the one -// that was recorded. +// Call this only after the writer has exited. Reading a log a writer can still +// append to can only produce a shorter window than the one that was recorded. // -// The parse rules are deliberately the crudest possible (§24.2): the header -// must be line 1 with a matching schema version, and ANY unparseable line — +// The parse rules are deliberately the crudest possible: the header must be +// line 1 with a matching schema version, and ANY unparseable line — // including the last one, and including a last line that follows a sentinel — // is an error, full stop. No heuristics, no discarding an untidy tail: a // truncated log is evidence that something killed the recorder, which is diff --git a/grafana-alertcheck/internal/gate/log_test.go b/grafana-alertcheck/internal/gate/log_test.go index 494c9535c..179adb5fe 100644 --- a/grafana-alertcheck/internal/gate/log_test.go +++ b/grafana-alertcheck/internal/gate/log_test.go @@ -48,7 +48,7 @@ func TestLogReduceKeepsOnlyAbnormalInstances(t *testing.T) { Instances: []Instance{ testInstance(StateNormal, "", "a"), testInstance(StateFiring, "", "b"), - // Both composites are canonical normal (P1.2a): they must NOT be + // Both composites are canonical normal: they must NOT be // retained as abnormal, and their reasons must still be counted. testInstance(StateNormal, "NoData", "c"), testInstance(StateNormal, "Error", "d"), @@ -67,11 +67,11 @@ func TestLogReduceKeepsOnlyAbnormalInstances(t *testing.T) { t.Errorf("Reasons = %v, want %v", p.Reasons, want) } // The histogram is a verbatim copy of the response totals — raw keys, no - // normalization (§4.9). + // normalization. if want := map[string]int{"alerting": 1, "normal": 2}; !reflect.DeepEqual(p.Histogram, want) { t.Errorf("Histogram = %v, want %v", p.Histogram, want) } - // Rule-level state and health stay raw and unnormalized (P1.2a). + // Rule-level state and health stay raw and unnormalized. if p.State != "firing" || p.Health != "ok" { t.Errorf("State/Health = %q/%q, want firing/ok", p.State, p.Health) } @@ -83,8 +83,8 @@ func TestLogReduceKeepsOnlyAbnormalInstances(t *testing.T) { } } -// A filtered response can hold several rules sharing one title (the known -// 2-way collision, §14.5), so the reducer must select by UID. +// A filtered response can hold several rules sharing one title, so the reducer +// must select by UID. func TestLogReduceSelectsRuleByUID(t *testing.T) { first := StateRule{UID: "ruleA", Title: "Same Title", Health: "ok", State: "inactive", LastEvaluation: testNow} second := StateRule{ @@ -120,7 +120,7 @@ func TestLogReduceRuleAbsentIsAuthoritative(t *testing.T) { } } -// H2: an instance that leaves the abnormal set is resolved against the SAME +// An instance that leaves the abnormal set is resolved against the SAME // response, and MissingSeries is a vanish, never a recovery. func TestTransitionMarkersClearedVersusVanished(t *testing.T) { badKey := instanceKey(testInstance(StateFiring, "", "b").Labels) @@ -249,8 +249,8 @@ func TestTransitionMarkersAreSortedAndPerRule(t *testing.T) { } } -// §3.2: the reduction depends on the state endpoint returning normal instances. -// If it ever stops, that must fail loudly at start, never be assumed. +// The reduction depends on the state endpoint returning normal instances. If it +// ever stops, that must fail loudly at start, never be assumed. func TestLogVerifyNormalInstancesVisible(t *testing.T) { cases := []struct { fixture string @@ -275,8 +275,8 @@ func TestLogVerifyNormalInstancesVisible(t *testing.T) { if err == nil { t.Fatalf("VerifyNormalInstancesVisible: want an error, got nil") } - if !strings.Contains(err.Error(), "§3.2") { - t.Errorf("error does not name §3.2: %v", err) + if !strings.Contains(err.Error(), "no longer returns normal instances") { + t.Errorf("error does not say the endpoint stopped returning normal instances: %v", err) } return } @@ -371,7 +371,7 @@ func TestLogModeCadenceComesFromTheHeader(t *testing.T) { } } -// watch polls a fleet concurrently through one Reducer (P6), so the marker +// watch polls a fleet concurrently through one Reducer, so the marker // state it holds per rule must be safe under -race — a latent data race here // surfaces as a wrong transition, which is the one thing markers exist to get // right. @@ -405,7 +405,7 @@ func TestLogReduceIsSafeForConcurrentUse(t *testing.T) { } // A not-found poll has no evaluation time, and the artifact is read by humans -// and jq (§21.3) — the zero time must not appear as though it were real. +// and jq — the zero time must not appear as though it were real. func TestLogPollOmitsTheZeroEvaluationTime(t *testing.T) { absent := NewReducer().Reduce("rule1", observation(testNow)) b, err := json.Marshal(pollRecord{Type: RecordPoll, Poll: absent}) @@ -504,13 +504,13 @@ func TestWriterReadLogRoundTrip(t *testing.T) { t.Fatalf("sentinel is nil after Stop") } // Stop stamps the recorder's own stop time and makes no comparison - // against `to` — watch never knows it (§4.5). + // against `to` — watch never knows it. if !sentinel.Equal(testNow.Add(2 * time.Minute)) { t.Errorf("sentinel = %s, want the writer's stop time %s", sentinel, testNow.Add(2*time.Minute)) } } -// §8: the log is append-only. A second run against the same path must never +// The log is append-only. A second run against the same path must never // destroy the evidence the first one recorded. func TestWriterAppendsAndNeverTruncates(t *testing.T) { path := filepath.Join(t.TempDir(), "log.jsonl") @@ -529,7 +529,7 @@ func TestWriterAppendsAndNeverTruncates(t *testing.T) { t.Fatalf("read: %v", err) } - // The P6 handoff: the parent wrote the header and closed; the child + // The handoff: the parent wrote the header and closed; the child // reopens the same path and appends without a second header. child, _ := newTestWriter(t, path) if err := child.WritePoll(Poll{RuleUID: "rule1", Found: true, GrafanaNow: testNow.Add(time.Minute)}); err != nil { @@ -633,8 +633,8 @@ func TestSentinelStopIsIdempotentAndLast(t *testing.T) { } } -// Close is the parent's handoff path in P6: a sentinel there would tell check -// the recording ended before the child had even started. +// Close is the parent's handoff path: a sentinel there would tell check the +// recording ended before the child had even started. func TestSentinelCloseWritesNone(t *testing.T) { path := filepath.Join(t.TempDir(), "log.jsonl") w, _ := newTestWriter(t, path) @@ -658,8 +658,9 @@ func TestSentinelCloseWritesNone(t *testing.T) { } // An unfinished recording reads cleanly with a nil sentinel — ReadLog reports -// the absence and P7 turns it into unobservable. It is never ReadLog's job to -// call that a failure, and never anyone's job to call it a pass. +// the absence and the coverage proof turns it into unobservable. It is never +// ReadLog's job to call that a failure, and never anyone's job to call it a +// pass. func TestReadLogWithoutASentinel(t *testing.T) { path := filepath.Join(t.TempDir(), "log.jsonl") w, _ := newTestWriter(t, path) @@ -682,7 +683,7 @@ func TestReadLogWithoutASentinel(t *testing.T) { } } -// The read rules are deliberately the crudest possible (§24.2): any unparseable +// The read rules are deliberately the crudest possible: any unparseable // line is an error, full stop — including the last one, and including a last // line that follows a sentinel. func TestReadLogRejectsBadLogs(t *testing.T) { @@ -778,7 +779,7 @@ func TestReadLogMissingFile(t *testing.T) { } } -// §22.3: per-poll log size must not grow across polls on a high-cardinality +// Per-poll log size must not grow across polls on a high-cardinality // rule, and the one firing instance among 2446 must still be attributed by its // labels. The reduction makes size independent of NORMAL cardinality — the // firing instances are still stored, which is why a clear shrinks the record. @@ -851,8 +852,8 @@ func TestLogSizeIsFlatAcrossPollsOnAHighCardinalityRule(t *testing.T) { } // The log must stay readable by anything that reads JSONL, one flat object per -// line with its type tag — an uploaded artifact (§21.3) is read by humans and -// by jq, not only by ReadLog. +// line with its type tag — an uploaded artifact is read by humans and by jq, +// not only by ReadLog. func TestLogRecordsAreFlatOneLineObjects(t *testing.T) { path := filepath.Join(t.TempDir(), "log.jsonl") w, _ := newTestWriter(t, path) diff --git a/grafana-alertcheck/internal/gate/parse_ruler.go b/grafana-alertcheck/internal/gate/parse_ruler.go index c27e5b9c4..ae878d4c9 100644 --- a/grafana-alertcheck/internal/gate/parse_ruler.go +++ b/grafana-alertcheck/internal/gate/parse_ruler.go @@ -7,9 +7,9 @@ import ( "time" ) -// RuleKind classifies a ruler-endpoint rule by shape, not by name (P1.3). -// P3 rejects KindDatasourceManaged and KindRecording, but only for rules a -// user actually named — ParseDefinitions itself never rejects. +// RuleKind classifies a ruler-endpoint rule by shape, not by name. Resolve +// rejects KindDatasourceManaged and KindRecording, but only for rules a user +// actually named — ParseDefinitions itself never rejects. type RuleKind int const ( @@ -22,8 +22,8 @@ const ( // (/api/ruler/grafana/api/v1/rules). IntervalSeconds, NoDataState and // ExecErrState live inside the grafana_alert block and are only populated for // KindGrafanaManaged — a datasource-managed rule has no such block by -// definition (§11.6 drops relativeTimeRange/keep_firing_for entirely; neither -// is parsed here). +// definition. relativeTimeRange and keep_firing_for are deliberately not +// parsed: nothing in the gate reads them. type Definition struct { UID, Title, Folder, FolderUID, Group string For time.Duration @@ -43,8 +43,8 @@ func ParseDefinitions(body []byte) ([]Definition, error) { } // Map iteration order is nondeterministic; sort namespace names so - // ParseDefinitions' output order is stable across calls (P3's candidate - // listings and any golden test depend on that). + // ParseDefinitions' output order is stable across calls — Resolve's + // candidate listings and the golden tests depend on that. names := make([]string, 0, len(namespaces)) for name := range namespaces { names = append(names, name) @@ -137,11 +137,11 @@ func parseDefinition(raw json.RawMessage, folder, group string) (Definition, err // Classify by the presence of "record" before requiring anything else. // no_data_state/exec_err_state/is_paused/intervalSeconds are alerting-only // concepts a recording rule may not carry at all — its real shape is - // unverified (none exist in the fleet capture) — and P3 refuses this - // Kind categorically before any of this would gate a release. Strict- - // parsing a recording rule into a hard error over fields it was never - // going to use would brick `list` and every resolve for rules nobody - // named (§11.6, "do not reject here"). + // unverified, none exist in the fleet capture — and Resolve refuses this + // Kind categorically before any of this would gate a release. + // Strict-parsing a recording rule into a hard error over fields it was + // never going to use would brick `list` and every resolve for rules nobody + // named. var record json.RawMessage if err := opt(ga, "record", &record); err != nil { return Definition{}, fmt.Errorf("rule %q: grafana_alert: %w", uid, err) diff --git a/grafana-alertcheck/internal/gate/parse_ruler_test.go b/grafana-alertcheck/internal/gate/parse_ruler_test.go index 3cf2e3261..b225ffc2c 100644 --- a/grafana-alertcheck/internal/gate/parse_ruler_test.go +++ b/grafana-alertcheck/internal/gate/parse_ruler_test.go @@ -20,7 +20,7 @@ func TestParseDefinitions_RulerRules(t *testing.T) { } // The real 2-way duplicate title: same folder, same group, same title, - // distinct UIDs (§17, §22.2). + // distinct UIDs — only uid: can tell them apart. a, ok := byUID["rule0000006a"] if !ok { t.Fatalf("missing rule0000006a") @@ -81,7 +81,8 @@ func TestParseDefinitions_DatasourceManaged(t *testing.T) { } // A datasource-managed rule has no uid in this shape; its only identity // is the Prometheus "alert" name — a synthetic UID would be invented - // shape, and an empty Title would make P3's refusal-by-name unreachable. + // shape, and an empty Title would make Resolve's refusal-by-name + // unreachable. if defs[0].Title != "ExampleTargetDown" { t.Errorf("Title = %q, want ExampleTargetDown", defs[0].Title) } diff --git a/grafana-alertcheck/internal/gate/parse_state.go b/grafana-alertcheck/internal/gate/parse_state.go index 7ffff1cfc..7c35c7e7d 100644 --- a/grafana-alertcheck/internal/gate/parse_state.go +++ b/grafana-alertcheck/internal/gate/parse_state.go @@ -7,7 +7,7 @@ import ( "time" ) -// State is the canonical instance state (P1.2a). It is distinct from the raw, +// State is the canonical instance state. It is distinct from the raw, // unnormalized vocabularies the API uses at the rule level and at the instance // level — see normalizeInstanceState. type State string @@ -22,12 +22,12 @@ const ( // Instance is one entry of a rule's alerts[]. State is always canonical; Reason // is the opaque suffix of a "State (Reason)" composite ("" when the API gave a -// bare state). Reason is reporting-only except for the H2 MissingSeries routing +// bare state). Reason is reporting-only except for the MissingSeries routing // done downstream in the log markers. // -// The json tags are for the JSONL log's abnormal-instance list (P5) only — -// parsing an API response never goes through them, because parseInstance -// decodes field by field through req/opt to keep H1's presence checks explicit. +// The json tags are for the JSONL log's abnormal-instance list only — parsing +// an API response never goes through them, because parseInstance decodes field +// by field through req/opt to keep the presence checks explicit. type Instance struct { Labels map[string]string `json:"labels"` State State `json:"state"` @@ -37,12 +37,12 @@ type Instance struct { } // StateRule is one rule from the state endpoint -// (/api/prometheus/grafana/api/v1/rules), fully and strictly parsed (H1). +// (/api/prometheus/grafana/api/v1/rules), fully and strictly parsed. type StateRule struct { UID, Title, Folder, Group string Interval time.Duration // State and Health are raw, lowercase, and reporting-only — never - // classified (P1.2a). State in particular is never normalized. + // classified. State in particular is never normalized. State, Health string LastError string LastEvaluation time.Time @@ -53,7 +53,7 @@ type StateRule struct { // ParseState strictly parses a state-endpoint response body into its rules. // A missing or unparseable required field (health, state, lastEvaluation on -// each rule; interval on each group) is an error, never a zero value (H1). +// each rule; interval on each group) is an error, never a zero value. func ParseState(body []byte) ([]StateRule, error) { var top map[string]json.RawMessage if err := json.Unmarshal(body, &top); err != nil { @@ -133,11 +133,10 @@ func parseStateRule(raw json.RawMessage, folder, group string, interval time.Dur if err := req(m, "health", &r.Health); err != nil { return StateRule{}, fmt.Errorf("rule %q: %w", uid, err) } - // isPaused is not one of H1's four named required fields, but this parser - // extends that contract to it: the zero-time rule below can't tell a - // paused rule from a broken one without it, and it's the primary - // in-window pause detector (H2/§12.2) — a silent false default would be - // exactly the fail-open bug H1 exists to kill. + // isPaused is required rather than optional: the zero-time rule below + // can't tell a paused rule from a broken one without it, and it's the + // primary in-window pause detector — a silent false default would be + // exactly the fail-open this parser's strictness exists to kill. if err := req(m, "isPaused", &r.IsPaused); err != nil { return StateRule{}, fmt.Errorf("rule %q: %w", uid, err) } @@ -150,7 +149,7 @@ func parseStateRule(raw json.RawMessage, folder, group string, interval time.Dur if err != nil { return StateRule{}, fmt.Errorf("rule %q: lastEvaluation: %w", uid, err) } - // The zero-time rule (§2.3): only a paused rule may report the zero time. + // Only a paused rule may report the zero time. if lastEval.IsZero() && !r.IsPaused { return StateRule{}, fmt.Errorf("rule %q: lastEvaluation is the zero time but isPaused is false", uid) } @@ -198,10 +197,10 @@ func parseInstance(raw json.RawMessage) (Instance, error) { return Instance{}, err } - // activeAt is also not in H1's named list, extended here for the same - // reason as StateRule.IsPaused: it's the onset time BadFor (P8) measures - // from, so a silently zeroed one would misclassify how long an instance - // has been bad rather than failing loudly. + // activeAt is required for the same reason as StateRule.IsPaused: it's the + // onset time BadFor measures from, so a silently zeroed one would + // misclassify how long an instance has been bad rather than failing + // loudly. var activeAtStr string if err := req(m, "activeAt", &activeAtStr); err != nil { return Instance{}, err @@ -224,8 +223,8 @@ func parseInstance(raw json.RawMessage) (Instance, error) { } // baseInstanceStates is the strict 5-value allowlist for the base of an -// instance state (P1.2a). Anything else — including an unrecognized base -// inside a "Base (Reason)" composite — is a parse error (H1, §2.7 control 3). +// instance state. Anything else — including an unrecognized base inside a +// "Base (Reason)" composite — is a parse error. var baseInstanceStates = map[string]State{ "Normal": StateNormal, "Alerting": StateFiring, diff --git a/grafana-alertcheck/internal/gate/parse_state_test.go b/grafana-alertcheck/internal/gate/parse_state_test.go index bc8a2a64c..4ccca1391 100644 --- a/grafana-alertcheck/internal/gate/parse_state_test.go +++ b/grafana-alertcheck/internal/gate/parse_state_test.go @@ -167,7 +167,7 @@ func TestParseState_HappyPaths(t *testing.T) { t.Fatalf("Instances = %+v, want one firing instance", r.Instances) } if r.Totals["normal"] == 0 { - t.Errorf(`Totals["normal"] = 0, want >0 (this is the §3.2 mismatch the fixture exists to capture)`) + t.Errorf(`Totals["normal"] = 0, want >0 (the totals/instances mismatch this fixture exists to capture)`) } }, }, @@ -192,11 +192,11 @@ func TestParseState_HappyPaths(t *testing.T) { } } -// TestParseState_MustError is the H1 regression suite: it doesn't just check -// err != nil (a stray comma in a fixture would keep that green forever while -// the actual check regressed) — it asserts the error names the specific -// offending field or value, so a real H1 check going missing fails loudly -// here instead of surviving unnoticed. +// The strict-parsing regression suite: it doesn't just check err != nil (a +// stray comma in a fixture would keep that green forever while the actual +// check regressed) — it asserts the error names the specific offending field +// or value, so a check going missing fails loudly here instead of surviving +// unnoticed. func TestParseState_MustError(t *testing.T) { cases := []struct { fixture string @@ -293,7 +293,7 @@ func TestInstanceKey_NoCollision(t *testing.T) { } } -// minimalStateBody is the smallest H1-legal state response: one group, one +// minimalStateBody is the smallest legal state response: one group, one // rule, no optional keys at all, plus whatever extra is spliced in verbatim // before the rule's closing brace — for isolating one optional key at a time // rather than relying on a fixture that removes several together. @@ -304,10 +304,10 @@ func minimalStateBody(extraRuleJSON string) []byte { `"lastEvaluation":"2026-01-01T00:00:00Z"%s}]}]}}`, extraRuleJSON) } -// §22.2: keepFiringFor is named alongside alerts/totals/labels as an optional -// key (§3.1), but state_missing_optional.json removes it together with -// everything else — never in isolation, so a regression that made it -// required specifically would not be caught by that fixture alone. +// keepFiringFor is optional alongside alerts/totals/labels, but +// state_missing_optional.json removes it together with everything else — never +// in isolation, so a regression that made it required specifically would not +// be caught by that fixture alone. func TestParseState_KeepFiringForIsOptional(t *testing.T) { tests := []struct { name string @@ -329,7 +329,7 @@ func TestParseState_KeepFiringForIsOptional(t *testing.T) { } } -// §22.2: labels is optional at the INSTANCE level (opt(m, "labels", ...) in +// labels is optional at the INSTANCE level (opt(m, "labels", ...) in // parseInstance), distinct from the rule-level labels state_missing_optional.json // already covers — an instance can exist with no labels of its own. func TestParseState_InstanceWithoutLabelsParses(t *testing.T) { @@ -349,9 +349,9 @@ func TestParseState_InstanceWithoutLabelsParses(t *testing.T) { // synthesizeHighCardinalityState builds a state response with a single rule // holding `alerting` Alerting instances and `normal` Normal instances, by // cloning the one real instance in state_one_instance.json. It is never -// committed (§3.2, §22.3, §22.6) — the 2446-instance rule this stands in for -// is ~600 KB and exists only to prove the parser and (in later phases) the -// reducer don't choke on real fleet cardinality. +// committed — the 2446-instance rule this stands in for is ~600 KB and exists +// only to prove the parser and the reducer don't choke on real fleet +// cardinality. func synthesizeHighCardinalityState(t *testing.T, alerting, normal int) []byte { t.Helper() base := readFixture(t, "state_one_instance.json") diff --git a/grafana-alertcheck/internal/gate/resolve.go b/grafana-alertcheck/internal/gate/resolve.go index 3c3710038..329644f95 100644 --- a/grafana-alertcheck/internal/gate/resolve.go +++ b/grafana-alertcheck/internal/gate/resolve.go @@ -7,8 +7,8 @@ import ( "strings" ) -// Resolve turns the operator-supplied alert names into resolved Definitions -// (§17). Order is load-bearing (§17.3): +// Resolve turns the operator-supplied alert names into resolved Definitions. +// Order is load-bearing: // // 1. Trim each name. // 2. Discard empty lines. @@ -18,9 +18,9 @@ import ( // user less than a failure). // // The caller-visible consequence: len(resolved) is the count *after* the -// collapse. A later phase's MinObserved must default from that length, never -// from len(names) — using the input line count would make one rule named -// twice turn an achievable default into an unsatisfiable one (§17.3). +// collapse, and MinObserved must default from that length, never from +// len(names) — using the input line count would make one rule named twice turn +// an achievable default into an unsatisfiable one. func Resolve(defs []Definition, names []string, folder string) (resolved []Definition, notes []string, err error) { seenUID := map[string]string{} // uid -> the first input name that resolved to it for _, raw := range names { @@ -45,15 +45,15 @@ func Resolve(defs []Definition, names []string, folder string) (resolved []Defin return resolved, notes, nil } -// resolveOne resolves a single trimmed, non-empty name against defs (§17.1): -// one match wins outright, zero is an error with suggestions, two or more is -// an error listing every candidate. folder scopes a bare title (no "/" in the -// name) to one folder; it is ignored for the "Folder/Title" and -// "Folder/Group/Title" forms, which already name their own folder. +// resolveOne resolves a single trimmed, non-empty name against defs: one match +// wins outright, zero is an error with suggestions, two or more is an error +// listing every candidate. folder scopes a bare title (no "/" in the name) to +// one folder; it is ignored for the "Folder/Title" and "Folder/Group/Title" +// forms, which already name their own folder. // -// Policy on unsupported kinds (datasource-managed, recording) — decided here -// because §17.1 only says to refuse them, not how they interact with the -// no-match/ambiguous surfaces: a name can still match an unsupported rule (so +// Unsupported kinds (datasource-managed, recording) are refused, and how that +// interacts with the no-match/ambiguous surfaces is decided here: a name can +// still match an unsupported rule (so // naming one by title still gets the specific, named refusal, not a bare "no // match"), but only *supported* candidates count for ambiguity — an // unsupported rule sharing a title with a supported one is resolved silently @@ -73,7 +73,7 @@ func resolveOne(defs []Definition, name, folder string) (Definition, error) { } // uid == "" falls through to the same message as "not found": several // Definition kinds legitimately carry UID == "" (datasource-managed - // rules have no uid at all, P1.3), so matching on an empty suffix + // rules have no uid at all), so matching on an empty suffix // would silently hit one of those and report a misleading // kind-specific refusal for what is really an empty/typo'd uid. This // deliberately does not go through noMatchError: that function's @@ -118,9 +118,9 @@ func resolveOne(defs []Definition, name, folder string) (Definition, error) { } } -// supportedDefs filters out the two kinds §17.1 refuses. Only these -// participate in name-based matching, the no-match rule count, and substring -// suggestions (see the policy note on resolveOne). +// supportedDefs filters out the two refused kinds. Only these participate in +// name-based matching, the no-match rule count, and substring suggestions (see +// the policy note on resolveOne). func supportedDefs(defs []Definition) []Definition { out := make([]Definition, 0, len(defs)) for _, d := range defs { @@ -132,7 +132,7 @@ func supportedDefs(defs []Definition) []Definition { } // classifyForm splits name into the Title | Folder/Title | Folder/Group/Title -// forms (§17). A bare title is scoped by folder when the caller supplied one; +// forms. A bare title is scoped by folder when the caller supplied one; // the two- and three-segment forms already carry their own folder and ignore // it. // @@ -158,8 +158,8 @@ func classifyForm(name, folder string) (wantFolder, wantGroup, wantTitle string, } } -// refuseUnsupportedKind rejects the two kinds §17.1 names explicitly with a -// clear, specific error — distinct from "no match" and from "ambiguous" — so +// refuseUnsupportedKind rejects the two unsupported kinds with a clear, +// specific error — distinct from "no match" and from "ambiguous" — so // an operator who names a recording or datasource-managed rule learns why, // not just that nothing matched. func refuseUnsupportedKind(name string, d Definition) (Definition, error) { @@ -174,8 +174,7 @@ func refuseUnsupportedKind(name string, d Definition) (Definition, error) { } // noMatchError reports a no-match with the count of rules the gate could see -// and, per Context decision 4, case-insensitive substring matches in place of -// the source plan's cut Levenshtein suggestions (§17.2). +// and case-insensitive substring matches as suggestions. func noMatchError(defs []Definition, name, wantTitle string) error { msg := fmt.Sprintf("no rule matched %q (%d rules available; run 'grafana-alertcheck list' to see titles)", name, len(defs)) @@ -194,8 +193,8 @@ func noMatchError(defs []Definition, name, wantTitle string) error { } // ambiguousError lists every candidate with its folder, its group, and the -// full copyable Folder/Group/Title (§17.1) — including the uid: form, which -// resolves unambiguously on the next attempt. +// full copyable Folder/Group/Title — including the uid: form, which resolves +// unambiguously on the next attempt. func ambiguousError(name string, candidates []Definition) error { sorted := append([]Definition(nil), candidates...) sort.Slice(sorted, func(i, j int) bool { return sorted[i].UID < sorted[j].UID }) diff --git a/grafana-alertcheck/internal/gate/resolve_test.go b/grafana-alertcheck/internal/gate/resolve_test.go index 192ed3dc2..d58ec73bd 100644 --- a/grafana-alertcheck/internal/gate/resolve_test.go +++ b/grafana-alertcheck/internal/gate/resolve_test.go @@ -143,8 +143,8 @@ func TestResolve_RejectsEmptySegments(t *testing.T) { } func TestResolve_UIDEmptySuffix(t *testing.T) { - // ruler_datasource_managed.json's only rule has UID == "" (P1.3: this - // shape has no uid at all). "uid:" with an empty suffix must not match it + // ruler_datasource_managed.json's only rule has UID == "" — that shape has + // no uid at all. "uid:" with an empty suffix must not match it // — that would report the misleading "datasource-managed rule, not // supported" for what is really a typo'd/empty uid. defs, err := ParseDefinitions(readFixture(t, "ruler_datasource_managed.json")) @@ -216,8 +216,7 @@ func TestResolve_UnsupportedHomonymResolvesSupportedSilently(t *testing.T) { func TestResolve_CollapseByUIDGivesNoteNotError(t *testing.T) { defs := rulerDefs(t) // The bare title and its Folder/Group/Title spelling both name the same - // rule (rule0000007) — a duplicate-name copy mistake, not an error - // (§17.3). + // rule (rule0000007) — a duplicate-name copy mistake, not an error. resolved, notes, err := Resolve(defs, []string{ "example_workflow_paused_rule", "ExampleObservability/Example Auth Production/example_workflow_paused_rule", @@ -233,9 +232,8 @@ func TestResolve_CollapseByUIDGivesNoteNotError(t *testing.T) { } } -// §22.2: "the same rule with two identical names ... must collapse to one -// rule" — the literal exact-duplicate-string case, distinct from the -// different-spellings case above. +// The same rule named twice with the identical string must collapse to one +// rule — distinct from the different-spellings case above. func TestResolve_IdenticalDuplicateNameCollapsesWithNote(t *testing.T) { defs := rulerDefs(t) resolved, notes, err := Resolve(defs, []string{ @@ -264,7 +262,7 @@ func TestResolve_MinObservedCountIsPostCollapse(t *testing.T) { if err != nil { t.Fatalf("Resolve: unexpected error: %v", err) } - // §17.3: the default MinObserved must come from len(resolved) (2 distinct + // The default MinObserved must come from len(resolved) (2 distinct // rules) — never len(names) (3 input lines), which would be unsatisfiable. if len(resolved) != 2 { t.Fatalf("resolved = %+v, want 2 distinct rules after collapse", resolved) diff --git a/grafana-alertcheck/internal/gate/schedule.go b/grafana-alertcheck/internal/gate/schedule.go index 0417af6a9..4541317e5 100644 --- a/grafana-alertcheck/internal/gate/schedule.go +++ b/grafana-alertcheck/internal/gate/schedule.go @@ -8,35 +8,30 @@ import ( "time" ) -// SkewHardLimit is one of §5's filled-in values (basis: §16; §22.11 asserts -// 120s errors, 30s does not). Defined here, in schedule.go's named-constants -// block, per §5's instruction — it moved out of source.go now that P4 exists; -// P2 needed it before this file did, so it started there. Exported (P10) so -// the CLI can report it verbatim next to a measured skew instead of keeping -// its own mirrored copy. +// SkewHardLimit is the largest clock skew between this runner and Grafana that +// a run tolerates before it errors out. Exported so the CLI can report it +// verbatim next to a measured skew instead of keeping its own mirrored copy. const SkewHardLimit = 60 * time.Second // fromFutureTolerance is how far ahead of the runner's own clock a supplied -// `from` may sit before check refuses it (§7: "from in the future, more than -// the skew tolerance — error"). §7 names no number, so this is the judgment -// call §5's table records: the same 60s as SkewHardLimit, because the only -// legitimate reason for a `from` in the future is clock disagreement between -// the deploy step and the check step, and that is bounded by the same figure. -// It is once-per-run input validation, not a per-rule coverage check, so -// Check applies it (P9) and proveCoverage does not. +// `from` may sit before check refuses it: the same 60s as SkewHardLimit, +// because the only legitimate reason for a `from` in the future is clock +// disagreement between the deploy step and the check step, and that is bounded +// by the same figure. It is once-per-run input validation, not a per-rule +// coverage check, so Check applies it and proveCoverage does not. const fromFutureTolerance = 60 * time.Second -// minDrainTimeout is §5's floor on drainTimeout: max(2 x max(intervalSeconds), -// 2m). Without the floor, a fleet of very tight rules would derive a -// drainTimeout too short to let a healthy in-flight poll land. +// minDrainTimeout is the floor on drainTimeout, which is otherwise +// 2 x max(intervalSeconds). Without the floor, a fleet of very tight rules +// would derive a drainTimeout too short to let a healthy in-flight poll land. const minDrainTimeout = 2 * time.Minute -// graceWarnFraction is §13.2's threshold for warning that transitionGrace eats -// too much of the requested window: "approximately one quarter of the window". +// graceWarnFraction is the share of the requested window above which +// transitionGrace is worth warning about. const graceWarnFraction = 0.25 -// ruleTimings groups the per-rule threshold values §5/§10.1/§14.1 derive from -// a rule's poll cadence and its own evaluation interval. +// ruleTimings groups the per-rule thresholds derived from a rule's poll +// cadence and its own evaluation interval. type ruleTimings struct { pollEvery time.Duration maxGap time.Duration @@ -45,26 +40,24 @@ type ruleTimings struct { } // globalTimings groups the values that apply to the whole run rather than to -// one rule: §13.1's transitionGrace and §19's drainTimeout are each derived -// once, across every non-skipped watched rule, not per rule. +// one rule: transitionGrace and drainTimeout are each derived once, across +// every non-skipped watched rule, not per rule. type globalTimings struct { transitionGrace time.Duration - // graceSource names, and already carries the `for` value of, the rule - // that set transitionGrace (§13.2 requires printing both) — one string - // field rather than a second (rule, duration) pair, matching this - // struct's fixed shape. "none" when no rule contributed (transitionGrace - // is then 0). + // graceSource names, and already carries the `for` value of, the rule that + // set transitionGrace — one string field rather than a second + // (rule, duration) pair, matching this struct's fixed shape. "none" when no + // rule contributed (transitionGrace is then 0). graceSource string drainTimeout time.Duration } // newRuleTimings derives one rule's thresholds from its fully-resolved poll -// cadence and its evaluation interval (§5, §10.1, §14.1). pollEvery arrives -// already resolved for the caller's mode — the §5 default, the operator's -// --poll-interval override, or (in log mode, a later phase) the cadence -// recorded in the log header. Deriving pollEvery inline here, instead of -// accepting it as an input, would let a caller in the wrong mode compute -// maxGap against the wrong authority — see the "Two authorities" note in P5. +// cadence and its evaluation interval. pollEvery arrives already resolved for +// the caller's mode — the default, the operator's --poll-interval override, or +// (in log mode) the cadence recorded in the log header. Deriving pollEvery +// inline here, instead of accepting it as an input, would let a caller in the +// wrong mode compute maxGap against the wrong authority. func newRuleTimings(pollEvery time.Duration, intervalSeconds int) ruleTimings { interval := time.Duration(intervalSeconds) * time.Second maxGap := 2 * pollEvery @@ -77,20 +70,20 @@ func newRuleTimings(pollEvery time.Duration, intervalSeconds int) ruleTimings { } } -// defaultPollEvery is §5's default per-rule cadence: half the rule's own +// defaultPollEvery is the default per-rule cadence: half the rule's own // evaluation interval. func defaultPollEvery(intervalSeconds int) time.Duration { return time.Duration(intervalSeconds) * time.Second / 2 } -// DeriveTimings computes every resolved rule's ruleTimings, keyed by UID, -// plus the shared globalTimings, from resolved definitions and watch's -// optional --poll-interval override (0 = no override: use each rule's §5 -// default of half its own interval). Per §5.1, a supplied override is used -// verbatim for every rule and is never clamped down to the default even when -// it exceeds intervalSeconds/2 — that case is reported back as a note, not -// silently corrected or refused, because clamping would defeat the one knob -// §5.1 gives an operator for making a tight schedule fit. +// DeriveTimings computes every resolved rule's ruleTimings, keyed by UID, plus +// the shared globalTimings, from resolved definitions and watch's optional +// --poll-interval override (0 = no override: use each rule's default of half +// its own interval). A supplied override is used verbatim for every rule and is +// never clamped down to the default even when it exceeds intervalSeconds/2 — +// that case is reported back as a note, not silently corrected or refused, +// because clamping would defeat the one knob an operator has for making a tight +// schedule fit. func DeriveTimings(defs []Definition, override time.Duration) (rules map[string]ruleTimings, global globalTimings, notes []string) { rules = make(map[string]ruleTimings, len(defs)) for _, d := range defs { @@ -124,14 +117,14 @@ func pausedSet(defs []Definition) map[string]bool { return paused } -// DeriveTimingsFromLog is DeriveTimings' log-mode counterpart, and the two -// authorities of P5 are the whole reason it exists as a separate function. -// pollEvery comes from the header — the cadence the recording ACTUALLY used, -// after any --poll-interval override — and maxGap and healthGrace follow from -// it. Re-deriving pollEvery from defs here would compare gaps recorded at the -// override cadence against thresholds computed from the default: exit 2 on a -// clean window when the override was slower, and, worse, a real recorder gap -// passing silently when it was faster. +// DeriveTimingsFromLog is DeriveTimings' log-mode counterpart, and having one +// authority for the cadence is the whole reason it exists as a separate +// function. pollEvery comes from the header — the cadence the recording +// ACTUALLY used, after any --poll-interval override — and maxGap and +// healthGrace follow from it. Re-deriving pollEvery from defs here would +// compare gaps recorded at the override cadence against thresholds computed +// from the default: exit 2 on a clean window when the override was slower, and, +// worse, a real recorder gap passing silently when it was faster. // // evalStaleAfter still comes from defs (2 x intervalSeconds): it is a property // of the rule's own evaluation cadence and is unaffected by how often the gate @@ -151,12 +144,12 @@ func pausedSet(defs []Definition) map[string]bool { // // It checks only the header-to-defs direction. The opposite direction — a // resolved definition absent from the header — is NOT this function's to -// judge: it is §19.1 step 3's log-identity validation, and it belongs to P9's -// Check, which is the only caller that knows both sets and can name the -// mismatch. Without that check a definition simply gets no timings entry, and -// a downstream lookup would read a zero maxGap: fail-closed (every gap -// exceeds it) but silent, so P9 must reject the set mismatch by name rather -// than let a rule fail for an unexplained reason. +// judge: it belongs to Check's log-identity validation, the only caller that +// knows both sets and can name the mismatch. Without that check a definition +// simply gets no timings entry, and a downstream lookup would read a zero +// maxGap: fail-closed (every gap exceeds it) but silent, so Check must reject +// the set mismatch by name rather than let a rule fail for an unexplained +// reason. func DeriveTimingsFromLog(h Header, defs []Definition) (rules map[string]ruleTimings, global globalTimings, err error) { byUID := make(map[string]Definition, len(defs)) for _, d := range defs { @@ -187,15 +180,13 @@ func DeriveTimingsFromLog(h Header, defs []Definition) (rules map[string]ruleTim return rules, deriveGlobalTimings(defs, h.pausedAtStart()), nil } -// deriveGlobalTimings computes transitionGrace and drainTimeout over defs -// (§5, §13.1, §19). +// deriveGlobalTimings computes transitionGrace and drainTimeout over defs. // -// A rule paused before the window opened — skipped, §12 — is excluded from the +// A rule paused before the window opened — skipped — is excluded from the // transitionGrace max: its `for` value can never fire during the window, so // counting it would only inflate the wait past what any watched rule actually -// needs (a judgment call the v2 plan makes explicitly for this formula; §19's -// drainTimeout carries no such exclusion, so it still runs over every resolved -// rule). +// needs. drainTimeout carries no such exclusion and still runs over every +// resolved rule. // // "Before the window opened" is the whole content of that exclusion, so the // authority is pausedAtStart and NEVER Definition.IsPaused: in log mode the @@ -226,9 +217,9 @@ func deriveGlobalTimings(defs []Definition, pausedAtStart map[string]bool) globa return g } -// Scheduler drives one per-rule schedule, never a global cycle (§5): a rule -// at intervalSeconds=10 alongside twenty at 300 keeps its own 5s cadence -// without forcing the same cadence onto the other twenty. +// Scheduler drives one per-rule schedule, never a global cycle: a rule at +// intervalSeconds=10 alongside twenty at 300 keeps its own 5s cadence without +// forcing the same cadence onto the other twenty. type Scheduler struct { next map[string]time.Time every map[string]time.Duration @@ -236,9 +227,9 @@ type Scheduler struct { // NewScheduler builds a Scheduler over per-rule cadences (keyed by UID), // staggering each rule's initial next-due time across [0, pollEvery) so the -// fleet does not start phase-aligned (§5's burst-bound proof depends on this: -// an already-staggered fleet only re-aligns by chance, briefly, not by -// construction). +// fleet does not start phase-aligned. The burst bound CheckBudget enforces +// depends on that: an already-staggered fleet only re-aligns by chance, +// briefly, not by construction. // // It takes cadences rather than whole ruleTimings on purpose: a scheduler // decides when to poll and nothing else, so it must not be handed maxGap, @@ -262,12 +253,12 @@ func NewScheduler(every map[string]time.Duration, now time.Time) *Scheduler { } // Due returns the UIDs whose next-due time has arrived, earliest-due-first. -// Ties (equal next-due time) break by tightest cadence first: the burst-bound -// proof in §5 assumes a newly-due tight rule waits at most for one in-flight -// request, which only holds if a simultaneous batch serves the tightest rule -// ahead of slacker ones. A tie-break that instead followed map iteration -// order would silently void that proof — nothing else would fail until a -// phase-aligned fleet opened a mid-run gap in production. +// Ties (equal next-due time) break by tightest cadence first: the burst bound +// assumes a newly-due tight rule waits at most for one in-flight request, which +// only holds if a simultaneous batch serves the tightest rule ahead of slacker +// ones. A tie-break that instead followed map iteration order would silently +// void that assumption — nothing else would fail until a phase-aligned fleet +// opened a mid-run gap in production. func (s *Scheduler) Due(now time.Time) []string { var due []string for uid, t := range s.next { @@ -303,11 +294,10 @@ func (s *Scheduler) Mark(uid string, now time.Time) error { } // earliestDue returns the earliest scheduled next-due time, and false when the -// scheduler holds no rules at all. The recorder's loop (P6) waits exactly that -// long instead of waking on a fixed tick: a fixed tick either polls a slack -// rule early — spending request budget the §5 formulas already accounted for — -// or wakes too late for the tightest rule and opens a gap inside its own -// maxGap. +// scheduler holds no rules at all. The recorder's loop waits exactly that long +// instead of waking on a fixed tick: a fixed tick either polls a slack rule +// early — spending request budget the schedule already accounted for — or wakes +// too late for the tightest rule and opens a gap inside its own maxGap. func (s *Scheduler) earliestDue() (time.Time, bool) { var earliest time.Time ok := false @@ -320,23 +310,21 @@ func (s *Scheduler) earliestDue() (time.Time, bool) { return earliest, ok } -// CheckBudget applies §5's error-at-start check to a fully resolved schedule. -// t and measured are both keyed by rule UID; measured must carry every UID in -// t; a rule this run never measured can't have its budget proved, and a -// silent zero-duration default would be exactly the kind of pass-on-an- -// unproven-window bug §5 exists to catch. CheckBudget fails when any of three -// conditions holds (sanity-checked against §22.3's mixed-interval regression -// in this phase's tests): +// CheckBudget proves at start that a fully resolved schedule can actually be +// served. t and measured are both keyed by rule UID, and measured must carry +// every UID in t: a rule this run never measured cannot have its budget +// proved, and a silent zero-duration default would be exactly the kind of +// pass-on-an-unproven-window bug this check exists to catch. It fails when any +// of three conditions holds: // // - utilization: the long-run request rate exceeds what concurrency serves; // - a single rule's own request cannot fit inside its own cadence; -// - the burst bound: the slowest measured request is slower than the -// fleet's tightest cadence, which — even under earliest-due-first -// ordering — can open a mid-run gap bigger than that rule's maxGap. +// - the burst bound: the slowest measured request is slower than the fleet's +// tightest cadence, which — even under earliest-due-first ordering — can +// open a mid-run gap bigger than that rule's maxGap. // -// The message never suggests a single interval (§5.1) — only the three -// controls an operator actually has: concurrency, poll-interval, and the -// alert list. +// The message names only the three controls an operator actually has: +// concurrency, poll-interval, and the alert list. func CheckBudget(t map[string]ruleTimings, measured map[string]time.Duration, concurrency int) error { if len(t) == 0 { return nil @@ -406,10 +394,11 @@ func CheckBudget(t map[string]ruleTimings, measured map[string]time.Duration, co return fmt.Errorf("%s", b.String()) } -// StartupSummary formats §13.2's required pre-run print: the total planned -// run time and the rule (with its `for` value) that set transitionGrace, plus -// a warning when the grace eats more than graceWarnFraction of the requested -// window. from/to are the requested classification window. +// StartupSummary formats the pre-run print an operator sees before the wait: +// the total planned run time and the rule (with its `for` value) that set +// transitionGrace, plus a warning when the grace eats more than +// graceWarnFraction of the requested window. from/to are the requested +// classification window. func StartupSummary(from, to time.Time, global globalTimings) (summary, warning string) { window := to.Sub(from) total := window + global.transitionGrace + global.drainTimeout diff --git a/grafana-alertcheck/internal/gate/schedule_test.go b/grafana-alertcheck/internal/gate/schedule_test.go index aee22f37f..70cc8817c 100644 --- a/grafana-alertcheck/internal/gate/schedule_test.go +++ b/grafana-alertcheck/internal/gate/schedule_test.go @@ -69,7 +69,7 @@ func TestDeriveTimings_TransitionGraceExcludesSkippedRule(t *testing.T) { } } -// §22.2/§22.3: `for: 1d` and `for: 1w` parse correctly (parse_ruler_test.go), +// `for: 1d` and `for: 1w` parse correctly (parse_ruler_test.go), // but that alone never proves they flow into transitionGrace — a Prometheus // duration parser that silently truncated to time.Duration's other units, or // a transitionGrace derivation that only ever saw hand-built values, could @@ -148,7 +148,7 @@ func TestDeriveTimingsFromLog_TransitionGraceFollowsTheHeaderNotTheDefinition(t }) t.Run("drainTimeout counts every rule either way", func(t *testing.T) { - // §19 puts no pause exclusion on drainTimeout, so both headers give the + // drainTimeout carries no pause exclusion, so both headers give the // same floor-bound value. for _, pausedAtStart := range []bool{false, true} { h := Header{Rules: []LoggedRule{loggedRule("r1", pausedAtStart)}} @@ -192,9 +192,8 @@ func TestDeriveTimings_DrainTimeoutAboveFloor(t *testing.T) { } } -// TestScheduler_DueOrderingTiesBreakByTightestCadence pins the ordering -// invariant the burst bound depends on (§5): when several rules become due at -// the exact same instant, Due must serve the tightest cadence first, not +// The ordering invariant the burst bound depends on: when several rules become +// due at the exact same instant, Due must serve the tightest cadence first, not // whatever order the underlying map happens to iterate in. A refactor that // loses this ordering must fail here, not in a production phase-aligned gap. func TestScheduler_DueOrderingTiesBreakByTightestCadence(t *testing.T) { @@ -261,8 +260,8 @@ func TestScheduler_MarkUnknownUIDFails(t *testing.T) { } // TestScheduler_PerRuleCadenceOverTime simulates a run and counts how often -// each rule comes due, pinning §5's core claim: schedules are per rule, never -// a global cycle. A tight rule must be polled at its own cadence regardless +// each rule comes due: schedules are per rule, never a global cycle. A tight +// rule must be polled at its own cadence regardless // of what slower rules in the same fleet need, and a slack rule must never be // forced onto the tight rule's cadence. func TestScheduler_PerRuleCadenceOverTime(t *testing.T) { @@ -350,10 +349,8 @@ func TestScheduler_EarliestDuePicksMinimum(t *testing.T) { } } -// TestCheckBudget_MixedIntervalRegression is §22.3's sanity check from the -// plan: one rule at 10s beside twenty at 300s, all measured ~1.8s, must not -// error at any reasonable concurrency — the exact case a naive worst-case-slot -// simulation would wrongly fail. +// One rule at 10s beside twenty at 300s, all measured ~1.8s, must not error at +// any reasonable concurrency — the exact case a naive worst-case-slot func TestCheckBudget_MixedIntervalRegression(t *testing.T) { timings := map[string]ruleTimings{"tight": {pollEvery: 5 * time.Second}} measured := map[string]time.Duration{"tight": 1800 * time.Millisecond} @@ -459,9 +456,9 @@ func TestCheckBudget_NonPositivePollIntervalIsAnError(t *testing.T) { } } -// assertBudgetMessage checks §5.1's required message contents: a measured -// duration is present, and all three controls are named — never a single -// suggested interval. +// assertBudgetMessage checks the message contents: a measured duration is +// present, and all three controls are named — never a single suggested +// interval. func assertBudgetMessage(t *testing.T, msg string) { t.Helper() for _, want := range []string{"measured", "concurrency", "poll-interval", "fewer"} { @@ -491,13 +488,10 @@ func TestStartupSummary_WarningWhenGraceTooLarge(t *testing.T) { } } -// §22.3: "a rule with for: 15m in a 10-minute window gives the warning about -// a large grace period" — no such rule exists in the real capture -// (testdata/README.md), so the test above pins the mechanism with a -// hand-built globalTimings. This drives the same warning off the real -// ruler_rules.json fixture's for:1w rule instead, tying ParseDefinitions and -// DeriveTimings into the warning end to end, not just the warning formula in -// isolation. +// The test above pins the warning formula with a hand-built globalTimings. +// This drives the same warning off the real ruler_rules.json fixture's for:1w +// rule instead, tying ParseDefinitions and DeriveTimings into the warning end +// to end. func TestStartupSummary_RealForOneWeekRuleTriggersWarning(t *testing.T) { defs := rulerDefs(t) _, global, notes := DeriveTimings(defs, 0) diff --git a/grafana-alertcheck/internal/gate/source.go b/grafana-alertcheck/internal/gate/source.go index d440d4938..82dcdc392 100644 --- a/grafana-alertcheck/internal/gate/source.go +++ b/grafana-alertcheck/internal/gate/source.go @@ -22,8 +22,8 @@ import ( // across retries. const maxResponseBytes = 25 << 20 // 25 MiB -// Clock is the seam that lets tests advance time without sleeping (§22) — the -// only two operations the gate ever needs from a clock. +// Clock is the seam that lets tests advance time without sleeping — the only +// two operations the gate ever needs from a clock. type Clock interface { Now() time.Time After(d time.Duration) <-chan time.Time @@ -37,9 +37,9 @@ func (SystemClock) After(d time.Duration) <-chan time.Time { return time.After(d // Observation is one successful poll of the state endpoint for a single rule. type Observation struct { - Rules []StateRule // may be empty — an authoritative 2xx saying the rule is absent (§14.5) - GrafanaNow time.Time // the Date header — H4 - Skew time.Duration // serverDate - (t_send+t_headers)/2, signed (§16) + Rules []StateRule // may be empty — an authoritative 2xx saying the rule is absent + GrafanaNow time.Time // the response's Date header + Skew time.Duration // serverDate - (t_send+t_headers)/2, signed SkewBound time.Duration // (t_headers-t_send)/2 — RTT/2 to the response headers Latency time.Duration // t_send through the full body read — see requestResult.Latency } @@ -48,7 +48,7 @@ type Observation struct { // network failure, or a body that failed to parse. It is never a deleted rule // (an authoritative 2xx with no matching rule is not this) and never a clock // problem (a missing/unparseable Date header or an out-of-bounds skew is a -// hard error instead — see doRequest). Never conflate them (§14.5). +// hard error instead — see doRequest). Never conflate them. type TransportError struct { Err error Status int // 0 when the failure never got a status (network/transport failure) @@ -67,10 +67,10 @@ func (e *TransportError) Unwrap() error { return e.Err } // too many sequential *TransportError failures. It deliberately does not // implement Unwrap into the underlying *TransportError: once retries are // exhausted the result is a hard, terminal failure, and -// errors.AsType[*TransportError] must never re-classify it as retryable — -// that is the exact conflation §19.3 case 1 forbids. Cause is still exposed -// as a plain field (and folded into Error()'s text) so a caller can log or -// inspect it; it just cannot flow back into the retry classification. +// errors.AsType[*TransportError] must never re-classify it as retryable. +// Cause is still exposed as a plain field (and folded into Error()'s text) so +// a caller can log or inspect it; it just cannot flow back into the retry +// classification. type RetryExhaustedError struct { Failures int Cause error @@ -81,7 +81,7 @@ func (e *RetryExhaustedError) Error() string { } // Source is everything the gate reads from Grafana. httpSource is the one -// production implementation; every later phase's tests use a scripted fake +// production implementation; the tests use a scripted fake // (source_fake_test.go) instead of real HTTP. type Source interface { Version(ctx context.Context) (string, error) @@ -137,7 +137,7 @@ func parseGrafanaVersion(s string) (grafanaVersion, error) { } // supportedGrafanaMin and supportedGrafanaMax bound the platform this gate is -// verified against (§2.7 control 2, §21.5): >= 13.0.0, < 14.0.0. +// verified against: >= 13.0.0, < 14.0.0. var ( supportedGrafanaMin = grafanaVersion{13, 0, 0} supportedGrafanaMax = grafanaVersion{14, 0, 0} // exclusive @@ -145,8 +145,9 @@ var ( // CheckGrafanaVersion enforces the supported range. An unparseable or // out-of-range version is a hard error naming both what was found and what is -// supported — trusting an unverified schema is exactly the deprecation risk -// §2.7 control 2 exists to catch. +// supported: the response schemas this gate parses are only verified against +// that range, and trusting an unverified one is how a deprecation turns into a +// silent misread. func CheckGrafanaVersion(version string) error { v, err := parseGrafanaVersion(version) if err != nil { @@ -160,12 +161,12 @@ func CheckGrafanaVersion(version string) error { return nil } -// httpSource is the production Source: stdlib net/http only, bearer auth -// from a token supplied at construction (the caller reads it from the -// environment — §20.2 — this type never touches env itself), and manual -// strict decoding via ParseState/ParseDefinitions (H1). The retry limit and -// backoff parameters are struct fields with production defaults set here, -// not package constants, so a test can shrink them without a hook. +// httpSource is the production Source: stdlib net/http only, bearer auth from +// a token supplied at construction (the caller reads it from the environment; +// this type never touches env itself), and manual strict decoding via +// ParseState/ParseDefinitions. The retry limit and backoff parameters are +// struct fields with production defaults set here, not package constants, so a +// test can shrink them without a hook. type httpSource struct { baseURL string token string @@ -177,9 +178,8 @@ type httpSource struct { backoffCap time.Duration } -// NewHTTPSource builds the production Source. token is never logged and -// never enters an error string (§20.2) — it is used only to set the -// Authorization header. +// NewHTTPSource builds the production Source. token is never logged and never +// enters an error string — it is used only to set the Authorization header. func NewHTTPSource(baseURL, token string, clock Clock) Source { return &httpSource{ baseURL: strings.TrimSuffix(baseURL, "/"), @@ -236,8 +236,8 @@ func (s *httpSource) RuleState(ctx context.Context, title string) (Observation, if parseErr != nil { // Treated as transient, not a schema break: an unparseable 2xx // is far more likely a mid-stream hiccup than a permanent shape - // change, and H1's strict parser already turns a real shape - // change into a loud per-field error the moment it's visible. + // change, and the strict parser already turns a real shape change + // into a loud per-field error the moment it's visible. return Observation{}, &TransportError{Err: fmt.Errorf("parse rule state: %w", parseErr)} } return Observation{ @@ -252,33 +252,32 @@ func (s *httpSource) RuleState(ctx context.Context, title string) (Observation, // requestResult is the outcome of one successful HTTP attempt in doRequest: // the raw body plus everything derived from timing the round trip against -// the response's own clock (§16). +// the response's own clock. type requestResult struct { Body []byte - ServerDate time.Time // the Date header — H4 + ServerDate time.Time // the response's Date header Skew time.Duration // serverDate - (t_send+t_headers)/2, signed SkewBound time.Duration // (t_headers-t_send)/2 — RTT/2 to the response headers - // Latency spans t_send through the full body read (§5.2's budget check - // needs the whole poll's wall time, or a schedule feasibility check that - // only sees header latency goes optimistic — fail-open). It does not - // include the caller's subsequent JSON parse (ParseState/ParseDefinitions - // run outside doRequest); if P4's budget accounting needs parse time - // folded in too, extend here rather than approximating it at the call - // site. + // Latency spans t_send through the full body read: the budget check needs + // the whole poll's wall time, or a schedule feasibility check that only + // sees header latency goes optimistic — fail-open. It does not include the + // caller's subsequent JSON parse (ParseState/ParseDefinitions run outside + // doRequest); if the budget accounting ever needs parse time folded in too, + // extend here rather than approximating it at the call site. Latency time.Duration } -// doRequest performs one HTTP GET and classifies the outcome (§14.5, §16): -// a network failure, a non-2xx status, or a body-read failure is retryable +// doRequest performs one HTTP GET and classifies the outcome: a network +// failure, a non-2xx status, or a body-read failure is retryable // (*TransportError); a missing or unparseable Date header, or a skew beyond // SkewHardLimit, is a hard error — retrying can never fix either, so neither -// may enter the backoff loop (H4). +// may enter the backoff loop. // // The Date-header/skew check runs for every endpoint this hits, including -// /api/health — broader than §16's own scope, which only discusses the state -// endpoint. Deliberate: a skewed clock discovered only once RuleState starts -// polling is a skew that has already masked whatever /api/health and the -// ruler read reported; failing closed at the first response catches it +// /api/health, and not only the state endpoint whose timestamps the gate +// actually compares. Deliberate: a skewed clock discovered only once RuleState +// starts polling is a skew that has already masked whatever /api/health and +// the ruler read reported; failing closed at the first response catches it // before any of that is trusted, and every response comes with a Date header // for free. func (s *httpSource) doRequest(ctx context.Context, path string) (requestResult, error) { @@ -317,7 +316,7 @@ func (s *httpSource) doRequest(ctx context.Context, path string) (requestResult, dateHeader := resp.Header.Get("Date") if dateHeader == "" { - return requestResult{}, fmt.Errorf("%s: response has no Date header (H4)", path) + return requestResult{}, fmt.Errorf("%s: response has no Date header", path) } serverDate, parseErr := http.ParseTime(dateHeader) if parseErr != nil { @@ -332,7 +331,7 @@ func (s *httpSource) doRequest(ctx context.Context, path string) (requestResult, absSkew = -absSkew } if absSkew > SkewHardLimit { - return requestResult{}, fmt.Errorf("%s: clock skew %s exceeds hard limit %s (§16)", path, absSkew, SkewHardLimit) + return requestResult{}, fmt.Errorf("%s: clock skew %s exceeds hard limit %s", path, absSkew, SkewHardLimit) } return requestResult{Body: b, ServerDate: serverDate, Skew: signedSkew, SkewBound: bound, Latency: latency}, nil @@ -341,9 +340,9 @@ func (s *httpSource) doRequest(ctx context.Context, path string) (requestResult, // retryTransport runs fn, retrying with backoff only while it fails with a // *TransportError — any other error is a hard error and returns immediately, // never retried. failures counts consecutive *TransportError results; -// exceeding maxFailures gives up with a wrapped hard error (§19.3 case 1). -// The wait between attempts goes through clock.After so a test with a fake -// Clock never sleeps on real time (§22). +// exceeding maxFailures gives up with a wrapped hard error. The wait between +// attempts goes through clock.After so a test with a fake Clock never sleeps on +// real time. func retryTransport[T any](ctx context.Context, clock Clock, maxFailures int, backoffBase, backoffCap time.Duration, fn func() (T, error)) (T, error) { var zero T failures := 0 @@ -368,7 +367,7 @@ func retryTransport[T any](ctx context.Context, clock Clock, maxFailures int, ba } // backoffDelay is 1s base, doubling per failure, capped at maxDelay, with -// ±20% jitter (§5's filled-in value for maxSequentialFailures). +// ±20% jitter. func backoffDelay(base, maxDelay time.Duration, failureCount int) time.Duration { d := base for i := 1; i < failureCount && d < maxDelay; i++ { diff --git a/grafana-alertcheck/internal/gate/source_fake_test.go b/grafana-alertcheck/internal/gate/source_fake_test.go index 40a2a8a05..97de92f02 100644 --- a/grafana-alertcheck/internal/gate/source_fake_test.go +++ b/grafana-alertcheck/internal/gate/source_fake_test.go @@ -9,15 +9,13 @@ import ( ) // fakeClock is a manually-advanced Clock — no test in this package sleeps on -// real time (§22). It is goroutine-safe (a concurrent fleet under -race must -// not trip on the double itself), but After always fires immediately, -// regardless of the requested duration or whether Advance was ever called. -// That is sufficient here: every retry/backoff test in this phase only needs -// to avoid a real sleep. It is NOT sufficient for a test that must prove a -// wait did not fire early — e.g. a P4 scheduler test asserting Due() doesn't -// return a rule before its next-due time. Use virtualClock below for that: it -// is the clock P6's recorder-loop tests needed, and it makes a wait and the -// passage of time the same event. +// real time. It is goroutine-safe (a concurrent fleet under -race must not trip +// on the double itself), but After always fires immediately, regardless of the +// requested duration or whether Advance was ever called. That is enough for the +// retry/backoff tests, which only need to avoid a real sleep. It is NOT enough +// for a test that must prove a wait did not fire early — e.g. asserting Due() +// does not return a rule before its next-due time. Use virtualClock below for +// that: it makes a wait and the passage of time the same event. type fakeClock struct { mu sync.Mutex now time.Time @@ -113,12 +111,11 @@ type scriptedObservation struct { err error } -// fakeSource is a scripted Source with no HTTP, goroutine-safe so a phase -// that polls several rules concurrently (P6) can share one instance across -// goroutines without tripping -race. P3 through at least P5 can construct -// one directly instead of talking to HTTP; a phase that needs it to behave -// like a live server under concurrent load beyond simple locking should -// verify that assumption rather than take this comment's word for it. +// fakeSource is a scripted Source with no HTTP, goroutine-safe so a test that +// polls several rules concurrently can share one instance across goroutines +// without tripping -race. A test that needs it to behave like a live server +// under concurrent load beyond simple locking should verify that assumption +// rather than take this comment's word for it. type fakeSource struct { mu sync.Mutex diff --git a/grafana-alertcheck/internal/gate/source_test.go b/grafana-alertcheck/internal/gate/source_test.go index 2998b4776..39181cfca 100644 --- a/grafana-alertcheck/internal/gate/source_test.go +++ b/grafana-alertcheck/internal/gate/source_test.go @@ -90,7 +90,7 @@ func TestCheckGrafanaVersion(t *testing.T) { } for _, want := range c.wantContains { if !strings.Contains(err.Error(), want) { - t.Errorf("CheckGrafanaVersion(%q): error %q does not mention %q (the plan requires naming both what was found and what is supported)", c.version, err.Error(), want) + t.Errorf("CheckGrafanaVersion(%q): error %q does not mention %q — it must name both what was found and what is supported", c.version, err.Error(), want) } } } @@ -170,7 +170,7 @@ func TestHTTPSource_RuleState_EmptyIsNotAnError(t *testing.T) { t.Fatalf("Rules = %+v, want empty (an authoritative 2xx is not a transport error)", obs.Rules) } if obs.GrafanaNow.IsZero() { - t.Fatalf("GrafanaNow is zero, want the response's Date header value (H4)") + t.Fatalf("GrafanaNow is zero, want the response's Date header value") } } @@ -274,7 +274,7 @@ func TestHTTPSource_MissingDateHeader(t *testing.T) { src := NewHTTPSource(srv.URL, "", clock) _, err := src.Version(context.Background()) if err == nil { - t.Fatalf("Version(): want error, got nil (H4: a missing Date header is a hard error)") + t.Fatalf("Version(): want error, got nil: a missing Date header is a hard error") } if calls.Load() != 1 { t.Fatalf("calls = %d, want 1 — a missing Date header must never be retried", calls.Load()) @@ -294,7 +294,7 @@ func TestHTTPSource_UnparseableDateHeader(t *testing.T) { src := NewHTTPSource(srv.URL, "", clock) _, err := src.Version(context.Background()) if err == nil { - t.Fatalf("Version(): want error, got nil (H4: an unparseable Date header is a hard error)") + t.Fatalf("Version(): want error, got nil: an unparseable Date header is a hard error") } if calls.Load() != 1 { t.Fatalf("calls = %d, want 1 — an unparseable Date header must never be retried", calls.Load()) @@ -343,14 +343,14 @@ func TestHTTPSource_ObservationTiming(t *testing.T) { t.Errorf("SkewBound = %v, want 1s (RTT/2 with a 2s round trip to headers)", obs.SkewBound) } if obs.Latency != 4*time.Second { - t.Errorf("Latency = %v, want 4s (send through full body read, §5.2) — not just the 2s header round trip", obs.Latency) + t.Errorf("Latency = %v, want 4s (send through full body read) — not just the 2s header round trip", obs.Latency) } }) } } -// §22.7/§16: a genuinely discriminating regression for "the gate compares -// staleness against the Date header, never the runner's clock." lastEvaluation +// A discriminating regression for "the gate compares staleness against the +// Date header, never the runner's clock". lastEvaluation // sits 100s behind Grafana's TRUE now (obs.GrafanaNow, from the Date header) // — under the 120s evalStaleAfter limit — but 130s behind the RUNNER's clock. // An implementation that leaked the runner's clock into the staleness @@ -564,8 +564,7 @@ func TestHTTPSource_NetworkFailureRetries(t *testing.T) { // it names how many failures it gave up after, and — the regression this // pins — it is never itself classified as a *TransportError. If it were, // something one layer up that also retries on *TransportError would treat an -// already-exhausted give-up as retryable again, the exact conflation §19.3 -// case 1 forbids. +// already-exhausted give-up as retryable again. func assertRetryExhausted(t *testing.T, err error, wantFailures int) { t.Helper() var reErr *RetryExhaustedError diff --git a/grafana-alertcheck/internal/gate/testdata/README.md b/grafana-alertcheck/internal/gate/testdata/README.md index fb437b3ae..1a6bf5877 100644 --- a/grafana-alertcheck/internal/gate/testdata/README.md +++ b/grafana-alertcheck/internal/gate/testdata/README.md @@ -1,6 +1,6 @@ # Fixture provenance -All fixtures are sanitized slices of the real Grafana 13.1.0 payloads captured next to the plan in +All fixtures are sanitized slices of real Grafana 13.1.0 payloads captured into `tmp/` (`tmp/state_all.json`, `tmp/ruler_all.json`, `tmp/health.json` — gitignored, never committed). Renames are consistent across files: the same real folder/rule keeps the same fake identity everywhere it appears (e.g. `folder0000002`/`rule0000002` is the same real paused rule in both @@ -23,45 +23,38 @@ here instead, for every fixture, for consistency. `rule0000002`/"Example Paused Rule". Unmodified: `isPaused:true`, zero `lastEvaluation`, `health:ok`, `state:inactive`, absent `alerts`/`labels`. - **state_health_error.json** — real `health:error` rule ("[JD] No Job Proposals", folder - `job-distributor`), highest priority per §22.1. Renamed to folder `ExampleService`/`folder0000003`, + `job-distributor`). Renamed to folder `ExampleService`/`folder0000003`, rule `rule0000003`/"Example No Data Source". Unmodified: `health:error`, `lastError` text, the single `Error` instance. - **state_health_nodata.json** — real `health:nodata` rule ("ARE test", folder `diegos_playground`). Renamed to folder `ExamplePlayground`/`folder0000004`, rule `rule0000004`/"Example NoData Rule". Unmodified: `health:nodata`, the single `NoData` instance. -- **state_reason_composite.json** — composite of two real instances combined under one rule for P1.2a - coverage: a real `"Normal (Error)"` instance (from a Flux-reconciliation rule; 14 of that state exist +- **state_reason_composite.json** — two real instances combined under one rule to cover composite + state parsing: a real `"Normal (Error)"` instance (from a Flux-reconciliation rule; 14 of that state exist in the capture) and a real `"Normal (NoData)"` instance (from a pod-liveness rule; 1091 of that state exist), plus one plain `"Normal"` instance for contrast. Renamed to folder `ExampleInfra`/`folder0000005`, rule `rule0000005`/"Example Composite Reasons". - **state_missing_optional.json** — derived from `state_one_instance.json`: `alerts`, `totals`, `totalsFiltered` and `labels` all removed. Must parse with `Instances=nil`, `Totals=nil`. -- **state_missing_health.json** — derived from `state_one_instance.json`: the required `health` key - removed. Must be a parse error (H1). -- **state_missing_lasteval.json** — derived from `state_one_instance.json`: the required - `lastEvaluation` key removed. Must be a parse error (H1). -- **state_missing_state.json** — derived from `state_one_instance.json`: the required rule-level - `state` key removed. Must be a parse error (H1). Closes must-error coverage for H1's four required - fields — a review pass found `health`/`lastEvaluation` covered but `state`/`interval` weren't, even - though the code already `req`'d them correctly. -- **state_missing_interval.json** — derived from `state_one_instance.json`: the required group-level - `interval` key removed. Must be a parse error (H1); same review-pass gap as above. +- **state_missing_health.json**, **state_missing_lasteval.json**, **state_missing_state.json**, + **state_missing_interval.json** — derived from `state_one_instance.json`, each with one of the four + required keys removed (`health`, `lastEvaluation`, rule-level `state`, group-level `interval`). Each + must be a parse error. - **state_missing_file.json** / **state_missing_name.json** — derived from `state_one_instance.json`: - the group-level `file`/`name` keys removed respectively. Not part of H1's four (those are `health`, - `state`, `lastEvaluation`, `interval`), but the code treats group identity as strict too, and the same - review pass flagged the gap — closed rather than deferred to a later §22 sweep since the fixture is - the same 10-line edit. + the group-level `file`/`name` keys removed respectively. Not among the four required fields above, + but the parser treats group identity as strict too. - **state_zerotime_unpaused.json** — derived from `state_one_instance.json`: `lastEvaluation` set to - the zero time while `isPaused` stays `false`. Must be a parse error (§2.3). + the zero time while `isPaused` stays `false`. Must be a parse error — only a paused rule may report + the zero time. - **state_unknown_state.json** — derived from `state_one_instance.json`: the instance state hand-edited to `"Weird (NoData)"`, a syntactically valid composite whose base isn't in the 5-value allowlist. Must - be a parse error (P1.2a). + be a parse error. - **state_only_active_instances.json** — derived from a real rule that genuinely had 1 `Alerting` + 22 `Normal` instances (`totals: {alerting:1, normal:22}`, rule `dfhp1t5pkosu8f`, folder `BCM`). `alerts[]` - trimmed to the single `Alerting` instance only, while `totals` is left **unchanged** — reproducing the - §3.2 violation shape (instance list says "only active" while totals disagrees). Renamed to folder - `ExampleTeam`/`folder0000001`, rule `rule0000006`. `ParseState` itself parses this fine; the §3.2 - verification lives in a later phase (P5/P9). + trimmed to the single `Alerting` instance only, while `totals` is left **unchanged** — the shape a + state endpoint that stopped returning normal instances would produce (the instance list says "only + active" while totals disagrees). Renamed to folder `ExampleTeam`/`folder0000001`, rule `rule0000006`. + `ParseState` itself parses this fine; `VerifyNormalInstancesVisible` is what rejects it. ## Ruler endpoint (`/api/ruler/grafana/api/v1/rules`) @@ -69,7 +62,7 @@ here instead, for every fixture, for consistency. - The real true 2-way title collision: namespace `CRE-BCM-Prod-Zone-A`, group `Gateway`, identical folder+group+title, distinct UIDs (`ffvabtvvbozcwf`/`efvabtwbxlvk0b`) — renamed to namespace `Example-Zone-A`, rules `rule0000006a`/`rule0000006b`, both titled "Example No Gateways Available". - Folder/Group/Title alone does **not** disambiguate this pair (§17, §22.2). + Folder/Group/Title alone does **not** disambiguate this pair; only `uid:` does. - The 3 real `is_paused:true` rules, renamed to `rule0000002`/`rule0000007`/`rule0000008`. `rule0000002` intentionally shares its identity (`folder0000002`) with `state_paused.json`. - A real `for:1d` rule (`afs438kjd4v7kd` → `rule0000009`). @@ -81,7 +74,7 @@ here instead, for every fixture, for consistency. Grafana represents a datasource-managed (native Prometheus-format) alerting rule. `ParseDefinitions` must classify it as `KindDatasourceManaged`, parse `Title` from `alert`, and leave `UID` empty (this shape has no uid at all — inventing one would be inventing shape) without rejecting the rule - (rejection is P3's job, only for rules a user actually named). + (rejection is `Resolve`'s job, and only for rules a user actually named). - **ruler_recording.json** — **DERIVED**, no recording rule exists in the capture (verified: 0 rules carry `grafana_alert.record`). Hand-built: a `grafana_alert` block with a `record` sub-object but deliberately *without* `no_data_state`/`exec_err_state`/`is_paused`/`intervalSeconds`/`namespace_uid` diff --git a/grafana-alertcheck/internal/gate/watch.go b/grafana-alertcheck/internal/gate/watch.go index f8726fec6..6f4b7fa15 100644 --- a/grafana-alertcheck/internal/gate/watch.go +++ b/grafana-alertcheck/internal/gate/watch.go @@ -14,7 +14,7 @@ import ( ) // DaemonChildFlag is the hidden flag the parent passes when it re-execs itself -// as the detached recorder (§4.4). It is deliberately absent from the CLI's +// as the detached recorder. It is deliberately absent from the CLI's // usage text: an operator never types it, and a child started by hand against // a log no parent prepared fails immediately on the header read. const DaemonChildFlag = "--daemon-child" @@ -26,7 +26,7 @@ const DaemonChildFlag = "--daemon-child" // loop — and not an assumption drawn from surviving a timer. A timer cannot // tell a healthy child from one that is about to die on a slow runner, and // getting that wrong means watch returns success over a recording that never -// happened (§4.3). +// happened. const ReadyFDFlag = "--ready-fd" // childReadyTimeout bounds that wait. Everything before the signal is local — @@ -43,7 +43,7 @@ const daemonLogTailBytes = 4096 // // It has no To field and must never gain one: watch writes the stopped // sentinel with its OWN stop time and makes no comparison against `to`, which -// only check knows (§4.5). Passing `to` here would give two components an +// only check knows. Passing `to` here would give two components an // opinion about the same comparison, and the recorder's opinion is the one // that cannot be trusted — it exits before the grace it would have to wait for. // @@ -55,18 +55,18 @@ const daemonLogTailBytes = 4096 // Header carries no States field for the same reason. type WatchConfig struct { // URL and Token are the connection details. The CLI reads both from the - // environment and never from a flag (§20.2); Token is never logged and - // never enters an error string. + // environment and never from a flag; Token is never logged and never + // enters an error string. URL, Token string - // Alerts are the operator-supplied names, one per line, in any of §17's - // forms. Empty lines are discarded by Resolve. + // Alerts are the operator-supplied names, one per line, in any of the forms + // Resolve accepts. Empty lines are discarded by Resolve. Alerts []string Folder string // Out is the JSONL log path. PidFile and DaemonLog default to // .pid and .daemon.log — the same convention check uses to find - // the recorder it must stop (P9), so nothing has to be wired by hand. + // the recorder it must stop, so nothing has to be wired by hand. Out string PidFile string DaemonLog string @@ -77,11 +77,10 @@ type WatchConfig struct { Until time.Time // PollEvery is the --poll-interval override, used verbatim for every rule - // and never clamped (§5.1). Zero means each rule polls at half its own - // evaluation interval. Whatever this resolves to is written into the header - // as the cadence actually used, and that header value — never a - // re-derivation from the definitions — is what check derives maxGap from - // (P5, "two authorities"). + // and never clamped. Zero means each rule polls at half its own evaluation + // interval. Whatever this resolves to is written into the header as the + // cadence actually used, and that header value — never a re-derivation from + // the definitions — is what check derives maxGap from. PollEvery time.Duration Concurrency int @@ -90,7 +89,7 @@ type WatchConfig struct { // Notes is where the parent prints what an operator has to see before the // deploy step runs: resolve notes, the cadence per rule, the rules it will // not wait for. nil discards them. The library prints nothing else — the - // CLI owns presentation (§20.2). + // CLI owns presentation. Notes io.Writer } @@ -138,20 +137,20 @@ func (cfg WatchConfig) validate() error { return nil } -// Watch is the record step's parent process (§4.3). It returns only once the -// window is genuinely being recorded: +// Watch is the record step's parent process. It returns only once the window +// is genuinely being recorded: // // version gate -> resolve definitions and names -> derive timings -> // open the log and write the header -> ONE observation of every non-skipped -// rule -> verify §3.2 -> check the schedule budget -> detach the child -> -// wait for the child to report that it is recording -> write the pidfile -> -// return. +// rule -> verify normal instances are visible -> check the schedule budget -> +// detach the child -> wait for the child to report that it is recording -> +// write the pidfile -> return. // // The first-observation wait is not a convenience. Returning before it would // leave the deploy inside [from, first_poll] with no evidence — the exact -// blind interval the two-phase model exists to remove — and it is also what -// surfaces auth, name-resolution and parse failures BEFORE deploy.sh runs -// rather than ten minutes later. +// blind interval the record-then-check split exists to remove — and it is +// also what surfaces auth, name-resolution and parse failures BEFORE +// deploy.sh runs rather than ten minutes later. func Watch(ctx context.Context, cfg WatchConfig) error { cfg = cfg.withDefaults() if err := cfg.validate(); err != nil { @@ -165,8 +164,8 @@ func Watch(ctx context.Context, cfg WatchConfig) error { } // Hand the log over with Close, never Stop: a sentinel here would tell - // check the recording ended before the child had even started (§4.5). - // Closing also releases the flock the child is about to take. + // check the recording ended before the child had even started. Closing + // also releases the flock the child is about to take. if err := prep.writer.Close(); err != nil { return err } @@ -181,8 +180,8 @@ func Watch(ctx context.Context, cfg WatchConfig) error { // The PARENT writes the pidfile, not the child: check must find the pid the // instant Watch returns, and a child writing its own would race the very - // next step of the pipeline. A deviation from P6's argv list, and the - // reason the child is never given --pidfile at all. + // next step of the pipeline. That is why the child is never given + // --pidfile at all. // // It is written only once the child has reported ready, so no path through // this function leaves a pidfile naming a process that is not recording. @@ -295,8 +294,8 @@ type preparedWatch struct { // prepareWatch is everything the parent does before it detaches. It takes a // Source rather than building one so the paused-rule, first-observation, -// §3.2 and budget behaviours are all testable with a scripted fake — only the -// process spawning needs a real binary. +// instance-visibility and budget behaviours are all testable with a scripted +// fake — only the process spawning needs a real binary. func prepareWatch(ctx context.Context, cfg WatchConfig, src Source) (*preparedWatch, error) { version, err := src.Version(ctx) if err != nil { @@ -325,7 +324,7 @@ func prepareWatch(ctx context.Context, cfg WatchConfig, src Source) (*preparedWa for _, d := range resolved { // A cadence of zero would make the child spin: every rule is due the // instant it was marked. It also cannot be written into the header, - // where check requires a positive value to derive maxGap from (P5). + // where check requires a positive value to derive maxGap from. if rt[d.UID].pollEvery <= 0 { return nil, fmt.Errorf("rule %q (%s) reports intervalSeconds=%d: there is no poll cadence to record at", d.Title, d.UID, d.IntervalSeconds) @@ -363,18 +362,18 @@ func openRecording(ctx context.Context, cfg WatchConfig, src Source, writer *Wri return nil, err } - // A rule whose DEFINITION says is_paused is skipped (§12): it is not - // waited for, not scheduled and never polled. Waiting for one either hangs - // forever or errors before the deploy (§4.3), and recording polls for it - // would report an in-window pause (coverage check 7) for a rule that was - // already paused when the window opened — turning §12's exit 1 into an - // exit 2. The header still names it, with is_paused true, so check reports - // it as skipped from the definitions. + // A rule whose DEFINITION says is_paused is skipped: it is not waited for, + // not scheduled and never polled. Waiting for one either hangs forever or + // errors before the deploy, and recording polls for it would report an + // in-window pause (coverage check 7) for a rule that was already paused + // when the window opened — turning a skipped rule's exit 1 into an exit 2. + // The header still names it, with is_paused true, so check reports it as + // skipped from the definitions. var active []Definition activeTimings := make(map[string]ruleTimings, len(resolved)) for _, d := range resolved { if d.IsPaused { - fmt.Fprintf(cfg.Notes, "note: rule %q (%s) is paused: recorded as skipped, not waited for (§4.3)\n", d.Title, d.UID) + fmt.Fprintf(cfg.Notes, "note: rule %q (%s) is paused: recorded as skipped, not waited for\n", d.Title, d.UID) continue } active = append(active, d) @@ -392,9 +391,9 @@ func openRecording(ctx context.Context, cfg WatchConfig, src Source, writer *Wri } } - // Budget last, on the latencies just measured — never on a fixed estimate - // (§5.2). Only the active rules count: a skipped rule is never polled and - // consumes none of the capacity. + // Budget last, on the latencies just measured — never on a fixed estimate. + // Only the active rules count: a skipped rule is never polled and consumes + // none of the capacity. if err := CheckBudget(activeTimings, measured, cfg.Concurrency); err != nil { return nil, err } @@ -404,9 +403,9 @@ func openRecording(ctx context.Context, cfg WatchConfig, src Source, writer *Wri // loggedRules snapshots the resolved definitions into the header's rule list. // Every field but PollEverySeconds is forensic — a resolve-time snapshot that -// makes an uploaded log self-describing (§21.3) — while PollEverySeconds is +// makes an uploaded log self-describing — while PollEverySeconds is // load-bearing: it is the cadence this recording actually used, and check -// derives maxGap from it rather than from the definitions (P5). +// derives maxGap from it rather than from the definitions. func loggedRules(defs []Definition, rt map[string]ruleTimings) []LoggedRule { out := make([]LoggedRule, 0, len(defs)) for _, d := range defs { @@ -427,17 +426,17 @@ func loggedRules(defs []Definition, rt map[string]ruleTimings) []LoggedRule { } // firstObservations takes one observation of every rule in active, verifies -// §3.2 against those very responses, and reduces each into the poll record -// that IS the window's first heartbeat — plus the measured latency of each, -// which is the only honest input to §5.2's budget check (a fixed estimate is -// worthless when one rule's payload is ~230x another's). +// that normal instances are visible in those very responses, and reduces each +// into the poll record that IS the window's first heartbeat — plus the measured +// latency of each, which is the only honest input to the budget check (a fixed +// estimate is worthless when one rule's payload is ~230x another's). // -// Both entry paths share it: watch's parent, before it detaches (§4.3), and +// Both entry paths share it: watch's parent, before it detaches, and // single-step check's measurement pass, which keeps the polls as evidence -// rather than writing them to a log (P9). Keeping one implementation is the -// point — the §3.2 verification and the "absent is a warning, not an error" -// rule are exactly the places where two copies would silently drift, and a -// drift in either direction is fail-open. +// rather than writing them to a log. Keeping one implementation is the point — +// the instance-visibility verification and the "absent is a warning, not an +// error" rule are exactly the places where two copies would silently drift, and +// a drift in either direction is fail-open. // // polls come back in `active` order, so a log written from them is byte-stable // for a given set of observations. @@ -456,7 +455,7 @@ func firstObservations(ctx context.Context, src Source, active []Definition, red return nil, nil, err } - // Verify §3.2 before anything downstream relies on it: if the state + // Verify this before anything downstream relies on it: if the state // endpoint ever stops returning normal instances, the reduction's "keep // the non-normal ones" silently becomes "keep everything it happened to // send" and the transition markers lose their ground truth. @@ -473,12 +472,12 @@ func firstObservations(ctx context.Context, src Source, active []Definition, red measured[d.UID] = obs.Latency poll := reducer.Reduce(d.UID, obs) if !poll.Found { - // Authoritative, not transient (P2 already retried transport - // failures): the rule resolved in the ruler API but the state - // endpoint does not serve it. Recorded as Found=false, which P7 - // check 8 turns into unobservable — a note rather than an error - // here, because the state endpoint can lag a freshly created rule - // and the coverage proof fails closed either way. + // Authoritative, not transient (the transport already retried + // every transient failure): the rule resolved in the ruler API but + // the state endpoint does not serve it. Recorded as Found=false, + // which the coverage proof turns into unobservable — a note rather + // than an error here, because the state endpoint can lag a freshly + // created rule and the coverage proof fails closed either way. fmt.Fprintf(notes, "warning: rule %q (%s) is absent from the state endpoint; recorded as not found\n", d.Title, d.UID) } polls = append(polls, poll) @@ -488,8 +487,9 @@ func firstObservations(ctx context.Context, src Source, active []Definition, red // observeAll polls every rule in uids concurrently, bounded by concurrency, // and returns one Observation per rule that answered. Every rule is polled by -// TITLE (the ?rule_name= filter, §2.8) and selected out of the response by -// UID (§14.5) — a filtered response can carry several rules sharing one title. +// TITLE (the ?rule_name= filter is a title filter) and selected out of the +// response by UID — a filtered response can carry several rules sharing one +// title. // // It returns the successful observations alongside the first error in UID // order, so a caller that wants to keep the good heartbeats can, and the error @@ -537,9 +537,9 @@ func observeAll(ctx context.Context, src Source, titles map[string]string, uids // parent already wrote — one source of truth, no parent/child drift, and it // exercises ReadLog's header path — and the connection details come from the // inherited environment. Only the run facts the header does not carry travel -// in argv (§4.4). +// in argv. type DaemonChildConfig struct { - URL, Token string // from the inherited environment, never from argv (§20.2) + URL, Token string // from the inherited environment, never from argv Out string Until time.Time Concurrency int @@ -568,7 +568,7 @@ func RunDaemonChild(ctx context.Context, cfg DaemonChildConfig) error { } // Safe to read: the parent closed its writer before spawning this process, - // and no other writer can hold the log's flock (§4.4 step 4). + // and no other writer can hold the log's flock. header, polls, sentinel, err := ReadLog(cfg.Out) if err != nil { return err @@ -576,7 +576,7 @@ func RunDaemonChild(ctx context.Context, cfg DaemonChildConfig) error { if sentinel != nil { return fmt.Errorf("log %s already carries a stopped sentinel: another recorder finished it", cfg.Out) } - // The header's URL is the log's identity (§19.1 step 3). Checking it here + // The header's URL is the log's identity. Checking it here // catches a child that inherited an environment pointing somewhere else, // before it appends a single poll from the wrong Grafana. if header.URL != cfg.URL { @@ -596,7 +596,7 @@ func RunDaemonChild(ctx context.Context, cfg DaemonChildConfig) error { reducer := NewReducer() reducer.seedFrom(polls) - // SIGTERM is how check stops the recorder (§4.4 step 1); SIGINT is the + // SIGTERM is how check stops the recorder; SIGINT is the // same request from a human at a terminal. Both are clean stops, so both // end with a sentinel. Registered before the readiness report, so a signal // arriving the moment the parent unblocks is already handled. @@ -642,10 +642,10 @@ func reportReady(fd int) error { // childSchedule derives what the child polls, and how often, from the header // alone. The cadence comes from PollEverySeconds — the cadence the recording -// actually uses — and is never re-derived from the rule's evaluation interval: -// that is P5's "two authorities", and getting it wrong is fail-open in the -// faster-override direction. Paused rules are excluded here for the same -// reason the parent never polls them (§4.3, §12). +// actually uses — and is never re-derived from the rule's evaluation interval, +// which would be a second authority for the same value and is fail-open in the +// faster-override direction. Paused rules are excluded here for the same reason +// the parent never polls them. // // It returns cadences and nothing else. maxGap, healthGrace and evalStaleAfter // are coverage thresholds applied by the pure layer at classification time, so @@ -672,7 +672,7 @@ func childSchedule(h Header) (titles map[string]string, cadence map[string]time. // watchLoopConfig is the child's working state: what to poll, how often, and // where to append it. There is no threshold in here and no policy — the child -// records and classifies nothing (H5). +// records and classifies nothing. type watchLoopConfig struct { Src Source Writer *Writer @@ -690,7 +690,7 @@ type watchLoopConfig struct { // // The sentinel policy is the load-bearing part. A clean stop (a signal, or // Until) writes it; a hard error does NOT. A recorder that died must look -// exactly like a coverage gap to check, because it is one (§4.5) — writing a +// exactly like a coverage gap to check, because it is one — writing a // sentinel on the way out of a failure would hand check a "recording finished" // claim about a window that stopped being observed. func watchLoop(ctx context.Context, cfg watchLoopConfig) error { @@ -735,8 +735,8 @@ func watchLoop(ctx context.Context, cfg watchLoopConfig) error { pollErr := cfg.pollBatch(ctx, due) if ctx.Err() != nil { // Signalled while a poll was in flight. The aborted poll's error is - // not a recorder failure, and a clean stop wins over it (§4.4 step - // 1: finish the in-flight write, then the sentinel). + // not a recorder failure, and a clean stop wins over it: finish the + // in-flight write, then the sentinel. return cfg.Writer.Stop() } if pollErr != nil { @@ -780,7 +780,7 @@ func untilNextPoll(sched *Scheduler, until, now time.Time) (time.Duration, bool) return max(next.Sub(now), 0), true } -// writePidFile records the child's pid where check looks for it (P9's +// writePidFile records the child's pid where check looks for it (its // --pidfile, default .pid). The format is the decimal pid and a newline, // so `kill $(cat log.jsonl.pid)` works and ReadPidFile stays trivial. func writePidFile(path string, pid int) error { @@ -791,7 +791,7 @@ func writePidFile(path string, pid int) error { } // ReadPidFile is the other side of that contract: the pid of the recorder -// check must stop before it may read the log (§4.4 steps 1-4). +// check must stop before it may read the log. func ReadPidFile(path string) (int, error) { b, err := os.ReadFile(path) if err != nil { diff --git a/grafana-alertcheck/internal/gate/watch_daemon_test.go b/grafana-alertcheck/internal/gate/watch_daemon_test.go index db35c505b..c3a91cd12 100644 --- a/grafana-alertcheck/internal/gate/watch_daemon_test.go +++ b/grafana-alertcheck/internal/gate/watch_daemon_test.go @@ -21,7 +21,7 @@ import ( // os.Executable(), which under `go test` is this binary, so the one integration // test below exercises the real thing — a real fork/exec, a real setsid, a real // inherited environment, a real SIGTERM — with this function standing in for -// the CLI's `watch --daemon-child` dispatch, which lands in P10. +// the CLI's `watch --daemon-child` dispatch. func TestMain(m *testing.M) { if path := os.Getenv(lockHolderEnv); path != "" { os.Exit(runTestLockHolder(path)) @@ -66,8 +66,8 @@ func runTestLockHolder(path string) int { } // runTestDaemonChild parses the child argv childArgs() writes, and reads the -// connection details from the environment — never from argv (§20.2). P10's -// `watch` FlagSet does the same four flags. +// connection details from the environment — never from argv. The CLI's `watch` +// FlagSet does the same four flags. func runTestDaemonChild(args []string) int { cfg := DaemonChildConfig{ URL: os.Getenv("GRAFANA_URL"), @@ -116,7 +116,7 @@ func runTestDaemonChild(args []string) int { } // testBearerToken is what every request to grafanaTestServer must carry. The -// child never receives it in argv (§20.2), so a request that arrives +// child never receives it in argv, so a request that arrives // authenticated is proof that the token reached the detached process through // the inherited environment — and a 401 is what a test sees if that ever // breaks. @@ -147,7 +147,7 @@ func grafanaTestServer(t *testing.T) *httptest.Server { _, _ = w.Write(ruler) case strings.HasPrefix(r.URL.Path, "/api/prometheus/"): if r.URL.Query().Get("rule_name") == "" { - // §2.8: the gate must never read the state endpoint unfiltered. + // The gate must never read the state endpoint unfiltered. http.Error(w, "unfiltered state read", http.StatusBadRequest) return } @@ -212,14 +212,14 @@ func waitFor(t *testing.T, what string, timeout time.Duration, cond func() bool) t.Fatalf("timed out after %s waiting for %s", timeout, what) } -// TestWatchSpawnsADetachedRecorder is P6's one integration test: everything -// from the version gate to the sentinel, through a real detached process. +// The one watch integration test: everything from the version gate to the +// sentinel, through a real detached process. // // It asserts the four things only a real spawn can show — the pidfile points // at a live process, that process is in its own session (setsid, not a bare // `&`), it keeps appending after Watch returned, and SIGTERM makes it finish -// the log in the §4.4 order — and it uses a 200ms --poll-interval to do it in -// about a second, which also exercises the unclamped-override path (§5.1). +// the log in the stop order — and it uses a 200ms --poll-interval to do it in +// about a second, which also exercises the unclamped-override path. func TestWatchSpawnsADetachedRecorder(t *testing.T) { srv := grafanaTestServer(t) t.Setenv("GRAFANA_URL", srv.URL) @@ -263,7 +263,7 @@ func TestWatchSpawnsADetachedRecorder(t *testing.T) { t.Errorf("recorder pgid = %d, want %d: it did not get its own session", pgid, pid) } - // The parent already wrote the first heartbeat before it returned (§4.3); + // The parent already wrote the first heartbeat before it returned; // these later ones prove the detached child is the one appending now. waitFor(t, "the detached recorder to append its own polls", 10*time.Second, func() bool { _, polls, _, err := ReadLog(out) @@ -294,7 +294,7 @@ func TestWatchSpawnsADetachedRecorder(t *testing.T) { t.Fatalf("poll %d = %+v, want a found observation of %s", i, p, watchActiveUID) } if p.GrafanaNow.IsZero() { - t.Fatalf("poll %d has no grafana_now; H4 needs the Date header of its own response", i) + t.Fatalf("poll %d has no grafana_now; every poll needs the Date header of its own response", i) } } if sentinel.Before(header.StartedAt) { diff --git a/grafana-alertcheck/internal/gate/watch_process.go b/grafana-alertcheck/internal/gate/watch_process.go index 629606188..169a4f65a 100644 --- a/grafana-alertcheck/internal/gate/watch_process.go +++ b/grafana-alertcheck/internal/gate/watch_process.go @@ -23,7 +23,7 @@ type detachedChild struct { logOffset int64 } -// spawnChild re-execs this binary as the detached recorder (§4.4). A trailing +// spawnChild re-execs this binary as the detached recorder. A trailing // `&` is NOT sufficient: the child would keep the parent's session and process // group, so it would still take the terminal's signals and, on a runner, die // with the step that started it. Setsid gives it a new session AND a new @@ -52,7 +52,7 @@ func spawnChild(cfg WatchConfig) (detachedChild, error) { logOffset = info.Size() } - // The readiness pipe (§ReadyFDFlag): the child gets the write end as + // The readiness pipe: the child gets the write end as // descriptor 3 and reports on it once it holds the log and is polling. readyRead, readyWrite, err := os.Pipe() if err != nil { @@ -64,9 +64,9 @@ func spawnChild(cfg WatchConfig) (detachedChild, error) { cmd.Stdout = logFile cmd.Stderr = logFile cmd.ExtraFiles = []*os.File{readyWrite} // descriptor 3 in the child - // The environment is how the connection details reach the child (§20.2): - // the token must never appear in argv, where it would land in the process - // table and in CI logs. + // The environment is how the connection details reach the child: the token + // must never appear in argv, where it would land in the process table and + // in CI logs. cmd.Env = os.Environ() cmd.SysProcAttr = &syscall.SysProcAttr{Setsid: true} diff --git a/grafana-alertcheck/internal/gate/watch_test.go b/grafana-alertcheck/internal/gate/watch_test.go index 5a0337a37..33c455af7 100644 --- a/grafana-alertcheck/internal/gate/watch_test.go +++ b/grafana-alertcheck/internal/gate/watch_test.go @@ -92,9 +92,9 @@ func countPolls(polls []Poll, uid string) int { return n } -// TestWatchLoopPollsEachRuleAtItsOwnCadence is §5's per-rule schedule seen -// from the recorder: a 10s rule beside a 300s one keeps its own 5s cadence -// instead of dragging the slack rule along with it or being slowed to its pace. +// The per-rule schedule seen from the recorder: a 10s rule beside a 300s one +// keeps its own 5s cadence instead of dragging the slack rule along with it or +// being slowed to its pace. func TestWatchLoopPollsEachRuleAtItsOwnCadence(t *testing.T) { const tightUID, slackUID = "tight", "slack" path := filepath.Join(t.TempDir(), "log.jsonl") @@ -148,9 +148,8 @@ func TestWatchLoopPollsEachRuleAtItsOwnCadence(t *testing.T) { } } -// TestWatchLoopHardErrorLeavesNoSentinel is §4.5's fail-closed rule from the -// recorder's side: a recorder that dies must look exactly like a coverage gap, -// so it must not sign off the log on its way out. +// Fail-closed from the recorder's side: a recorder that dies must look exactly +// like a coverage gap, so it must not sign off the log on its way out. func TestWatchLoopHardErrorLeavesNoSentinel(t *testing.T) { path := filepath.Join(t.TempDir(), "log.jsonl") clock := newVirtualClock(testNow) @@ -191,10 +190,9 @@ func TestWatchLoopHardErrorLeavesNoSentinel(t *testing.T) { } } -// TestWatchLoopSignalDuringPollIsACleanStop pins §4.4 step 1: SIGTERM arriving -// while a poll is in flight is a clean stop, so the aborted poll's error must -// not suppress the sentinel — otherwise every normal check run, which stops the -// recorder exactly this way, would end unobservable. +// SIGTERM arriving while a poll is in flight is a clean stop, so the aborted +// poll's error must not suppress the sentinel — otherwise every normal check +// run, which stops the recorder exactly this way, would end unobservable. func TestWatchLoopSignalDuringPollIsACleanStop(t *testing.T) { path := filepath.Join(t.TempDir(), "log.jsonl") clock := newVirtualClock(testNow) @@ -311,7 +309,7 @@ func TestWatchLoopPollBatchKeepsTheHeartbeatsItGot(t *testing.T) { } } -// TestReducerSeedFromKeepsMarkersAcrossTheHandoff is H2 at the one seam P6 +// The vanish-versus-clear distinction at the one seam the parent/child handoff // introduces. The parent observes a firing instance; the child starts with a // fresh Reducer and sees the instance gone. Seeded, that is a vanish — a // discontinuity. Unseeded, it is nothing at all, and the instance silently @@ -321,7 +319,7 @@ func TestReducerSeedFromKeepsMarkersAcrossTheHandoff(t *testing.T) { key := instanceKey(firing.Labels) parentPoll := Poll{RuleUID: "r1", Found: true, Abnormal: []Instance{firing}} // The child's first response: the instance is gone from the response - // entirely, which is a vanish and never a clear (§4.7). + // entirely, which is a vanish and never a clear. childObs := observation(testNow, testStateRule("r1", "Example", time.Minute, testNow)) t.Run("seeded", func(t *testing.T) { @@ -385,10 +383,9 @@ func liveObservation(grafanaNow time.Time) Observation { testInstance(StateNormal, "", "a"))) } -// TestPrepareWatchDoesNotWaitForPausedRules is §22.4's regression test: a rule -// paused in its definition is skipped, never waited for. Waiting for one either -// hangs forever or errors before the deploy — and the header must still name -// it, so check can report it as skipped rather than lose it. +// A rule paused in its definition is skipped, never waited for. Waiting for one +// either hangs forever or errors before the deploy — and the header must still +// name it, so check can report it as skipped rather than lose it. func TestPrepareWatchDoesNotWaitForPausedRules(t *testing.T) { var notes strings.Builder cfg := watchTestConfig(t, ¬es, "uid:"+watchActiveUID, "uid:"+watchPausedUID) @@ -423,16 +420,16 @@ func TestPrepareWatchDoesNotWaitForPausedRules(t *testing.T) { } // One poll, for the live rule only — and it is already in the log before - // prepareWatch returned, which is the whole point of §4.3. + // prepareWatch returned, which is the whole point of the record step. if len(polls) != 1 || polls[0].RuleUID != watchActiveUID { t.Fatalf("polls = %+v, want exactly one first observation of %s", polls, watchActiveUID) } if !polls[0].Found || !polls[0].GrafanaNow.Equal(testNow) { t.Errorf("first poll = %+v, want a found observation at %s", polls[0], testNow) } - // §22.3: "the poll record holds the state histogram. Assert that watch - // writes it" — through a real prepareWatch()/Reducer call, not just - // log_test.go's hand-built Writer/ReadLog round trip. + // The poll record holds the state histogram, asserted through a real + // prepareWatch()/Reducer call rather than log_test.go's hand-built + // Writer/ReadLog round trip. if want := map[string]int{"normal": 1}; !maps.Equal(polls[0].Histogram, want) { t.Errorf("Histogram = %v, want %v: watch must record the state histogram on every poll it writes", polls[0].Histogram, want) } @@ -441,9 +438,9 @@ func TestPrepareWatchDoesNotWaitForPausedRules(t *testing.T) { } } -// TestPrepareWatchHeaderRecordsTheOverriddenCadence is P5's "two authorities" -// from the writing side: whatever --poll-interval resolves to is what the -// header records, because that is the only value check may derive maxGap from. +// One authority for the cadence, from the writing side: whatever +// --poll-interval resolves to is what the header records, because that is the +// only value check may derive maxGap from. func TestPrepareWatchHeaderRecordsTheOverriddenCadence(t *testing.T) { var notes strings.Builder cfg := watchTestConfig(t, ¬es, "uid:"+watchActiveUID) @@ -467,8 +464,8 @@ func TestPrepareWatchHeaderRecordsTheOverriddenCadence(t *testing.T) { } } -// TestPrepareWatchFailsWhenTheScheduleDoesNotFit: the budget check runs on the -// latencies the parent just measured, before the deploy runs (§5.2). +// The budget check runs on the latencies the parent just measured, before the +// deploy runs. func TestPrepareWatchFailsWhenTheScheduleDoesNotFit(t *testing.T) { var notes strings.Builder cfg := watchTestConfig(t, ¬es, "uid:"+watchActiveUID) @@ -484,9 +481,9 @@ func TestPrepareWatchFailsWhenTheScheduleDoesNotFit(t *testing.T) { assertBudgetMessage(t, err.Error()) } -// TestPrepareWatchVerifiesNormalInstancesAreVisible is the §3.2 check at the -// one place it can still be cheap: the first observation. If the state endpoint -// stops returning normal instances, the reduction's predicate quietly inverts. +// Normal instances are verified visible at the one place it is still cheap: +// the first observation. If the state endpoint stops returning them, the +// reduction's predicate quietly inverts. func TestPrepareWatchVerifiesNormalInstancesAreVisible(t *testing.T) { var notes strings.Builder cfg := watchTestConfig(t, ¬es, "uid:"+watchActiveUID) @@ -499,8 +496,8 @@ func TestPrepareWatchVerifiesNormalInstancesAreVisible(t *testing.T) { if err == nil { t.Fatal("prepareWatch: no error when totals claim normal instances the response omitted") } - if !strings.Contains(err.Error(), "3.2") { - t.Errorf("error does not name §3.2: %v", err) + if !strings.Contains(err.Error(), "no longer returns normal instances") { + t.Errorf("error does not say the endpoint stopped returning normal instances: %v", err) } // The failure happens before any poll is appended, so the log holds a @@ -525,9 +522,9 @@ func TestPrepareWatchRejectsAnUnsupportedGrafana(t *testing.T) { } } -// TestPrepareWatchNotesAnAbsentRule: a rule that resolved in the ruler API but -// is absent from the state endpoint is recorded as Found=false — authoritative -// evidence P7 turns into unobservable — not silently dropped. +// A rule that resolved in the ruler API but is absent from the state endpoint +// is recorded as Found=false — authoritative evidence the coverage proof turns +// into unobservable — not silently dropped. func TestPrepareWatchNotesAnAbsentRule(t *testing.T) { var notes strings.Builder cfg := watchTestConfig(t, ¬es, "uid:"+watchActiveUID) @@ -601,8 +598,8 @@ func TestWatchConfigValidation(t *testing.T) { }) } -// TestChildScheduleUsesTheRecordedCadence is P5's fail-open direction, checked -// on the child's side: a log recorded at 5s on a 300s rule must schedule at 5s. +// The fail-open direction, checked on the child's side: a log recorded at 5s on +// a 300s rule must schedule at 5s. // Re-deriving from the interval would give 150s — and every real 250s hole in // that recording would pass. func TestChildScheduleUsesTheRecordedCadence(t *testing.T) { @@ -616,7 +613,7 @@ func TestChildScheduleUsesTheRecordedCadence(t *testing.T) { t.Fatalf("childSchedule: %v", err) } if _, ok := titles["paused"]; ok { - t.Error("the child scheduled a rule that was paused when the window opened (§4.3)") + t.Error("the child scheduled a rule that was paused when the window opened") } if got := cadence["fast"]; got != 5*time.Second { t.Errorf("pollEvery = %s, want 5s from the header, not %s from the interval", got, defaultPollEvery(300)) From 80d92ee7131b17fe089ddaaf49e3e1b05b6dec48 Mon Sep 17 00:00:00 2001 From: Bartek Tofel Date: Wed, 2 Sep 2026 13:29:34 +0200 Subject: [PATCH 2/5] fix: merge conflict --- grafana-alertcheck/internal/gate/classify.go | 17 ++--------------- 1 file changed, 2 insertions(+), 15 deletions(-) diff --git a/grafana-alertcheck/internal/gate/classify.go b/grafana-alertcheck/internal/gate/classify.go index 5d28c16ce..ca4a4222c 100644 --- a/grafana-alertcheck/internal/gate/classify.go +++ b/grafana-alertcheck/internal/gate/classify.go @@ -613,18 +613,6 @@ func decide(h Header, polls []Poll, sentinel *time.Time, defs []Definition, }) } -<<<<<<< HEAD - // MinObserved (§12): default len(defs) after the collapse (already done - // by Resolve before decide ever sees defs). skipped rules count against - // it unless AllowPaused says otherwise. A shortfall counts toward exit 1 - // (§9.1), never exit 2 — decide never returns an error for this — and H7 - // requires it to surface through Violations like any other fail reason, - // so a shortfall always produces at least one, even when no rule is - // paused at all (an operator-supplied MinObserved that simply exceeds - // what could ever be resolved). - counted := watchedCount - var attributable []Definition -======= // MinObserved defaults to len(defs) after duplicate names collapse // (already done by Resolve before decide ever sees defs). Skipped rules // count against it unless AllowPaused says otherwise. A shortfall counts @@ -633,9 +621,8 @@ func decide(h Header, polls []Poll, sentinel *time.Time, defs []Definition, // a shortfall always produces at least one, even when no rule is paused at // all (an operator-supplied MinObserved that simply exceeds what could ever // be resolved). - counted := observedCount - var chargeable []Definition ->>>>>>> 641701bb (chore: more concise comments) + counted := watchedCount + var attributable []Definition if pol.AllowPaused { counted += len(skippedRules) } else { From 34040dfc241a8bd2fe2adc2b815e46ad79e94eca Mon Sep 17 00:00:00 2001 From: Bartek Tofel Date: Fri, 4 Sep 2026 17:01:53 +0200 Subject: [PATCH 3/5] chore: shorten comments --- grafana-alertcheck/internal/gate/check.go | 211 ++++++------------- grafana-alertcheck/internal/gate/classify.go | 155 +++++--------- grafana-alertcheck/internal/gate/coverage.go | 149 ++++--------- grafana-alertcheck/internal/gate/log.go | 105 +++------ grafana-alertcheck/internal/gate/schedule.go | 175 ++++++--------- grafana-alertcheck/internal/gate/source.go | 57 ++--- grafana-alertcheck/internal/gate/watch.go | 122 ++++------- 7 files changed, 320 insertions(+), 654 deletions(-) diff --git a/grafana-alertcheck/internal/gate/check.go b/grafana-alertcheck/internal/gate/check.go index 0f21b4d86..eab5d931b 100644 --- a/grafana-alertcheck/internal/gate/check.go +++ b/grafana-alertcheck/internal/gate/check.go @@ -14,26 +14,21 @@ import ( // Check returns (Result, error) and no exit code: the code is a presentation // decision the CLI makes. err != nil is exit 2 unconditionally, even alongside // real violations; violations with err == nil is exit 1; neither is exit 0. -// -// Check never reads the environment either. The URL and token are read by the -// CLI and passed in as fields, and the token must never reach a *flag.FlagSet. +// Check never reads the environment — the CLI reads URL/token and passes them +// in, and the token must never reach a *flag.FlagSet. // countdownEvery is how often the collection loop reports what it is waiting -// for. A silent wait is indistinguishable from a hung process, and the wait -// after `to` is the longest silence in the whole run. +// for; a silent wait is indistinguishable from a hung process. const countdownEvery = 30 * time.Second -// recorderStopTimeout bounds the wait for the recorder's exit. Everything the -// recorder does after SIGTERM is local (finish the in-flight write, append the -// sentinel, fsync) and an in-flight poll aborts through the child's own -// context, so the real figure is milliseconds; this is loose enough for an -// overloaded runner. The timeout is a hard error rather than a longer wait — a -// log a writer may still hold cannot be read at all. +// recorderStopTimeout bounds the wait for the recorder's exit after SIGTERM. +// Everything after the signal is local (finish the in-flight write, sentinel, +// fsync), so this is loose; it stays a hard error because a log a writer still +// holds cannot be read. const recorderStopTimeout = 30 * time.Second -// recorderStopPoll is how often that wait re-checks the pid. There is no -// wait(2) available: the recorder is a detached session leader, not this -// process's child, so its exit can only be observed by polling. +// recorderStopPoll is how often the wait re-checks the lock. With no wait(2) +// on a detached session leader, its exit is observable only by polling. const recorderStopPoll = 100 * time.Millisecond // Config is check's whole input. It is the CLI's view of a run, and it is @@ -117,14 +112,10 @@ func (cfg Config) namedAlerts() []string { } // Check is the I/O shell: HTTP, signals, the pidfile, file reads, the -// countdown print. Every correctness question it touches is answered -// elsewhere — by proveCoverage and decide, which are pure — and that split is -// the most important seam in the project. Check therefore needs two -// integration tests; decide carries the suite. -// -// A pass is exactly len(Violations) == 0 && err == nil. Every error path below -// leaves err non-nil, and no path anywhere in this file converts an error into -// an empty Result with a nil error. +// countdown print. Every correctness question it touches is answered elsewhere +// (proveCoverage, decide — both pure), which is the most important seam in the +// project. A pass is exactly len(Violations) == 0 && err == nil; every error +// path leaves err non-nil. func Check(ctx context.Context, cfg Config) (Result, error) { cfg = cfg.withDefaults() if err := cfg.validate(); err != nil { @@ -160,10 +151,8 @@ func (cfg Config) validate() error { } now := cfg.Clock.Now() - // from mirrors what check() will use, so the two window checks below judge - // the window that will really be classified. The fallback is not written - // back into cfg: check() re-reads the clock at the same point, and one - // authority for that value is better than two that could disagree. + // from mirrors what check() will use, so the window checks below judge the + // window that will really be classified. from := cfg.From switch { case from.IsZero() && cfg.Log != "": @@ -185,13 +174,10 @@ func (cfg Config) validate() error { from.Format(time.RFC3339), fromFutureTolerance, now.Format(time.RFC3339)) } - // A `to` already in the past is not a special mode WITH a log: the - // collection loop's condition is simply already true and the evidence is - // classified immediately. Without one it is a different thing entirely — a - // request to prove a window that nothing observed. Refusing it is not - // pedantry: the coverage window would end before the first observation, - // every heartbeat gap inside it would measure negative, and the run would - // report a proved window it never saw. + // A `to` in the past is fine WITH a log (the collection loop is already + // done). Without one it is a request to prove a window nothing observed: + // every heartbeat gap would measure negative, and the run would report a + // proved window it never saw. if cfg.Log == "" && !cfg.To.After(now) { return fmt.Errorf("check: `to` %s has already passed and there is no recorded log: a window that ended before check started can only be classified from a recording", cfg.To.Format(time.RFC3339)) @@ -232,11 +218,9 @@ func check(ctx context.Context, cfg Config, src Source) (Result, error) { // ---- With a log, validate its identity. ------------------------------- // The header is read early — line 1 only, the one line a writer can never - // change (ReadLogHeader) — so a wrong URL or a rule that no longer - // resolves fails closed NOW rather than after the whole window has - // elapsed. It is advisory: the authoritative header comes from the single - // full ReadLog once collection is over and the writer has exited, and the - // identity is validated again against that one. + // change — so a wrong URL or an unresolvable rule fails closed NOW. It is + // advisory: the authoritative header is re-read once collection ends and + // the writer has exited. var ( resolved []Definition notes []string @@ -291,11 +275,8 @@ func check(ctx context.Context, cfg Config, src Source) (Result, error) { fmt.Fprintf(cfg.Notes, "warning: %s\n", warning) } - // The measurement pass and the budget check belong to single-step mode - // alone: in recorder mode watch already took one observation of every rule - // and checked the budget against those measured latencies before it - // detached, and repeating it here would spend a second poll of every rule - // to re-answer a question already answered. + // The measurement pass and budget check are single-step only: in recorder + // mode watch already measured and checked the budget before detaching. var ( header Header initial []Poll @@ -364,10 +345,8 @@ func check(ctx context.Context, cfg Config, src Source) (Result, error) { } collected, err := collectUntil(ctx, cfg, windowEnd, poller) if err != nil { - // The failure limit was exceeded (retryTransport already gave every - // transient failure its backoff), or the context ended. Nothing - // collected is classified — the count is there so an operator can tell - // a run that failed at once from one that failed at minute nine. + // Nothing collected is classified; the count lets an operator tell a + // run that failed at once from one that failed at minute nine. return Result{}, fmt.Errorf("collect evidence after %d poll(s): %w", len(collected), err) } @@ -385,29 +364,20 @@ func check(ctx context.Context, cfg Config, src Source) (Result, error) { return Result{}, err } header, polls, sentinel, err = ReadLog(cfg.Log) - // The lock stays held across the read, so no writer can appear between - // the proof that there was none and the read itself. Released here - // rather than deferred: everything past this point works from bytes - // already in memory, and the drain wait below can take minutes. + // Held across the read so no writer can appear mid-read, then released + // (everything past here works from memory, and the drain wait is minutes). _ = heldLog.Close() if err != nil { return Result{}, err } - // The authoritative header, validated the same way the advisory one - // was — and its result is KEPT. Everything from here on judges the - // header ReadLog returned, so nothing downstream rests on the advisory - // read having been right. That read is what it claims to be: a - // fail-fast, and no part of the verdict depends on it. + // The authoritative header wins: the advisory read was only a fail-fast. resolved, _, err = resolveFromLog(allDefs, header, cfg) if err != nil { return Result{}, err } - // rt is re-derived because it depends on the header: PollEverySeconds - // is the one load-bearing value the advisory read supplied. windowEnd - // is deliberately NOT recomputed from the gt this returns: the - // collection loop has already stopped at the earlier value, and moving - // the end of the window afterwards would prove a window this run did - // not collect. + // rt is re-derived from the authoritative header. windowEnd is NOT + // recomputed: the loop already stopped at the earlier value, and moving + // it afterwards would prove a window this run did not collect. if rt, gt, err = DeriveTimingsFromLog(header, resolved); err != nil { return Result{}, fmt.Errorf("log identity: %w", err) } @@ -453,24 +423,14 @@ func check(ctx context.Context, cfg Config, src Source) (Result, error) { return result, errors.Join(decideErr, drainErr) } -// resolveFromLog turns a log header into the resolved definitions, and is the -// log's identity check in practice. Three things are verified: the URL -// matches, the schema version matches (ReadLog/ReadLogHeader own that), and -// every header UID still resolves against the fresh ruler read. The alert set -// is TAKEN from the log, never compared — with Alerts required empty in log -// mode there is nothing to compare it against, and refusing a log recorded -// against a different alert set is exactly this URL-and-UID failure. +// resolveFromLog is the log's identity check in practice: the URL must match +// and every header UID must still resolve against a fresh ruler read. The +// alert set is TAKEN from the log, never compared against --alerts (which the +// validator requires empty in log mode). Resolving through Resolve by uid: +// keeps one implementation of the resolution rules. // -// Resolving through Resolve, by uid:, rather than by a private lookup, keeps -// one implementation of the resolution rules: a header naming a recording or -// datasource-managed rule gets the same specific refusal an operator would, -// and a header naming the same UID twice collapses with a note -// (DeriveTimingsFromLog rejects that case outright, so the note is belt and -// braces). -// -// Only the header-to-defs direction needs checking. The opposite direction -// cannot fail here: resolved is BUILT from the header, so no resolved -// definition can be absent from it. +// Only the header-to-defs direction can fail: resolved is BUILT from the +// header, so no resolved definition can be absent from it. func resolveFromLog(allDefs []Definition, h Header, cfg Config) ([]Definition, []string, error) { if h.URL != cfg.URL { return nil, nil, fmt.Errorf("log identity: %s recorded url %q but this run is configured for %q", @@ -619,32 +579,22 @@ func collectUntil(ctx context.Context, cfg Config, deadline time.Time, p *livePo } } -// stopRecorder signals the recorder and waits for it to go. Nothing here is -// best-effort: the log may not be read until the writer has provably gone, so -// every failure to reach that state is a hard error. -// -// It returns the log held under an exclusive flock. The caller must keep that -// file open across ReadLog and close it afterwards — the lock is the proof -// that no writer exists, and holding it across the read also shuts out a new -// one appearing between the proof and the read. +// stopRecorder signals the recorder and waits for it to go; the log may not be +// read until the writer has provably gone, so every failure is a hard error. +// It returns the log held under an exclusive flock, which the caller must keep +// open across ReadLog — the lock is the proof that no writer exists. // -// Two authorities, and only one of them is evidence: +// Two authorities, only one of which is evidence: // -// - The PIDFILE says whether a recording was ever started, and an absent or -// unparseable one must never read as "there was nothing to stop". The -// parent writes the pidfile only AFTER the child reports that it holds the -// log and is polling, and removes it on every failing path, so a missing -// one means watch failed and this run has no evidence at all. -// - The FLOCK says whether a writer exists RIGHT NOW. Nothing removes the -// pidfile when a recorder exits cleanly — the parent has long returned and -// the child never learns the path — so after a --until run, a supported -// flow, the pidfile names a pid nobody owns. Signalling it would SIGTERM -// whatever same-user process inherited that pid. The kernel releases a -// flock when its holder exits, crash included, so the lock cannot go -// stale that way. +// - the PIDFILE says whether a recording ever started (it is written only +// after the child reports ready, and removed on failure). +// - the FLOCK says whether a writer exists right now. A pidfile can go stale +// — nothing removes it on a clean --until stop, so it may name a pid +// somebody else now owns — but the kernel drops a flock when the holder +// exits, so the lock is always authoritative. // -// So: read the pidfile to learn that a recording happened, then ask the lock -// whether it is still running, and signal only if it is. +// So: read the pidfile to learn a recording happened, ask the lock whether it +// is still running, and signal only if it is. func stopRecorder(ctx context.Context, cfg Config) (*os.File, error) { pid, err := ReadPidFile(cfg.PidFile) if err != nil { @@ -724,27 +674,15 @@ type drainVerdict struct { note string } -// drainWait is the final instance of the liveness check, asking each rule the -// last question — did you evaluate through the end of the window? A rule that -// cannot answer within drainTimeout is unobservable, never a pass. +// drainWait is the final liveness check: did each rule evaluate through the +// end of the window? A rule that cannot answer within drainTimeout is +// unobservable, never a pass. It returns one verdict per rule it could not +// clear (keyed by UID); an error only for a hard failure of the wait itself. // -// It returns one verdict per rule it could not clear, keyed by UID, which the -// caller folds into the Result. It returns an error only for a hard failure of -// the wait itself; a rule that simply never catches up is reported, not -// raised. -// -// Two kinds of rule are excluded before the wait starts, both because draining -// them could not change a verdict: -// -// - a rule the HEADER says was already paused when the recording opened: it -// is skipped, it was not evaluating, and it never was — there is no -// evaluation to wait for. The header and not the definition, for decide's -// reason (Header.pausedAtStart): a rule the header says was active must be -// drained or faulted, because a pause somebody applied after the window is -// not evidence about the window; -// - a rule whose last poll says Found == false: the rule-absent coverage -// check already makes it unobservable, so the only thing draining it could -// add is drainTimeout of waiting before the same answer. +// Two kinds of rule are excluded up front because draining them could not +// change a verdict: a rule the HEADER says was paused at the window open (the +// header, not the late-resolved definitions — see Header.pausedAtStart), and a +// rule whose last poll says Found == false (already unobservable via rule_absent). func drainWait(ctx context.Context, cfg Config, src Source, defs []Definition, pausedAtStart map[string]bool, rt map[string]ruleTimings, polls []Poll, windowEnd time.Time, timeout time.Duration) (map[string]drainVerdict, error) { @@ -872,16 +810,11 @@ func anyPollEvaluatedThrough(polls []Poll, windowEnd time.Time) bool { return false } -// evaluatedThrough is the drain wait's one comparison, and it is cross-domain: -// lastEvaluation is a Grafana timestamp and windowEnd is runner-domain, so the -// Grafana value is translated by its own poll's skew. The skew BOUND is then -// subtracted rather than added — the pessimistic end of the uncertainty — so -// an evaluation that only might have reached the end of the window does not -// count as one that did. Understating it costs a few more seconds of waiting; -// overstating it would pass an unproven window. -// -// A zero lastEvaluation never satisfies the wait: only a paused rule may -// legitimately report it, and a paused rule has nothing to drain. +// evaluatedThrough is the drain wait's cross-domain comparison: a Grafana +// lastEvaluation is translated by its poll's skew, and the bound is SUBTRACTED +// (the pessimistic end) so an evaluation that only *might* have reached the +// window end is not counted as having reached it. A zero lastEvaluation never +// satisfies the wait. func evaluatedThrough(lastEval time.Time, skew, bound time.Duration, windowEnd time.Time) bool { if lastEval.IsZero() { return false @@ -889,16 +822,10 @@ func evaluatedThrough(lastEval time.Time, skew, bound time.Duration, windowEnd t return !lastEval.Add(-skew).Add(-bound).Before(windowEnd) } -// mergeDrainTimeouts folds the I/O drain wait's verdicts into the pure layer's -// Result. It runs immediately after decide rather than before it, because -// decide owns proveCoverage and therefore builds the Coverage map itself; that -// keeps decide a pure function of its arguments. -// -// It returns its own error rather than mutating decide's, so neither hides the -// other: a run with one rule unobservable from the coverage proof and another -// from the drain wait must name both. The error says "at the drain wait" for -// that reason — the two are joined into one message, and two counts under one -// identical phrase read as a contradiction rather than as two findings. +// mergeDrainTimeouts folds the drain wait's I/O verdicts into the pure Result, +// running after decide so that function stays pure of its arguments. It returns +// its own error rather than mutating decide's so neither hides the other: a run +// faulted by both the coverage proof and the drain wait must name both. func mergeDrainTimeouts(res Result, drained map[string]drainVerdict) (Result, error) { if len(drained) == 0 { return res, nil diff --git a/grafana-alertcheck/internal/gate/classify.go b/grafana-alertcheck/internal/gate/classify.go index ca4a4222c..9d881334a 100644 --- a/grafana-alertcheck/internal/gate/classify.go +++ b/grafana-alertcheck/internal/gate/classify.go @@ -7,19 +7,15 @@ import ( "time" ) -// ReasonNodata is decide's own unobservable reason: proveCoverage deliberately -// never sets it — health=nodata is a note there, never fatal, because -// escalating it needs Policy.NodataIsUnobservable, and the pure coverage layer -// has no Policy to consult (coverage.go, check 5). decide is the seam that DOES -// have a Policy, so the escalation lives here. +// ReasonNodata is decide's own unobservable reason: proveCoverage never sets it +// — health=nodata is a note there, never fatal — because escalating it needs +// Policy.NodataIsUnobservable, which only decide (the Policy-holding seam) has. const ReasonNodata UnobservableReason = "nodata" -// Outcome is the verdict of one instance's timeline, and — after decide takes -// the worst across a rule's instances — of the rule itself. It is a published -// JSON output: the three fail values stay distinct even though v1 maps all -// three to exit 1, because a later reason string cannot recover the -// information a single "fail" value would have thrown away, and because -// splitting them later would break a published interface for no gain. +// Outcome is the verdict of one instance's timeline, and (after decide takes +// the worst across instances) of the rule. It is a published JSON output: the +// fail values stay distinct even though v1 maps them all to exit 1, so a later +// version can split them without breaking the interface. type Outcome string const ( @@ -62,28 +58,20 @@ type Violation struct { State State Health string // raw, reporting-only, like Poll.Health LastError string - // FirstSeen is the episode's onset, in the runner domain: activeAt - // translated by its poll's own skew when the episode opened strictly - // inside the window, or `from` itself when the instance was already bad - // at window-open (preexisting) — never a raw, untranslated Grafana - // timestamp. + // FirstSeen is the episode's onset in the runner domain (translated by the + // poll's own skew), or `from` when preexisting — never a raw Grafana time. FirstSeen time.Time - // ClearedAt is zero unless the episode closed via a genuine Cleared - // event, also translated to the runner domain. + // ClearedAt is zero unless the episode closed via a genuine Cleared event. ClearedAt time.Time InstanceLabels map[string]string - // Note carries an explanation for a Violation that has no instance - // behind it — the synthetic MinObserved shortfall entry decide emits when - // the deficit exceeds what any named paused rule explains. LastError is - // reporting-only rule state from a real poll and must not double as a - // message field for a Violation that never touched one. + // Note explains a Violation with no instance behind it — decide's synthetic + // MinObserved shortfall — and must not double as LastError (reporting-only + // rule state from a real poll). Note string } -// RuleVerdict is one rule's worst-of outcome, always present for every -// resolved rule — Verdicts includes the passes, not only the failures — so a -// human reading the table sees every alert that was asked for, not only the -// ones that misbehaved. +// RuleVerdict is one rule's worst-of outcome, present for every resolved rule +// (passes included) so the table shows every alert asked for. type RuleVerdict struct { Alert, RuleUID string Outcome Outcome @@ -176,27 +164,17 @@ type instanceTimeline struct { episodes []episode } -// runnerTime translates a Grafana-domain timestamp recorded on poll p into the -// runner domain, undoing that poll's own measured skew. GrafanaNow and -// ActiveAt come from the same response, so the same poll's skew applies to -// both. This is the single implementation of that translation for the package -// (same drift argument as pollsForRule): coverage.go's window membership test -// and heartbeat boundary segments call it too, rather than each keeping its own -// copy of `p.GrafanaNow.Add(-p.Skew())` that could silently diverge from this -// one. +// runnerTime translates a Grafana-domain timestamp into the runner domain by +// undoing poll p's measured skew. The single implementation for the package — +// coverage.go's window-membership and heartbeat-boundary checks use it too. func runnerTime(p Poll, grafanaDomain time.Time) time.Time { return grafanaDomain.Add(-p.Skew()) } // classifyRule builds every instance timeline for one rule across -// [from, windowEnd] and reduces them to the rule's worst outcome, its merged -// BadFor, and the Violations the preexisting policy actually charges against -// the run. It is PURE: no I/O, no clock reads — decide supplies windowEnd -// (to + transitionGrace) rather than this function deriving it, so a test can -// pin the boundary directly. -// -// polls need not be pre-filtered to this rule, matching proveCoverage's own -// contract: selection is by def.UID. +// [from, windowEnd] and reduces them to the rule's worst outcome, merged +// BadFor, and the Violations the preexisting policy charges against the run. +// PURE: no I/O, no clock reads; polls need not be pre-filtered to this rule. func classifyRule(def Definition, polls []Poll, from, windowEnd time.Time, badStates map[State]bool, pol PreexistingPolicy) (Outcome, time.Duration, []Violation) { rulePolls := pollsForRule(polls, def.UID) inWindow := inWindowPolls(rulePolls, from, windowEnd) @@ -204,11 +182,9 @@ func classifyRule(def Definition, polls []Poll, from, windowEnd time.Time, badSt timelines := make(map[string]*instanceTimeline) order := make([]string, 0) - // get backfills labels the first time a real Instance is seen: a key can - // be created earlier by a bare Cleared/Vanished marker, which carries - // no labels of its own, and the instance later re-firing must not report - // an empty InstanceLabels just because of which event happened to create - // the timeline first. + // get backfills labels on the first real Instance: a bare Cleared/Vanished + // marker can create the timeline first (with no labels), and a later re-fire + // must not report an empty InstanceLabels. get := func(key string, labels map[string]string) *instanceTimeline { tl, ok := timelines[key] if !ok { @@ -228,30 +204,26 @@ func classifyRule(def Definition, polls []Poll, from, windowEnd time.Time, badSt tl.episodeStart = start } closeEpisode := func(tl *instanceTimeline, end time.Time, real bool) { - // inWindowPolls admits a poll whose translated time is up to its own - // skew bound PAST windowEnd (the membership test widens the boundary - // outward). Without this clamp a genuine Cleared event on such a - // poll would produce an episode.end slightly beyond windowEnd, - // contradicting the episode type's own "clamped to - // [from, windowEnd]" contract. + // inWindowPolls widens its boundary outward by the skew bound, so a + // translated end can land past windowEnd or before episodeStart; clamp + // both, otherwise mergeDurations gets an inverted span. if end.After(windowEnd) { end = windowEnd } - // Different polls can carry different measured skews. In theory a - // closing poll's translated time could land before the opening - // poll's — skew is capped at SkewHardLimit (60s), so this is remote, - // not impossible — and a negative span would feed mergeDurations a - // duration that subtracts instead of adds. Clamp rather than trust - // the arithmetic never to invert. if end.Before(tl.episodeStart) { end = tl.episodeStart } tl.episodes = append(tl.episodes, episode{start: tl.episodeStart, end: end, closedByRealClear: real}) tl.badOpen = false } +<<<<<<< HEAD // onsetOf resolves a fresh episode's start: the instance's own ActiveAt, // translated to the runner domain by this poll's skew, clamped to // [from, windowEnd]. +======= + // onsetOf is a fresh episode's start: the translate ActiveAt, clamped to + // never read as starting before the window opened. +>>>>>>> 056b9146 (chore: shorten comments) onsetOf := func(p Poll, inst Instance) time.Time { start := runnerTime(p, inst.ActiveAt) if start.Before(from) { @@ -276,12 +248,9 @@ func classifyRule(def Definition, polls []Poll, from, windowEnd time.Time, badSt case !tl.seen: tl.seen = true if bad { - // Fail-closed: only call an onset "preexisting" when - // even the worst-case skew error still puts it at or - // before `from`. An onset that might really have landed - // just inside the window must classify as a new episode, - // never earn the `recovered` benefit of the doubt it - // would get if it later clears. + // Fail-closed: "preexisting" only when even the worst-case + // skew error places the onset at or before `from`; an onset + // that might be in-window must classify as a new episode. activeAtRunner := runnerTime(p, inst.ActiveAt) tl.preexisting = !activeAtRunner.Add(p.SkewBound()).After(from) if tl.preexisting { @@ -301,11 +270,8 @@ func classifyRule(def Definition, polls []Poll, from, windowEnd time.Time, badSt for _, key := range p.Cleared { tl := get(key, nil) if !tl.seen { - // Cleared on the very first mention means the transition - // happened between the poll just before this one (possibly - // pre-window) and this one: there is no window-internal - // evidence that it was ever bad, so it is neither - // preexisting nor a new episode. + // Cleared on first mention: the transition happened pre-window, + // with no in-window evidence it was ever bad. tl.seen = true continue } @@ -315,9 +281,8 @@ func classifyRule(def Definition, polls []Poll, from, windowEnd time.Time, badSt tl.lastHealth, tl.lastError = p.Health, p.LastError } - // Vanished is a deliberate no-op: freeze whatever badOpen/preexisting - // already holds. An instance that vanishes while bad must stay bad, and - // one that vanishes while never having been bad must stay uninteresting. + // Vanished is a deliberate no-op: freeze badOpen/preexisting as-is, so a + // vanish while bad stays bad (never reading as a recovery). for _, key := range p.Vanished { tl := get(key, nil) tl.seen = true @@ -325,10 +290,8 @@ func classifyRule(def Definition, polls []Poll, from, windowEnd time.Time, badSt } } - // Multiple instances can appear for the first time within the same poll, - // and map iteration order is nondeterministic; sort so this pure - // function's Violations/BadFor output is stable across runs given the - // same input, like log.go sorts Cleared/Vanished for the same reason. + // Map iteration order is nondeterministic; sort so Violations/BadFor output + // is stable for a given input (like log.go sorts Cleared/Vanished). slices.Sort(order) var ( @@ -357,9 +320,8 @@ func classifyRule(def Definition, polls []Poll, from, windowEnd time.Time, badSt instOutcome = OutcomePersistentlyBad } default: - // A genuinely new onset always fails, whether or not it later - // clears within the window: only a PREEXISTING condition earns - // the benefit of `recovered`. + // A genuinely new onset fails whether or not it clears in-window; + // only a preexisting condition earns `recovered`. instOutcome = OutcomeNewlyBad } @@ -529,12 +491,9 @@ func decide(h Header, polls []Poll, sentinel *time.Time, defs []Definition, if s < 0 { s = -s } - // The bound travels with ITS OWN poll's skew, never the largest bound - // seen overall (Result.ClockSkewBound's doc comment) — so it is only - // ever overwritten in lockstep with ClockSkew, on the same poll. >= - // rather than > on top of skewSeen: a strict > would never assign the - // bound at all when every poll's skew is exactly 0, understating the - // real measurement uncertainty as an unearned "bound ±0s". + // The bound travels with its own poll's skew (see Result.ClockSkewBound), + // overwritten in lockstep. >= rather than > so a bound is still assigned + // when every poll's skew is exactly 0. if !skewSeen || s > result.ClockSkew { result.ClockSkew = s result.ClockSkewBound = p.SkewBound() @@ -556,13 +515,10 @@ func decide(h Header, polls []Poll, sentinel *time.Time, defs []Definition, unobservableNames []string ) - // `skipped` is decided from the header, never from defs. defs are - // resolved after the window has closed, so Definition.IsPaused describes - // the present; Header.pausedAtStart describes the moment the recording - // opened, which is the only moment "paused before the window opened" can - // mean. Reading the late definition instead let a rule that fired and was - // then paused report as skipped, with its firing never classified — and - // under AllowPaused that was a pass. + // `skipped` is decided from the header, never from defs: defs are resolved + // after the window closed, so Definition.IsPaused describes the present, + // while Header.pausedAtStart describes the window open — the only moment + // "paused before the window opened" can mean. pausedAtStart := h.pausedAtStart() for _, def := range defs { @@ -613,14 +569,9 @@ func decide(h Header, polls []Poll, sentinel *time.Time, defs []Definition, }) } - // MinObserved defaults to len(defs) after duplicate names collapse - // (already done by Resolve before decide ever sees defs). Skipped rules - // count against it unless AllowPaused says otherwise. A shortfall counts - // toward exit 1, never exit 2 — decide never returns an error for this — - // and it has to surface through Violations like any other fail reason, so - // a shortfall always produces at least one, even when no rule is paused at - // all (an operator-supplied MinObserved that simply exceeds what could ever - // be resolved). + // MinObserved defaults to len(defs) (post-collapse). A shortfall counts + // toward exit 1, never exit 2, and surfaces through Violations — so it + // always produces at least one, even when no rule is paused. counted := watchedCount var attributable []Definition if pol.AllowPaused { diff --git a/grafana-alertcheck/internal/gate/coverage.go b/grafana-alertcheck/internal/gate/coverage.go index 5c82458ee..527720bfe 100644 --- a/grafana-alertcheck/internal/gate/coverage.go +++ b/grafana-alertcheck/internal/gate/coverage.go @@ -8,25 +8,15 @@ import ( // keepLastReason is the instance Reason that check 9 watches for. const keepLastReason = "KeepLast" -// Two things this file deliberately leaves to its callers: -// -// - "from more than fromFutureTolerance ahead of the runner's clock" is a -// hard error, but it is once-per-run input validation rather than a -// per-rule coverage check, and this function has no error return. -// Config.validate (check.go) applies it; check 2 below owns only the -// "from < StartedAt" half. -// - A rule paused before the window opened is never scheduled or polled, so -// it would reach this function with zero polls and read as one large -// heartbeat_gap rather than as skipped (pinned by -// TestProveCoverage_SkippedRuleWithZeroPollsPinnedAsHeartbeatGap). decide -// returns before it ever calls proveCoverage for such a rule, reading -// skipped from the log header (Header.pausedAtStart) and NOT from -// Definition.IsPaused — the definitions are re-resolved after the window -// closed, so they cannot answer what was paused when it opened. +// Two things this file leaves to its callers: the "from too far ahead" bound is +// Config.validate's once-per-run input validation (check 2 owns only the +// "from < StartedAt" half), and a rule paused at the window open never reaches +// proveCoverage — decide reads `skipped` from Header.pausedAtStart first, so a +// paused rule's zero polls read as skipped, not as one large heartbeat gap. // UnobservableReason names why proveCoverage could not prove a rule's window. -// It is machine-readable — this reaches the JSON output, so it is a published -// vocabulary like Outcome; prose belongs in Notes. +// It reaches the JSON output, so it is a published vocabulary like Outcome; +// prose belongs in Notes. type UnobservableReason string const ( @@ -61,31 +51,18 @@ type CoverageResult struct { BlindFor time.Duration } -// proveCoverage applies the nine coverage checks to one rule's polls and is -// PURE: no HTTP, no files, no clock reads — everything it needs arrives as an -// argument, which is what lets its tests build []Poll literals instead of a -// fixture server. -// -// polls need not be pre-filtered to this rule: proveCoverage selects by -// def.UID itself, exactly as Reduce selects by UID rather than by title — a -// caller handing it a whole log's polls must not have to pre-filter to get a -// correct answer. -// -// Every check always runs, even once an earlier one has already set -// Unobservable: LargestGap and the notes are diagnostics an operator reads on -// exit 2 regardless of which check actually failed. Reason names the FIRST -// check, in the order below, that failed; a later failure still adds its own -// Note. +// proveCoverage applies the nine coverage checks to one rule's polls. PURE: no +// HTTP, no files, no clock reads — everything arrives as an argument. polls need +// not be pre-filtered to this rule (selection is by def.UID). Every check runs +// even after Unobservable is set, so LargestGap and the notes are complete on +// exit 2; Reason names only the FIRST check that failed. func proveCoverage(h Header, polls []Poll, sentinel *time.Time, t ruleTimings, def Definition, from, to time.Time, grace time.Duration) CoverageResult { windowEnd := to.Add(grace) - // pollsForRule (classify.go) is the single filter+sort implementation for - // "select one rule's polls, stably ordered by GrafanaNow" — proveCoverage - // and classifyRule must never carry two independent copies of this - // selection, or one drifting from the other becomes exactly the kind of - // silent membership mismatch this file's checks exist to prevent. + // pollsForRule (classify.go) is the single filter+sort implementation; this + // and classifyRule must not carry two independent copies. rulePolls := pollsForRule(polls, def.UID) var res CoverageResult @@ -119,10 +96,7 @@ func proveCoverage(h Header, polls []Poll, sentinel *time.Time, t ruleTimings, d "requested from %s is before recording started at %s", from.Format(time.RFC3339), h.StartedAt.Format(time.RFC3339))) } - // Filtered once, here, and threaded through every remaining check — - // ruleHeartbeatGap included — rather than re-filtered per check: two - // independent filters over the same polls would only invite one of them - // drifting from the other's membership test. + // Filtered once and threaded through every remaining check. inWindow := inWindowPolls(rulePolls, from, windowEnd) // Check 3 — heartbeat continuity. Data at both ends with a hole in between @@ -144,30 +118,19 @@ func proveCoverage(h Header, polls []Poll, sentinel *time.Time, t ruleTimings, d } } - // Check 5 — health=="nodata". Never fatal here: 96% of the fleet runs - // no_data_state:OK, so treating this as fatal by default would block - // nearly every healthy deploy in an idle environment. Escalating it under - // Policy.NodataIsUnobservable is decide's job, applied directly against - // the raw polls — this pure function has no Policy to consult and must not - // invent one. + // Check 5 — health=="nodata". Never fatal here (most of the fleet runs + // no_data_state:OK, so it would block healthy idle deploys). Escalating + // under Policy.NodataIsUnobservable is decide's job, since this pure + // function has no Policy to consult. if _, sawAny := longestHealthRun(inWindow, "nodata"); sawAny { res.Notes = append(res.Notes, fmt.Sprintf("rule %q: health=nodata observed (not fatal; see --nodata-is-unobservable)", def.Title)) } - // Check 6 — liveness. Absolute only, per poll: GrafanaNow and - // LastEvaluation are both Grafana-domain reads off the SAME response, so - // this is a same-domain comparison and uses raw values — never a delta - // against a previous poll, which reports stale on ~half the polls of a - // perfectly healthy rule (polling runs at intervalSeconds/2). - // - // Skipped only for a poll whose own flags SAY there is nothing to check: - // IsPaused (a zero LastEvaluation is legal only while paused; check 7 is - // its detector) or !Found (no rule, no evaluation; check 8 is its - // detector). Deliberately NOT skipped merely because LastEvaluation is - // zero: ReadLog does no field validation, so a corrupted or hand-edited - // log line can claim found:true, is_paused:false and still carry a zero - // LastEvaluation, and that combination must read as maximally stale - // rather than being silently waved through. + // Check 6 — liveness. Same-domain (GrafanaNow and LastEvaluation are from + // the SAME response), so raw values — never a delta against a previous + // poll, which reports stale ~half the polls of a healthy rule. Skipped only + // for IsPaused (check 7) or !Found (check 8); a zero LastEvaluation on a + // found, unpaused poll is treated as maximally stale, not waved through. var staleCount int var worstStale time.Duration var worstStaleAt time.Time @@ -236,12 +199,10 @@ func proveCoverage(h Header, polls []Poll, sentinel *time.Time, t ruleTimings, d fail(ReasonRuleAbsent, fmt.Sprintf("state endpoint returned no rule on %d poll(s), first at %s", absentCount, absentAt.Format(time.RFC3339))) } - // Check 9 — KeepLast. Two distinct notes, both non-fatal: - // - // DECLARED: the rule's own no_data_state/exec_err_state is configured as - // KeepLast — a standing blind spot whether or not it is ever exercised - // during this particular window. This reads def, not polls, so it fires - // exactly once regardless of poll content. + // Check 9 — KeepLast. Two non-fatal notes: DECLARED (the rule is configured + // with no_data_state/exec_err_state=KeepLast, read from def so it fires once), + // and OBSERVED (an instance reported KeepLast in-window; Reasons keys can be + // comma-joined, so membership via reasonsContain, never a literal index). nds, ees := def.NoDataState, def.ExecErrState for _, lr := range h.Rules { if lr.UID == def.UID { @@ -253,10 +214,6 @@ func proveCoverage(h Header, polls []Poll, sentinel *time.Time, t ruleTimings, d res.Notes = append(res.Notes, fmt.Sprintf( "rule %q: configured with no_data_state/exec_err_state=KeepLast — a stale state can continue past a real fault", def.Title)) } - // OBSERVED: an instance actually reported the KeepLast reason during the - // window. It surfaces only as an instance Reason, and Reasons keys can be - // comma-joined composites, so membership (reasonsContain) is required — - // indexing "KeepLast" directly would miss "KeepLast, MissingSeries". for _, p := range inWindow { if reasonsContain(p.Reasons, keepLastReason) { res.Notes = append(res.Notes, fmt.Sprintf( @@ -269,17 +226,11 @@ func proveCoverage(h Header, polls []Poll, sentinel *time.Time, t ruleTimings, d return res } -// inWindowPolls filters polls to those inside [from, windowEnd] using the -// CROSS-DOMAIN membership test: each poll's Grafana-domain GrafanaNow is -// translated to the runner domain by its OWN skew, and its own skew bound is -// the membership tolerance, so a poll that is genuinely inside the window is -// never excluded by ordinary clock imprecision. -// -// Everything downstream of this filter (health runs, liveness, pause, absence) -// reads the poll's raw fields: GrafanaNow paired with LastEvaluation on the -// SAME response, or one poll's GrafanaNow against the next's, are same-domain -// comparisons and need no translation. Only window membership and check 3's -// two boundary segments cross domains. +// inWindowPolls filters to polls inside [from, windowEnd] via the cross-domain +// membership test: each GrafanaNow is translated to the runner domain by its +// own skew, widened by its skew bound, so clock imprecision never excludes a +// genuinely in-window poll. Everything downstream reads same-domain raw fields; +// only this filter and check 3's boundary segments cross domains. func inWindowPolls(polls []Poll, from, windowEnd time.Time) []Poll { var out []Poll for _, p := range polls { @@ -294,21 +245,14 @@ func inWindowPolls(polls []Poll, from, windowEnd time.Time) []Poll { } // ruleHeartbeatGap finds the largest unobserved span inside [from, windowEnd], -// including the two boundary segments — which is why "data at both ends with a -// hole in the middle" still fails: the segment between the polls just inside -// each edge is exactly what this measures. in must already be filtered to this -// window (inWindowPolls) and sorted by GrafanaNow — proveCoverage computes that -// filter once and threads it through every check, this one included, rather -// than each check re-filtering. +// including the two boundary segments — which is why data at both ends with a +// hole in the middle still fails. in must be filtered (inWindowPolls) and +// sorted by GrafanaNow. // -// The two boundary segments compare a Grafana-domain poll time against the -// runner-domain from/windowEnd, so each is translated by its own poll's skew -// AND widened by that same poll's skew bound — on the side that makes the -// segment larger, never smaller, so an uncertain boundary reads as at least as -// big a gap as it might really be. Understating it by up to the bound would be -// fail-open. The spacing BETWEEN consecutive polls compares two Grafana-domain -// reads to each other — same domain — and uses the raw GrafanaNow difference, -// no bound needed. +// Boundary segments are cross-domain, so each translated poll time is widened +// by its skew bound on the side that makes the gap LARGER (never smaller — an +// uncertain boundary must read as at least as big a gap as it might be). +// Consecutive-poll spacing is same-domain and uses the raw GrafanaNow diff. func ruleHeartbeatGap(in []Poll, from, windowEnd time.Time) (largestGap time.Duration, largestGapAt time.Time) { if len(in) == 0 { return windowEnd.Sub(from), from @@ -332,15 +276,10 @@ func ruleHeartbeatGap(in []Poll, from, windowEnd time.Time) (largestGap time.Dur return largestGap, largestGapAt } -// longestHealthRun returns the longest contiguous wall-clock span during which -// polls — already sorted by GrafanaNow, same-domain spacing — read the given -// rule-level Health, and whether any poll matched it at all. -// -// It detects the span as it accumulates rather than waiting for the run to -// end, so an open-ended run that is still failing at the last poll in the -// window is measured correctly without needing data past the window: waiting -// for the run to "end" would have to assume the best case about what happens -// next, which is exactly what this gate must not do. +// longestHealthRun returns the longest contiguous span of polls reading the +// given rule-level Health, measured incrementally so a run still failing at the +// last in-window poll is measured correctly without assuming anything past the +// window. func longestHealthRun(polls []Poll, health string) (longest time.Duration, sawAny bool) { var runStart time.Time for _, p := range polls { diff --git a/grafana-alertcheck/internal/gate/log.go b/grafana-alertcheck/internal/gate/log.go index 2fdb71d5f..f8f52d01f 100644 --- a/grafana-alertcheck/internal/gate/log.go +++ b/grafana-alertcheck/internal/gate/log.go @@ -47,20 +47,15 @@ type LoggedRule struct { // definitions from the ruler API. ForSeconds float64 `json:"for_seconds"` IntervalSeconds int `json:"interval_seconds"` - // IsPaused is NOT forensic, and is the second load-bearing field here - // beside PollEverySeconds. It is the pause state at record start, which is - // the only moment `skipped` can honestly mean, and decide reads it through - // Header.pausedAtStart rather than reading Definition.IsPaused off a ruler - // read taken after the window had already closed. See that method for what - // goes wrong the other way. + // IsPaused is load-bearing (beside PollEverySeconds): the pause state at + // record start, the only moment `skipped` can honestly mean. decide reads + // it via Header.pausedAtStart, never a ruler read taken after the window. IsPaused bool `json:"is_paused"` NoDataState string `json:"no_data_state"` ExecErrState string `json:"exec_err_state"` - // PollEverySeconds is the cadence this recording ACTUALLY used, after any - // --poll-interval override. Load-bearing, not forensic: check derives - // maxGap from it and never re-derives it from the definitions. Getting - // that wrong is fail-open in the faster-override direction — a real - // recorder gap would pass silently. + // PollEverySeconds is the cadence this recording ACTUALLY used. Load-bearing: + // check derives maxGap from it, never from the definitions — getting that + // wrong is fail-open in the faster-override direction. PollEverySeconds float64 `json:"poll_every_seconds"` } @@ -128,15 +123,11 @@ type Poll struct { IsPaused bool `json:"is_paused"` Histogram map[string]int `json:"histogram,omitempty"` // written, never analysed // Reasons counts this poll's non-empty instance reasons, e.g. - // {"NoData":1091,"Error":14}; nil when none. Reporting-only, and the ONLY - // place composite states stay visible: they are canonical normal (so they - // are dropped from Abnormal) and `totals` never carries composite keys. - // - // The KEYS are raw reason strings and can be comma-joined composites - // ("KeepLast, MissingSeries") — newer Grafana versions join several - // reasons into one. So any consumer, the coverage proof's KeepLast note - // included, must test membership across the keys with reasonNames and must - // NEVER index a literal key: reasons["KeepLast"] misses every composite. + // {"NoData":1091,"Error":14}; nil when none. Reporting-only, and the only + // place composite states stay visible (they are canonical normal, dropped + // from Abnormal). Keys are raw reason strings and can be comma-joined + // composites ("KeepLast, MissingSeries"), so consumers must test membership + // via reasonNames and never index a literal key. Reasons map[string]int `json:"reasons,omitempty"` // Abnormal holds the instances whose CANONICAL state is not normal. // "Normal (NoData)" and "Normal (Error)" are canonical normal and are @@ -290,16 +281,10 @@ func (r *Reducer) seedFrom(polls []Poll) { } } -// stateRuleByUID picks one rule out of a state-endpoint response BY UID, and -// nil means the response is an authoritative "the rule is absent". -// -// Never by title: the ?rule_name= filter is a title filter, and a filtered -// response can carry several rules sharing one title (the known 2-way -// collision), so picking the first would silently watch the wrong rule. This -// is the single implementation of that selection for the package — Reduce -// above and the drain wait (check.go) both call it, for the same reason -// pollsForRule (classify.go) is shared between proveCoverage and classifyRule: -// two copies of a membership test are two chances for one to drift. +// stateRuleByUID picks one rule out of a state response BY UID (nil = the +// authoritative "rule absent"). Never by title: the ?rule_name= filter is a +// title filter and can return several rules sharing a title. The single +// selection for the package — Reduce and the drain wait both use it. func stateRuleByUID(rules []StateRule, uid string) *StateRule { for i := range rules { if rules[i].UID == uid { @@ -322,18 +307,11 @@ func reasonNames(reason, want string) bool { } // VerifyNormalInstancesVisible checks, on a first observation, that the state -// endpoint really does return normal instances and not only the abnormal ones. -// If it ever stops doing so, the reduction's "keep the non-normal instances" -// becomes "keep everything the API happened to send" and the transition markers -// lose their ground truth — a silent fail-open. So this is verified at start, -// never assumed. -// -// The counts are summed over every totals key whose LOWERCASED name is -// "normal" or "inactive". Never index one literal key: the captured -// vocabulary is mixed across rules ({"alerting":445,"normal":2004} on one, -// {"firing":2,"inactive":363} on another) and its case has already drifted -// from the original recon. Composite states never appear in totals — Grafana -// counts a "Normal (NoData)" instance under normal. +// endpoint really returns normal instances: if it ever stops, the reduction's +// "keep the non-normal" becomes "keep everything the API sent", a silent +// fail-open in the transition markers. Counts are summed over every totals key +// whose lowercased name is "normal" or "inactive" — never a literal key, since +// the vocabulary is mixed and its case has drifted. func VerifyNormalInstancesVisible(rules []StateRule) error { for _, r := range rules { var claimed int @@ -453,18 +431,12 @@ func (w *Writer) WritePoll(p Poll) error { return nil } -// Stop finishes recording in a fixed order that must not be rearranged: let -// the in-flight write finish (the mutex), append the stopped sentinel, fsync, -// then release. Any other order can leave a log whose last durable byte is a -// sentinel that was never actually preceded by the polls it vouches for. -// -// Stop writes the sentinel with the recorder's OWN stop time and makes no -// comparison against `to` — watch never knows `to` or the transition grace. -// check does that comparison, after this writer has exited. -// -// Calling Stop twice is a no-op: watch reaches it from both a signal handler -// and a defer, and a second sentinel would be indistinguishable from a second -// writer. +// Stop finishes recording in a fixed order that must not be rearranged: let the +// in-flight write finish, append the sentinel, fsync, release — any other order +// can leave a sentinel that was never preceded by the polls it vouches for. +// The sentinel uses the writer's OWN stop time; check does the `to` comparison +// after this has exited. Calling Stop twice is a no-op (watch reaches it from a +// signal handler and a defer). func (w *Writer) Stop() error { w.mu.Lock() defer w.mu.Unlock() @@ -505,18 +477,11 @@ func (w *Writer) Close() error { return nil } -// ReadLogHeader reads ONLY line 1 and is the one read of a log that a writer -// may still hold. That is safe for exactly one line and for no other: the -// header is written once, by watch's parent, before any child appends a byte, -// and the file is opened O_APPEND and never O_TRUNC, so line 1 is complete and -// immutable for the whole life of the recording. -// -// It exists so check can fail closed EARLY: the log's identity, the rule set -// and the cadences are all knowable at the start, and discovering a wrong URL -// or an unresolvable rule after a ten-minute wait helps nobody. It is advisory -// only — the authoritative read is still ReadLog, once, after the writer has -// exited, and check re-validates the identity against that header rather than -// trusting this one. +// ReadLogHeader reads ONLY line 1 — the one read safe while a writer may still +// hold the log. The header is written once by watch's parent before any child +// appends a byte, so line 1 is immutable. It lets check fail closed EARLY on a +// wrong URL or unresolvable rule; it is advisory only, and the authoritative +// identity read is still ReadLog after the writer exits. func ReadLogHeader(path string) (Header, error) { f, err := os.Open(path) if err != nil { @@ -555,11 +520,9 @@ func ReadLogHeader(path string) (Header, error) { // append to can only produce a shorter window than the one that was recorded. // // The parse rules are deliberately the crudest possible: the header must be -// line 1 with a matching schema version, and ANY unparseable line — -// including the last one, and including a last line that follows a sentinel — -// is an error, full stop. No heuristics, no discarding an untidy tail: a -// truncated log is evidence that something killed the recorder, which is -// exactly what must not pass. +// line 1 with a matching schema version, and ANY unparseable line — including +// the last, or one after a sentinel — is an error, full stop. A truncated log +// is evidence something killed the recorder, which must not pass. func ReadLog(path string) (Header, []Poll, *time.Time, error) { b, err := os.ReadFile(path) if err != nil { diff --git a/grafana-alertcheck/internal/gate/schedule.go b/grafana-alertcheck/internal/gate/schedule.go index 4541317e5..f2b9bd759 100644 --- a/grafana-alertcheck/internal/gate/schedule.go +++ b/grafana-alertcheck/internal/gate/schedule.go @@ -8,26 +8,22 @@ import ( "time" ) -// SkewHardLimit is the largest clock skew between this runner and Grafana that -// a run tolerates before it errors out. Exported so the CLI can report it -// verbatim next to a measured skew instead of keeping its own mirrored copy. +// SkewHardLimit is the largest runner↔Grafana clock skew a run tolerates. +// Exported so the CLI reports it verbatim next to a measured skew. const SkewHardLimit = 60 * time.Second -// fromFutureTolerance is how far ahead of the runner's own clock a supplied -// `from` may sit before check refuses it: the same 60s as SkewHardLimit, -// because the only legitimate reason for a `from` in the future is clock -// disagreement between the deploy step and the check step, and that is bounded -// by the same figure. It is once-per-run input validation, not a per-rule -// coverage check, so Check applies it and proveCoverage does not. +// fromFutureTolerance is how far ahead of the runner's clock a `from` may sit +// before check refuses it — the same 60s as SkewHardLimit, since a future `from` +// can only be clock disagreement. Once-per-run input validation, not a coverage +// check, so Check applies it and proveCoverage does not. const fromFutureTolerance = 60 * time.Second -// minDrainTimeout is the floor on drainTimeout, which is otherwise -// 2 x max(intervalSeconds). Without the floor, a fleet of very tight rules -// would derive a drainTimeout too short to let a healthy in-flight poll land. +// minDrainTimeout floors drainTimeout (otherwise 2 × max intervalSeconds) so a +// fleet of tight rules still lets a healthy in-flight poll land. const minDrainTimeout = 2 * time.Minute -// graceWarnFraction is the share of the requested window above which -// transitionGrace is worth warning about. +// graceWarnFraction is the share of the window above which transitionGrace is +// worth warning about. const graceWarnFraction = 0.25 // ruleTimings groups the per-rule thresholds derived from a rule's poll @@ -52,12 +48,9 @@ type globalTimings struct { drainTimeout time.Duration } -// newRuleTimings derives one rule's thresholds from its fully-resolved poll -// cadence and its evaluation interval. pollEvery arrives already resolved for -// the caller's mode — the default, the operator's --poll-interval override, or -// (in log mode) the cadence recorded in the log header. Deriving pollEvery -// inline here, instead of accepting it as an input, would let a caller in the -// wrong mode compute maxGap against the wrong authority. +// newRuleTimings derives one rule's thresholds from its resolved cadence and +// evaluation interval. pollEvery is an input — already resolved for the +// caller's mode — so no caller can compute maxGap against the wrong authority. func newRuleTimings(pollEvery time.Duration, intervalSeconds int) ruleTimings { interval := time.Duration(intervalSeconds) * time.Second maxGap := 2 * pollEvery @@ -76,14 +69,10 @@ func defaultPollEvery(intervalSeconds int) time.Duration { return time.Duration(intervalSeconds) * time.Second / 2 } -// DeriveTimings computes every resolved rule's ruleTimings, keyed by UID, plus -// the shared globalTimings, from resolved definitions and watch's optional -// --poll-interval override (0 = no override: use each rule's default of half -// its own interval). A supplied override is used verbatim for every rule and is -// never clamped down to the default even when it exceeds intervalSeconds/2 — -// that case is reported back as a note, not silently corrected or refused, -// because clamping would defeat the one knob an operator has for making a tight -// schedule fit. +// DeriveTimings computes every resolved rule's ruleTimings (keyed by UID) plus +// the shared globalTimings. A non-zero override is used verbatim for every rule +// and never clamped to the default — an override above intervalSeconds/2 widens +// maxGap and is reported as a note, not corrected. func DeriveTimings(defs []Definition, override time.Duration) (rules map[string]ruleTimings, global globalTimings, notes []string) { rules = make(map[string]ruleTimings, len(defs)) for _, d := range defs { @@ -99,10 +88,8 @@ func DeriveTimings(defs []Definition, override time.Duration) (rules map[string] } rules[d.UID] = newRuleTimings(pollEvery, d.IntervalSeconds) } - // In this mode the defs ARE the start-of-step snapshot — watch resolves - // them before it detaches, and single-step check before its first - // observation — so they can answer what was paused when the window opened. - // Only the log-mode counterpart below has to look elsewhere. + // In this mode defs ARE the start-of-step snapshot, so they answer what was + // paused at the window open; only the log-mode counterpart uses the header. return rules, deriveGlobalTimings(defs, pausedSet(defs)), notes } @@ -117,39 +104,19 @@ func pausedSet(defs []Definition) map[string]bool { return paused } -// DeriveTimingsFromLog is DeriveTimings' log-mode counterpart, and having one -// authority for the cadence is the whole reason it exists as a separate -// function. pollEvery comes from the header — the cadence the recording -// ACTUALLY used, after any --poll-interval override — and maxGap and -// healthGrace follow from it. Re-deriving pollEvery from defs here would -// compare gaps recorded at the override cadence against thresholds computed -// from the default: exit 2 on a clean window when the override was slower, and, -// worse, a real recorder gap passing silently when it was faster. +// DeriveTimingsFromLog is DeriveTimings' log-mode counterpart: pollEvery comes +// from the header (the cadence actually used), not the definitions — re-deriving +// it here would compare recorded gaps against default-cadence thresholds, an +// exit 2 on a clean window (slower override) or a silently passing recorder gap +// (faster override). evalStaleAfter still comes from defs (2 × intervalSeconds). // -// evalStaleAfter still comes from defs (2 x intervalSeconds): it is a property -// of the rule's own evaluation cadence and is unaffected by how often the gate -// polled. +// Three header shapes are hard errors rather than a best-effort derivation, +// because each would silently widen a threshold: a rule with no matching +// definition, a non-positive recorded cadence, and a UID listed twice +// (last-one-wins would widen maxGap on a corrupt log). // -// Three shapes of header are errors rather than a best-effort derivation, -// because each one would otherwise widen a threshold silently: -// -// - a rule with no matching definition — a log that names a rule nobody can -// resolve cannot have that rule's coverage proved; -// - a non-positive recorded cadence — a log that cannot say how often it was -// written cannot have maxGap derived, and defaulting the cadence would -// prove a window that was never observed; -// - the same UID twice — last-one-wins would take whichever cadence happened -// to be written last, and a slower duplicate widens maxGap. That is a -// fail-open reachable through nothing but log corruption. -// -// It checks only the header-to-defs direction. The opposite direction — a -// resolved definition absent from the header — is NOT this function's to -// judge: it belongs to Check's log-identity validation, the only caller that -// knows both sets and can name the mismatch. Without that check a definition -// simply gets no timings entry, and a downstream lookup would read a zero -// maxGap: fail-closed (every gap exceeds it) but silent, so Check must reject -// the set mismatch by name rather than let a rule fail for an unexplained -// reason. +// It checks only the header-to-defs direction. A definition absent from the +// header is Check's log-identity validation to judge, not this function's. func DeriveTimingsFromLog(h Header, defs []Definition) (rules map[string]ruleTimings, global globalTimings, err error) { byUID := make(map[string]Definition, len(defs)) for _, d := range defs { @@ -180,23 +147,15 @@ func DeriveTimingsFromLog(h Header, defs []Definition) (rules map[string]ruleTim return rules, deriveGlobalTimings(defs, h.pausedAtStart()), nil } -// deriveGlobalTimings computes transitionGrace and drainTimeout over defs. +// deriveGlobalTimings computes transitionGrace and drainTimeout over defs. A +// rule skipped at the window open is excluded from the transitionGrace max (its +// `for` can never fire in-window); drainTimeout runs over every resolved rule. // -// A rule paused before the window opened — skipped — is excluded from the -// transitionGrace max: its `for` value can never fire during the window, so -// counting it would only inflate the wait past what any watched rule actually -// needs. drainTimeout carries no such exclusion and still runs over every -// resolved rule. -// -// "Before the window opened" is the whole content of that exclusion, so the -// authority is pausedAtStart and NEVER Definition.IsPaused: in log mode the -// definitions are re-resolved after the window closed. Reading them instead -// was a fail-open, and a quiet one. transitionGrace is what lets a condition -// arising just before `to` be seen when it surfaces at to + `for`, and -// windowEnd is BOTH the classification bound and the collection deadline — so -// a rule somebody paused after `to` dropped out of the max, the grace -// collapsed, the surfacing poll was never even recorded, and the run reported -// clean. With one watched rule the shrink is total. +// The exclusion authority is pausedAtStart, never Definition.IsPaused: log-mode +// defs are re-resolved after the window closed. Reading the late definitions +// was a quiet fail-open — a rule paused after `to` would drop out of the max, +// collapse the grace past windowEnd (the classification bound AND collection +// deadline), and pass a window the surfacing poll was never recorded for. func deriveGlobalTimings(defs []Definition, pausedAtStart map[string]bool) globalTimings { var g globalTimings var maxInterval time.Duration @@ -225,17 +184,11 @@ type Scheduler struct { every map[string]time.Duration } -// NewScheduler builds a Scheduler over per-rule cadences (keyed by UID), -// staggering each rule's initial next-due time across [0, pollEvery) so the -// fleet does not start phase-aligned. The burst bound CheckBudget enforces -// depends on that: an already-staggered fleet only re-aligns by chance, -// briefly, not by construction. -// -// It takes cadences rather than whole ruleTimings on purpose: a scheduler -// decides when to poll and nothing else, so it must not be handed maxGap, -// healthGrace or evalStaleAfter. Those are coverage thresholds, they are -// applied by the pure layer at classification time, and the recorder that -// drives this scheduler never applies them at all. +// NewScheduler builds a Scheduler over per-rule cadences, staggering each +// rule's initial next-due time across [0, pollEvery) so a phase-aligned fleet +// (which would void CheckBudget's burst bound) never arises by construction. +// It takes cadences, not ruleTimings: a scheduler only decides when to poll and +// must not be handed coverage thresholds it never applies. func NewScheduler(every map[string]time.Duration, now time.Time) *Scheduler { s := &Scheduler{ next: make(map[string]time.Time, len(every)), @@ -252,13 +205,10 @@ func NewScheduler(every map[string]time.Duration, now time.Time) *Scheduler { return s } -// Due returns the UIDs whose next-due time has arrived, earliest-due-first. -// Ties (equal next-due time) break by tightest cadence first: the burst bound -// assumes a newly-due tight rule waits at most for one in-flight request, which -// only holds if a simultaneous batch serves the tightest rule ahead of slacker -// ones. A tie-break that instead followed map iteration order would silently -// void that assumption — nothing else would fail until a phase-aligned fleet -// opened a mid-run gap in production. +// Due returns the due UIDs, earliest-due-first. Ties break by tightest cadence +// first: the burst bound assumes a newly-due tight rule waits at most one +// in-flight request, which only holds if a simultaneous batch serves the +// tightest rule first. A map-order tie-break would silently void that. func (s *Scheduler) Due(now time.Time) []string { var due []string for uid, t := range s.next { @@ -293,11 +243,9 @@ func (s *Scheduler) Mark(uid string, now time.Time) error { return nil } -// earliestDue returns the earliest scheduled next-due time, and false when the -// scheduler holds no rules at all. The recorder's loop waits exactly that long -// instead of waking on a fixed tick: a fixed tick either polls a slack rule -// early — spending request budget the schedule already accounted for — or wakes -// too late for the tightest rule and opens a gap inside its own maxGap. +// earliestDue returns the earliest next-due time (false when empty). The loop +// waits exactly that long instead of a fixed tick, which would poll slack rules +// early (wasting budget) or wake late for the tightest rule (opening a gap). func (s *Scheduler) earliestDue() (time.Time, bool) { var earliest time.Time ok := false @@ -310,21 +258,18 @@ func (s *Scheduler) earliestDue() (time.Time, bool) { return earliest, ok } -// CheckBudget proves at start that a fully resolved schedule can actually be -// served. t and measured are both keyed by rule UID, and measured must carry -// every UID in t: a rule this run never measured cannot have its budget -// proved, and a silent zero-duration default would be exactly the kind of -// pass-on-an-unproven-window bug this check exists to catch. It fails when any -// of three conditions holds: +// CheckBudget proves at start that a fully resolved schedule can be served. t +// and measured are keyed by UID, and measured must carry every UID in t (a rule +// never measured cannot have its budget proved). It fails on any of three +// conditions: // -// - utilization: the long-run request rate exceeds what concurrency serves; -// - a single rule's own request cannot fit inside its own cadence; -// - the burst bound: the slowest measured request is slower than the fleet's -// tightest cadence, which — even under earliest-due-first ordering — can -// open a mid-run gap bigger than that rule's maxGap. +// - utilization — the long-run request rate exceeds the concurrency; +// - a single rule's request cannot fit inside its own cadence; +// - the burst bound — the slowest measured request is slower than the fleet's +// tightest cadence, which can open a mid-run gap beyond that rule's maxGap. // -// The message names only the three controls an operator actually has: -// concurrency, poll-interval, and the alert list. +// The message names only the three operator controls: concurrency, +// poll-interval, and the alert list. func CheckBudget(t map[string]ruleTimings, measured map[string]time.Duration, concurrency int) error { if len(t) == 0 { return nil diff --git a/grafana-alertcheck/internal/gate/source.go b/grafana-alertcheck/internal/gate/source.go index 82dcdc392..bebb1839e 100644 --- a/grafana-alertcheck/internal/gate/source.go +++ b/grafana-alertcheck/internal/gate/source.go @@ -44,11 +44,9 @@ type Observation struct { Latency time.Duration // t_send through the full body read — see requestResult.Latency } -// TransportError marks a failure worth retrying: a non-2xx response, a -// network failure, or a body that failed to parse. It is never a deleted rule -// (an authoritative 2xx with no matching rule is not this) and never a clock -// problem (a missing/unparseable Date header or an out-of-bounds skew is a -// hard error instead — see doRequest). Never conflate them. +// TransportError marks a failure worth retrying: a non-2xx response, a network +// failure, or a body that failed to parse. Not a deleted rule (an authoritative +// 2xx) and not a clock problem (a hard error — see doRequest). type TransportError struct { Err error Status int // 0 when the failure never got a status (network/transport failure) @@ -63,14 +61,10 @@ func (e *TransportError) Error() string { func (e *TransportError) Unwrap() error { return e.Err } -// RetryExhaustedError is what retryTransport returns once it gives up after -// too many sequential *TransportError failures. It deliberately does not -// implement Unwrap into the underlying *TransportError: once retries are -// exhausted the result is a hard, terminal failure, and -// errors.AsType[*TransportError] must never re-classify it as retryable. -// Cause is still exposed as a plain field (and folded into Error()'s text) so -// a caller can log or inspect it; it just cannot flow back into the retry -// classification. +// RetryExhaustedError is the hard, terminal failure retryTransport returns once +// it gives up. It deliberately omits Unwrap into *TransportError so +// errors.AsType can never re-classify it as retryable; Cause stays a plain +// field for logging only. type RetryExhaustedError struct { Failures int Cause error @@ -259,27 +253,20 @@ type requestResult struct { Skew time.Duration // serverDate - (t_send+t_headers)/2, signed SkewBound time.Duration // (t_headers-t_send)/2 — RTT/2 to the response headers // Latency spans t_send through the full body read: the budget check needs - // the whole poll's wall time, or a schedule feasibility check that only - // sees header latency goes optimistic — fail-open. It does not include the - // caller's subsequent JSON parse (ParseState/ParseDefinitions run outside - // doRequest); if the budget accounting ever needs parse time folded in too, - // extend here rather than approximating it at the call site. + // the whole poll's wall time (header-only latency would be fail-open). The + // caller's JSON parse runs outside doRequest; extend here if that ever must + // be folded in. Latency time.Duration } -// doRequest performs one HTTP GET and classifies the outcome: a network -// failure, a non-2xx status, or a body-read failure is retryable -// (*TransportError); a missing or unparseable Date header, or a skew beyond -// SkewHardLimit, is a hard error — retrying can never fix either, so neither -// may enter the backoff loop. +// doRequest performs one HTTP GET and classifies the outcome: network failure, +// non-2xx, or body-read failure is retryable (*TransportError); a missing or +// unparseable Date header or a skew beyond SkewHardLimit is a hard error — +// retrying can never fix either, so neither enters the backoff loop. // -// The Date-header/skew check runs for every endpoint this hits, including -// /api/health, and not only the state endpoint whose timestamps the gate -// actually compares. Deliberate: a skewed clock discovered only once RuleState -// starts polling is a skew that has already masked whatever /api/health and -// the ruler read reported; failing closed at the first response catches it -// before any of that is trusted, and every response comes with a Date header -// for free. +// The Date/skew check runs on every endpoint (even /api/health): a skew only +// noticed once RuleState starts polling has already masked earlier reads, so it +// fails closed on the first response. func (s *httpSource) doRequest(ctx context.Context, path string) (requestResult, error) { req, buildErr := http.NewRequestWithContext(ctx, http.MethodGet, s.baseURL+path, nil) if buildErr != nil { @@ -337,12 +324,10 @@ func (s *httpSource) doRequest(ctx context.Context, path string) (requestResult, return requestResult{Body: b, ServerDate: serverDate, Skew: signedSkew, SkewBound: bound, Latency: latency}, nil } -// retryTransport runs fn, retrying with backoff only while it fails with a -// *TransportError — any other error is a hard error and returns immediately, -// never retried. failures counts consecutive *TransportError results; -// exceeding maxFailures gives up with a wrapped hard error. The wait between -// attempts goes through clock.After so a test with a fake Clock never sleeps on -// real time. +// retryTransport runs fn, retrying with backoff only on *TransportError — any +// other error returns immediately. failures counts consecutive *TransportError +// results; exceeding maxFailures gives up with a wrapped hard error. Waits go +// through clock.After so a fake Clock never sleeps real time. func retryTransport[T any](ctx context.Context, clock Clock, maxFailures int, backoffBase, backoffCap time.Duration, fn func() (T, error)) (T, error) { var zero T failures := 0 diff --git a/grafana-alertcheck/internal/gate/watch.go b/grafana-alertcheck/internal/gate/watch.go index 6f4b7fa15..74dd3945b 100644 --- a/grafana-alertcheck/internal/gate/watch.go +++ b/grafana-alertcheck/internal/gate/watch.go @@ -20,19 +20,13 @@ import ( const DaemonChildFlag = "--daemon-child" // ReadyFDFlag names the inherited descriptor the child reports readiness on. -// The parent passes the write end of a pipe as descriptor 3 and waits for one -// byte, so "the recorder is running" is a POSITIVE signal from the child -// itself — it has read the header, taken the log's flock and entered its poll -// loop — and not an assumption drawn from surviving a timer. A timer cannot -// tell a healthy child from one that is about to die on a slow runner, and -// getting that wrong means watch returns success over a recording that never -// happened. +// "Ready" is a POSITIVE byte from the child (header read, flock taken, in its +// poll loop), never a timer heuristic — a timer can't tell a healthy child from +// one about to die on a slow runner. const ReadyFDFlag = "--ready-fd" -// childReadyTimeout bounds that wait. Everything before the signal is local — -// fork, exec, one read of a log holding a header and a handful of polls — so -// the real figure is milliseconds; this is loose enough for a badly overloaded -// runner and still fails closed rather than hanging the pipeline. +// childReadyTimeout bounds that wait. Everything before the signal is local, so +// it is loose enough for an overloaded runner yet still fails closed. const childReadyTimeout = 30 * time.Second // daemonLogTailBytes bounds how much of a dead child's output the parent @@ -41,18 +35,12 @@ const daemonLogTailBytes = 4096 // WatchConfig is the record step's whole input. // -// It has no To field and must never gain one: watch writes the stopped -// sentinel with its OWN stop time and makes no comparison against `to`, which -// only check knows. Passing `to` here would give two components an -// opinion about the same comparison, and the recorder's opinion is the one -// that cannot be trusted — it exits before the grace it would have to wait for. +// It has no To field (and must never gain one): watch writes the sentinel with +// its OWN stop time and makes no `to` comparison — only check knows `to`, and +// the recorder exits before the grace it would have to wait for. // -// It has no States field either, and watch has no --states flag: recording is -// deliberately unfiltered. The reduction keeps every non-normal instance and -// the transition markers key off the same predicate, so neither consults -// States. The payoff is real — because the log is raw evidence, one recording -// can be re-classified under different --states without re-recording — and the -// Header carries no States field for the same reason. +// It has no States field either: recording is unfiltered, so the same raw log +// can be re-classified under different --states without re-recording. type WatchConfig struct { // URL and Token are the connection details. The CLI reads both from the // environment and never from a flag; Token is never logged and never @@ -137,20 +125,11 @@ func (cfg WatchConfig) validate() error { return nil } -// Watch is the record step's parent process. It returns only once the window -// is genuinely being recorded: -// -// version gate -> resolve definitions and names -> derive timings -> -// open the log and write the header -> ONE observation of every non-skipped -// rule -> verify normal instances are visible -> check the schedule budget -> -// detach the child -> wait for the child to report that it is recording -> -// write the pidfile -> return. -// -// The first-observation wait is not a convenience. Returning before it would -// leave the deploy inside [from, first_poll] with no evidence — the exact -// blind interval the record-then-check split exists to remove — and it is -// also what surfaces auth, name-resolution and parse failures BEFORE -// deploy.sh runs rather than ten minutes later. +// Watch is the record step's parent process, returning only once the window is +// genuinely being recorded (version gate, resolve, write header, one +// observation per rule, budget check, then detach and await the child's +// readiness). The first-observation wait is what surfaces auth, name-resolution +// and parse failures before deploy.sh runs, rather than ten minutes later. func Watch(ctx context.Context, cfg WatchConfig) error { cfg = cfg.withDefaults() if err := cfg.validate(); err != nil { @@ -362,13 +341,10 @@ func openRecording(ctx context.Context, cfg WatchConfig, src Source, writer *Wri return nil, err } - // A rule whose DEFINITION says is_paused is skipped: it is not waited for, - // not scheduled and never polled. Waiting for one either hangs forever or - // errors before the deploy, and recording polls for it would report an - // in-window pause (coverage check 7) for a rule that was already paused - // when the window opened — turning a skipped rule's exit 1 into an exit 2. - // The header still names it, with is_paused true, so check reports it as - // skipped from the definitions. + // A rule whose definition says is_paused is skipped (never waited for or + // polled); polling it would report an in-window pause (check 7) for a rule + // already paused at the open. The header still names it, so check reports + // it skipped from the definitions. var active []Definition activeTimings := make(map[string]ruleTimings, len(resolved)) for _, d := range resolved { @@ -425,21 +401,11 @@ func loggedRules(defs []Definition, rt map[string]ruleTimings) []LoggedRule { return out } -// firstObservations takes one observation of every rule in active, verifies -// that normal instances are visible in those very responses, and reduces each -// into the poll record that IS the window's first heartbeat — plus the measured -// latency of each, which is the only honest input to the budget check (a fixed -// estimate is worthless when one rule's payload is ~230x another's). -// -// Both entry paths share it: watch's parent, before it detaches, and -// single-step check's measurement pass, which keeps the polls as evidence -// rather than writing them to a log. Keeping one implementation is the point — -// the instance-visibility verification and the "absent is a warning, not an -// error" rule are exactly the places where two copies would silently drift, and -// a drift in either direction is fail-open. -// -// polls come back in `active` order, so a log written from them is byte-stable -// for a given set of observations. +// firstObservations takes one observation of every active rule, verifies normal +// instances are visible in those very responses, and reduces each into the +// window's first heartbeat, plus measured latency (the only honest budget +// input). Watches parent and single-step check both share it. Polls return in +// `active` order, so a log written from them is byte-stable. func firstObservations(ctx context.Context, src Source, active []Definition, reducer *Reducer, concurrency int, notes io.Writer) ([]Poll, map[string]time.Duration, error) { @@ -486,14 +452,11 @@ func firstObservations(ctx context.Context, src Source, active []Definition, red } // observeAll polls every rule in uids concurrently, bounded by concurrency, -// and returns one Observation per rule that answered. Every rule is polled by -// TITLE (the ?rule_name= filter is a title filter) and selected out of the -// response by UID — a filtered response can carry several rules sharing one -// title. -// -// It returns the successful observations alongside the first error in UID -// order, so a caller that wants to keep the good heartbeats can, and the error -// message is the same on every run. +// returning one Observation per rule that answered. Each rule is polled by +// TITLE (the ?rule_name= filter is a title filter) and selected by UID — a +// filtered response can carry several rules sharing a title. Returns the +// successes alongside the first error in UID order, so a caller can keep the +// good heartbeats. func observeAll(ctx context.Context, src Source, titles map[string]string, uids []string, concurrency int) (map[string]Observation, error) { if concurrency < 1 { concurrency = 1 @@ -641,15 +604,10 @@ func reportReady(fd int) error { } // childSchedule derives what the child polls, and how often, from the header -// alone. The cadence comes from PollEverySeconds — the cadence the recording -// actually uses — and is never re-derived from the rule's evaluation interval, -// which would be a second authority for the same value and is fail-open in the -// faster-override direction. Paused rules are excluded here for the same reason -// the parent never polls them. -// -// It returns cadences and nothing else. maxGap, healthGrace and evalStaleAfter -// are coverage thresholds applied by the pure layer at classification time, so -// the recorder must not carry them: it would only be able to misuse them. +// alone. Cadence comes from PollEverySeconds (the cadence actually used), never +// re-derived from the evaluation interval; paused rules are excluded. It +// returns cadences only — the recorder must not carry coverage thresholds it +// has no business applying. func childSchedule(h Header) (titles map[string]string, cadence map[string]time.Duration, err error) { titles = make(map[string]string, len(h.Rules)) cadence = make(map[string]time.Duration, len(h.Rules)) @@ -684,15 +642,13 @@ type watchLoopConfig struct { Clock Clock } -// watchLoop is the child's whole working life: poll the rules that are due, -// reduce each observation to one poll record, append it, and — on a clean stop -// only — finish the log with the stopped sentinel. +// watchLoop is the child's whole working life: poll due rules, reduce each +// observation to a poll record, append it, and — on a clean stop only — finish +// the log with the stopped sentinel. // -// The sentinel policy is the load-bearing part. A clean stop (a signal, or -// Until) writes it; a hard error does NOT. A recorder that died must look -// exactly like a coverage gap to check, because it is one — writing a -// sentinel on the way out of a failure would hand check a "recording finished" -// claim about a window that stopped being observed. +// The sentinel policy is load-bearing: a clean stop (signal or Until) writes +// it; a hard error does NOT. A recorder that died must look exactly like a +// coverage gap to check, because it is one. func watchLoop(ctx context.Context, cfg watchLoopConfig) error { sched := NewScheduler(cfg.Cadence, cfg.Clock.Now()) From 75e4fc88d4de9040042c2303820b4aeeb89d6cfa Mon Sep 17 00:00:00 2001 From: Bartek Tofel Date: Mon, 7 Sep 2026 17:18:27 +0200 Subject: [PATCH 4/5] fix: resolve conflict --- grafana-alertcheck/internal/gate/classify.go | 5 ----- 1 file changed, 5 deletions(-) diff --git a/grafana-alertcheck/internal/gate/classify.go b/grafana-alertcheck/internal/gate/classify.go index 9d881334a..d5d9c8e12 100644 --- a/grafana-alertcheck/internal/gate/classify.go +++ b/grafana-alertcheck/internal/gate/classify.go @@ -216,14 +216,9 @@ func classifyRule(def Definition, polls []Poll, from, windowEnd time.Time, badSt tl.episodes = append(tl.episodes, episode{start: tl.episodeStart, end: end, closedByRealClear: real}) tl.badOpen = false } -<<<<<<< HEAD // onsetOf resolves a fresh episode's start: the instance's own ActiveAt, // translated to the runner domain by this poll's skew, clamped to // [from, windowEnd]. -======= - // onsetOf is a fresh episode's start: the translate ActiveAt, clamped to - // never read as starting before the window opened. ->>>>>>> 056b9146 (chore: shorten comments) onsetOf := func(p Poll, inst Instance) time.Time { start := runnerTime(p, inst.ActiveAt) if start.Before(from) { From f9d48c51af0c1aa5d91225fde8ccc2e6e2e3b793 Mon Sep 17 00:00:00 2001 From: Bartek Tofel Date: Wed, 9 Sep 2026 11:57:31 +0200 Subject: [PATCH 5/5] chore: use testify's require in tests (#2792) * chore: use testify's require in tests * chore: move remaining assumptions to testify * chore: address code review comments * chore: fix logging and std out printing (#2798) * chore: fix logging and std out printing * chore: truncate to seconds when comparing from time * chore: add centralized docs (#2802) * chore: add centralized docs * chore: further update docs * chore: address code review comments (#2807) * chore: address code review comments * chore: get rid of goreleaser --- .../workflows/grafana-alertcheck-release.yml | 34 - grafana-alertcheck/.goreleaser.yaml | 33 - grafana-alertcheck/README.md | 46 +- .../cmd/{grafana-alertcheck => }/check.go | 15 +- .../{grafana-alertcheck => }/check_test.go | 34 +- .../cmd/{grafana-alertcheck => }/common.go | 0 .../cmd/{grafana-alertcheck => }/env.go | 0 .../cmd/grafana-alertcheck/main_test.go | 60 -- .../cmd/grafana-alertcheck/version.go | 29 - .../cmd/grafana-alertcheck/version_test.go | 32 - .../cmd/{grafana-alertcheck => }/list.go | 0 .../cmd/{grafana-alertcheck => }/list_test.go | 31 +- .../cmd/{grafana-alertcheck => }/main.go | 4 +- grafana-alertcheck/cmd/main_test.go | 43 ++ grafana-alertcheck/cmd/style.go | 125 ++++ .../cmd/{grafana-alertcheck => }/table.go | 77 +- .../{grafana-alertcheck => }/table_test.go | 77 +- .../cmd/{grafana-alertcheck => }/watch.go | 10 +- .../{grafana-alertcheck => }/watch_test.go | 35 +- grafana-alertcheck/docs/_category_.yaml | 8 + grafana-alertcheck/docs/advanced.md | 41 ++ grafana-alertcheck/docs/architecture.md | 68 ++ .../docs/how-alerts-are-evaluated.md | 85 +++ grafana-alertcheck/docs/index.md | 87 +++ .../docs/reference/_category_.yaml | 8 + grafana-alertcheck/docs/reference/cli.md | 91 +++ .../docs/reference/log-format.md | 98 +++ grafana-alertcheck/go.mod | 4 + grafana-alertcheck/go.sum | 4 + grafana-alertcheck/internal/gate/check.go | 31 +- .../internal/gate/check_test.go | 685 ++++++------------ grafana-alertcheck/internal/gate/classify.go | 2 +- .../internal/gate/classify_test.go | 410 ++++------- grafana-alertcheck/internal/gate/coverage.go | 16 +- .../internal/gate/coverage_test.go | 263 +++---- .../internal/gate/duration_test.go | 16 +- .../internal/gate/flock_test.go | 18 +- .../internal/gate/jsonreq_test.go | 54 +- grafana-alertcheck/internal/gate/log_test.go | 467 ++++-------- .../internal/gate/parse_ruler.go | 2 +- .../internal/gate/parse_ruler_test.go | 103 +-- .../internal/gate/parse_state_test.go | 264 ++----- .../internal/gate/resolve_test.go | 222 ++---- grafana-alertcheck/internal/gate/schedule.go | 21 +- .../internal/gate/schedule_test.go | 271 ++----- grafana-alertcheck/internal/gate/source.go | 34 +- .../internal/gate/source_test.go | 294 +++----- .../internal/gate/watch_daemon_test.go | 119 +-- .../internal/gate/watch_test.go | 340 +++------ 49 files changed, 2034 insertions(+), 2777 deletions(-) delete mode 100644 .github/workflows/grafana-alertcheck-release.yml delete mode 100644 grafana-alertcheck/.goreleaser.yaml rename grafana-alertcheck/cmd/{grafana-alertcheck => }/check.go (94%) rename grafana-alertcheck/cmd/{grafana-alertcheck => }/check_test.go (84%) rename grafana-alertcheck/cmd/{grafana-alertcheck => }/common.go (100%) rename grafana-alertcheck/cmd/{grafana-alertcheck => }/env.go (100%) delete mode 100644 grafana-alertcheck/cmd/grafana-alertcheck/main_test.go delete mode 100644 grafana-alertcheck/cmd/grafana-alertcheck/version.go delete mode 100644 grafana-alertcheck/cmd/grafana-alertcheck/version_test.go rename grafana-alertcheck/cmd/{grafana-alertcheck => }/list.go (100%) rename grafana-alertcheck/cmd/{grafana-alertcheck => }/list_test.go (73%) rename grafana-alertcheck/cmd/{grafana-alertcheck => }/main.go (91%) create mode 100644 grafana-alertcheck/cmd/main_test.go create mode 100644 grafana-alertcheck/cmd/style.go rename grafana-alertcheck/cmd/{grafana-alertcheck => }/table.go (55%) rename grafana-alertcheck/cmd/{grafana-alertcheck => }/table_test.go (52%) rename grafana-alertcheck/cmd/{grafana-alertcheck => }/watch.go (95%) rename grafana-alertcheck/cmd/{grafana-alertcheck => }/watch_test.go (73%) create mode 100644 grafana-alertcheck/docs/_category_.yaml create mode 100644 grafana-alertcheck/docs/advanced.md create mode 100644 grafana-alertcheck/docs/architecture.md create mode 100644 grafana-alertcheck/docs/how-alerts-are-evaluated.md create mode 100644 grafana-alertcheck/docs/index.md create mode 100644 grafana-alertcheck/docs/reference/_category_.yaml create mode 100644 grafana-alertcheck/docs/reference/cli.md create mode 100644 grafana-alertcheck/docs/reference/log-format.md create mode 100644 grafana-alertcheck/go.sum diff --git a/.github/workflows/grafana-alertcheck-release.yml b/.github/workflows/grafana-alertcheck-release.yml deleted file mode 100644 index 4f946b18f..000000000 --- a/.github/workflows/grafana-alertcheck-release.yml +++ /dev/null @@ -1,34 +0,0 @@ -name: Grafana Alertcheck Release - -on: - push: - tags: - - grafana-alertcheck/v* - -jobs: - release: - name: Build and Release - runs-on: ubuntu-latest - environment: integration - permissions: - id-token: write - contents: write - steps: - - name: Checkout repo - uses: actions/checkout@v7 - with: - fetch-depth: 0 - - name: Set up Go - uses: actions/setup-go@v7 - with: - go-version-file: ./grafana-alertcheck/go.mod - cache-dependency-path: ./grafana-alertcheck/go.mod - - name: Goreleaser Release - uses: goreleaser/goreleaser-action@f06c13b6b1a9625abc9e6e439d9c05a8f2190e94 # v7.2.3 - with: - distribution: goreleaser-pro - version: "~> v2" - args: release --clean -f ./grafana-alertcheck/.goreleaser.yaml - env: - GITHUB_TOKEN: ${{ secrets.GITHUB_TOKEN }} - GORELEASER_KEY: ${{ secrets.GORELEASER_KEY }} diff --git a/grafana-alertcheck/.goreleaser.yaml b/grafana-alertcheck/.goreleaser.yaml deleted file mode 100644 index 0890d9f82..000000000 --- a/grafana-alertcheck/.goreleaser.yaml +++ /dev/null @@ -1,33 +0,0 @@ -# yaml-language-server: $schema=https://goreleaser.com/static/schema-pro.json -version: 2 -project_name: grafana-alertcheck - -dist: grafana-alertcheck/dist - -monorepo: - tag_prefix: grafana-alertcheck/ - dir: grafana-alertcheck - -builds: - - id: grafana-alertcheck - main: ./cmd/grafana-alertcheck/main.go - ldflags: - - -s - - -w - - -X github.com/smartcontractkit/chainlink-testing-framework/grafana-alertcheck/cmd/grafana-alertcheck.version={{.Version}} - - -X github.com/smartcontractkit/chainlink-testing-framework/grafana-alertcheck/cmd/grafana-alertcheck.commit={{.ShortCommit}} - - -X github.com/smartcontractkit/chainlink-testing-framework/grafana-alertcheck/cmd/grafana-alertcheck.date={{.CommitDate}} - - -X github.com/smartcontractkit/chainlink-testing-framework/grafana-alertcheck/cmd/grafana-alertcheck.builtBy=goreleaser - goos: - - linux - - darwin - goarch: - - amd64 - - arm64 - binary: grafana-alertcheck - env: - - CGO_ENABLED=0 - -before: - hooks: - - sh -c "cd grafana-alertcheck && go mod tidy" diff --git a/grafana-alertcheck/README.md b/grafana-alertcheck/README.md index 43cc03fc4..96aa0e999 100644 --- a/grafana-alertcheck/README.md +++ b/grafana-alertcheck/README.md @@ -1,6 +1,46 @@ # grafana-alertcheck -A CD quality gate for Grafana alerts: bookend a release with `watch` (record) and `check` (classify) to -answer whether any watched alert was in a bad state during the release window. +A CD quality gate for Grafana alerts. It bookends a release with two commands — `watch` (record) and +`check` (classify) — and answers whether any watched alert was in a bad state during the release window. -Under construction. +``` +watch → your work → check +``` + +`watch` starts a background recorder that polls each named alert into a JSONL log. After the work emits a +`from`/`to` pair, `check` proves continuous coverage of that window, classifies each alert's state +timeline, and exits `0`, `1`, or `2`. + +It **fails closed**: if it cannot get an answer, it stops the release — never a pass on an unproven window. + +## Quickstart + +```bash +export GRAFANA_URL=https://grafana.example.com +export GRAFANA_TOKEN=… + +grafana-alertcheck watch --out /tmp/run.jsonl --alerts alerts.txt +./deploy.sh # emits deployed_at= +./verify.sh # emits finished_at= +grafana-alertcheck check --in /tmp/run.jsonl --from "$deployed_at" --to "$finished_at" +``` + +Requires Grafana >= 13.0.0 and < 14.0.0. Connection details come from the environment only — the token is +never a flag. + +## Documentation + +| Doc | Covers | +| --- | ------ | +| [`docs/index.md`](./docs/index.md) | Overview, quickstarts, exit codes, common surprises | +| [`docs/how-alerts-are-evaluated.md`](./docs/how-alerts-are-evaluated.md) | Verdict model, coverage proof, health/liveness | +| [`docs/advanced.md`](./docs/advanced.md) | Check budget, scheduling, why history isn't queried | +| [`docs/architecture.md`](./docs/architecture.md) | Design invariants, the pure-function seam, recorder lifecycle | +| [`docs/reference/cli.md`](./docs/reference/cli.md) | Full CLI reference — subcommands, flags, naming | +| [`docs/reference/log-format.md`](./docs/reference/log-format.md) | The JSONL log schema, for debugging artifacts | + +## Build + +```bash +go build ./... && go test ./... +``` diff --git a/grafana-alertcheck/cmd/grafana-alertcheck/check.go b/grafana-alertcheck/cmd/check.go similarity index 94% rename from grafana-alertcheck/cmd/grafana-alertcheck/check.go rename to grafana-alertcheck/cmd/check.go index 3f2d295d6..e95fce376 100644 --- a/grafana-alertcheck/cmd/grafana-alertcheck/check.go +++ b/grafana-alertcheck/cmd/check.go @@ -3,6 +3,7 @@ package main import ( "context" "encoding/json" + "errors" "flag" "fmt" "io" @@ -40,6 +41,13 @@ func runCheck(args []string, stdin io.Reader, stdout, stderr io.Writer) int { output := fs.String("output", "", `"json" writes the machine-readable Result to stdout in addition to the table; default is the table alone`) if err := fs.Parse(args); err != nil { + if errors.Is(err, flag.ErrHelp) { + return 0 + } + return 2 + } + if fs.NArg() != 0 { + fmt.Fprintf(stderr, "check: unexpected arguments %v\n", fs.Args()) return 2 } if *output != "" && *output != "json" { @@ -82,7 +90,7 @@ func runCheck(args []string, stdin io.Reader, stdout, stderr io.Writer) int { PidFile: *pidfile, Concurrency: *common.concurrency, Clock: gate.SystemClock{}, - Notes: stderr, + Notes: newNoteStyler(stderr), } if *to == "" { fmt.Fprintln(stderr, "check: --to is required") @@ -113,11 +121,10 @@ func runCheck(args []string, stdin io.Reader, stdout, stderr io.Writer) int { result, checkErr := gate.Check(ctx, cfg) - if err := renderTable(stderr, result); err != nil { - fmt.Fprintln(stderr, err) - } if checkErr != nil { fmt.Fprintln(stderr, checkErr) + } else if err := renderTable(stderr, result); err != nil { + fmt.Fprintln(stderr, err) } if *output == "json" { enc := json.NewEncoder(stdout) diff --git a/grafana-alertcheck/cmd/grafana-alertcheck/check_test.go b/grafana-alertcheck/cmd/check_test.go similarity index 84% rename from grafana-alertcheck/cmd/grafana-alertcheck/check_test.go rename to grafana-alertcheck/cmd/check_test.go index 99e8ebc3a..c7c797a0b 100644 --- a/grafana-alertcheck/cmd/grafana-alertcheck/check_test.go +++ b/grafana-alertcheck/cmd/check_test.go @@ -4,10 +4,10 @@ import ( "bytes" "errors" "os" - "strings" "testing" "github.com/smartcontractkit/chainlink-testing-framework/grafana-alertcheck/internal/gate" + "github.com/stretchr/testify/require" ) // The exit-code mapping, pinned directly against exitCode with no network @@ -27,9 +27,7 @@ func TestExitCode(t *testing.T) { } for _, tt := range tests { t.Run(tt.name, func(t *testing.T) { - if got := exitCode(tt.res, tt.err); got != tt.want { - t.Fatalf("exitCode(...) = %d, want %d", got, tt.want) - } + require.Equal(t, tt.want, exitCode(tt.res, tt.err)) }) } } @@ -37,9 +35,7 @@ func TestExitCode(t *testing.T) { func writeTempAlerts(t *testing.T) string { t.Helper() path := t.TempDir() + "/alerts.txt" - if err := os.WriteFile(path, []byte("Some Alert\n"), 0o644); err != nil { - t.Fatal(err) - } + require.NoError(t, os.WriteFile(path, []byte("Some Alert\n"), 0o644)) return path } @@ -100,12 +96,8 @@ func TestRunCheck_FlagValidation(t *testing.T) { var stdout, stderr bytes.Buffer args := append([]string{"check"}, tt.args(t)...) code := run(args, &stdout, &stderr) - if code != 2 { - t.Fatalf("code = %d, want 2; stderr = %q", code, stderr.String()) - } - if !strings.Contains(stderr.String(), tt.wantErr) { - t.Fatalf("stderr = %q, want it to contain %q", stderr.String(), tt.wantErr) - } + require.Equal(t, 2, code) + require.Contains(t, stderr.String(), tt.wantErr) }) } } @@ -121,12 +113,8 @@ func TestRunCheck_ToInPastNoLog(t *testing.T) { "--from", "1999-01-01T00:00:00Z", "--to", "2000-01-01T00:00:00Z", "--alerts", writeTempAlerts(t), }, &stdout, &stderr) - if code != 2 { - t.Fatalf("code = %d, want 2; stderr = %q", code, stderr.String()) - } - if !strings.Contains(stderr.String(), "already passed") { - t.Fatalf("stderr = %q, want the past-`to` refusal", stderr.String()) - } + require.Equal(t, 2, code) + require.Contains(t, stderr.String(), "already passed") } // --output json never writes to stdout when Check was never reached, because @@ -137,10 +125,6 @@ func TestRunCheck_NoResultOnConfigError(t *testing.T) { t.Setenv("GRAFANA_TOKEN", "") var stdout, stderr bytes.Buffer code := run([]string{"check", "--to", "2026-01-01T00:00:00Z", "--output", "json"}, &stdout, &stderr) - if code != 2 { - t.Fatalf("code = %d, want 2", code) - } - if stdout.Len() != 0 { - t.Fatalf("stdout = %q, want empty", stdout.String()) - } + require.Equal(t, 2, code) + require.Empty(t, stdout.String()) } diff --git a/grafana-alertcheck/cmd/grafana-alertcheck/common.go b/grafana-alertcheck/cmd/common.go similarity index 100% rename from grafana-alertcheck/cmd/grafana-alertcheck/common.go rename to grafana-alertcheck/cmd/common.go diff --git a/grafana-alertcheck/cmd/grafana-alertcheck/env.go b/grafana-alertcheck/cmd/env.go similarity index 100% rename from grafana-alertcheck/cmd/grafana-alertcheck/env.go rename to grafana-alertcheck/cmd/env.go diff --git a/grafana-alertcheck/cmd/grafana-alertcheck/main_test.go b/grafana-alertcheck/cmd/grafana-alertcheck/main_test.go deleted file mode 100644 index 1d7906674..000000000 --- a/grafana-alertcheck/cmd/grafana-alertcheck/main_test.go +++ /dev/null @@ -1,60 +0,0 @@ -package main - -import ( - "bytes" - "strings" - "testing" -) - -func TestRun_NoArgs(t *testing.T) { - var stdout, stderr bytes.Buffer - code := run(nil, &stdout, &stderr) - if code != 2 { - t.Fatalf("code = %d, want 2", code) - } - if !strings.Contains(stderr.String(), "usage") { - t.Fatalf("stderr = %q, want a usage message", stderr.String()) - } -} - -func TestRun_Help(t *testing.T) { - for _, flag := range []string{"-h", "-help", "--help"} { - t.Run(flag, func(t *testing.T) { - var stdout, stderr bytes.Buffer - code := run([]string{flag}, &stdout, &stderr) - if code != 0 { - t.Fatalf("code = %d, want 0 (requested help is not a could-not-check condition)", code) - } - if !strings.Contains(stdout.String(), "usage") { - t.Fatalf("stdout = %q, want a usage message", stdout.String()) - } - if stderr.String() != "" { - t.Fatalf("stderr = %q, want empty — help goes to stdout", stderr.String()) - } - }) - } -} - -func TestRun_UnknownSubcommand(t *testing.T) { - var stdout, stderr bytes.Buffer - code := run([]string{"bogus"}, &stdout, &stderr) - if code != 2 { - t.Fatalf("code = %d, want 2", code) - } - if !strings.Contains(stderr.String(), `"bogus"`) { - t.Fatalf("stderr = %q, want it to name the unknown subcommand", stderr.String()) - } -} - -func TestRun_List_MissingEnv(t *testing.T) { - t.Setenv("GRAFANA_URL", "") - t.Setenv("GRAFANA_TOKEN", "") - var stdout, stderr bytes.Buffer - code := run([]string{"list"}, &stdout, &stderr) - if code != 2 { - t.Fatalf("code = %d, want 2", code) - } - if !strings.Contains(stderr.String(), "GRAFANA_URL") { - t.Fatalf("stderr = %q, want it to name the missing env var", stderr.String()) - } -} diff --git a/grafana-alertcheck/cmd/grafana-alertcheck/version.go b/grafana-alertcheck/cmd/grafana-alertcheck/version.go deleted file mode 100644 index 5cf68b6af..000000000 --- a/grafana-alertcheck/cmd/grafana-alertcheck/version.go +++ /dev/null @@ -1,29 +0,0 @@ -package main - -import ( - "fmt" - "io" -) - -// Build metadata. These are package-level variables so goreleaser's ldflags -// (-X) can stamp them at build time; left unstamped they fall back to the -// "dev" defaults below, which is what a plain `go build` produces. -var ( - version = "dev" - commit = "unknown" - date = "unknown" - builtBy = "unknown" -) - -// runVersion prints the build metadata to stdout. Unlike list/watch/check it -// needs no Grafana connection, so it never touches the environment or the -// network; it exists purely so operators can answer "what am I running?" -// against a deployed binary. -func runVersion(args []string, stdout, stderr io.Writer) int { - if len(args) != 0 { - fmt.Fprintf(stderr, "version takes no arguments, got %v\n", args) - return 2 - } - fmt.Fprintf(stdout, "version: %s\ncommit: %s\ndate: %s\nbuiltBy: %s\n", version, commit, date, builtBy) - return 0 -} diff --git a/grafana-alertcheck/cmd/grafana-alertcheck/version_test.go b/grafana-alertcheck/cmd/grafana-alertcheck/version_test.go deleted file mode 100644 index 4122c2f13..000000000 --- a/grafana-alertcheck/cmd/grafana-alertcheck/version_test.go +++ /dev/null @@ -1,32 +0,0 @@ -package main - -import ( - "bytes" - "strings" - "testing" -) - -func TestRunVersion(t *testing.T) { - var stdout, stderr bytes.Buffer - code := run([]string{"version"}, &stdout, &stderr) - if code != 0 { - t.Fatalf("code = %d, want 0", code) - } - out := stdout.String() - for _, want := range []string{"version:", "commit:", "date:", "builtBy:"} { - if !strings.Contains(out, want) { - t.Errorf("stdout = %q, want it to contain %q", out, want) - } - } - if stderr.String() != "" { - t.Errorf("stderr = %q, want empty", stderr.String()) - } -} - -func TestRunVersion_RejectsArgs(t *testing.T) { - var stdout, stderr bytes.Buffer - code := run([]string{"version", "extra"}, &stdout, &stderr) - if code != 2 { - t.Fatalf("code = %d, want 2", code) - } -} diff --git a/grafana-alertcheck/cmd/grafana-alertcheck/list.go b/grafana-alertcheck/cmd/list.go similarity index 100% rename from grafana-alertcheck/cmd/grafana-alertcheck/list.go rename to grafana-alertcheck/cmd/list.go diff --git a/grafana-alertcheck/cmd/grafana-alertcheck/list_test.go b/grafana-alertcheck/cmd/list_test.go similarity index 73% rename from grafana-alertcheck/cmd/grafana-alertcheck/list_test.go rename to grafana-alertcheck/cmd/list_test.go index 119326c2c..62aa200d2 100644 --- a/grafana-alertcheck/cmd/grafana-alertcheck/list_test.go +++ b/grafana-alertcheck/cmd/list_test.go @@ -5,8 +5,9 @@ import ( "fmt" "net/http" "net/http/httptest" - "strings" "testing" + + "github.com/stretchr/testify/require" ) const rulerBody = `{ @@ -60,19 +61,11 @@ func TestRunList_HappyPath(t *testing.T) { var stdout, stderr bytes.Buffer code := run([]string{"list"}, &stdout, &stderr) - if code != 0 { - t.Fatalf("code = %d, want 0; stderr = %q", code, stderr.String()) - } + require.Equal(t, 0, code) out := stdout.String() - if !strings.Contains(out, "rule0000006a") { - t.Errorf("stdout = %q, want it to list rule0000006a", out) - } - if !strings.Contains(out, "Example No Gateways Available") { - t.Errorf("stdout = %q, want it to list the rule title", out) - } - if !strings.Contains(out, "grafana-managed") { - t.Errorf("stdout = %q, want it to name the rule kind", out) - } + require.Contains(t, out, "rule0000006a") + require.Contains(t, out, "Example No Gateways Available") + require.Contains(t, out, "grafana-managed") } func TestRunList_UnsupportedVersion(t *testing.T) { @@ -82,12 +75,8 @@ func TestRunList_UnsupportedVersion(t *testing.T) { var stdout, stderr bytes.Buffer code := run([]string{"list"}, &stdout, &stderr) - if code != 2 { - t.Fatalf("code = %d, want 2", code) - } - if !strings.Contains(stderr.String(), "12.5.0") { - t.Fatalf("stderr = %q, want it to name the unsupported version", stderr.String()) - } + require.Equal(t, 2, code) + require.Contains(t, stderr.String(), "12.5.0") } func TestRunList_RejectsArgs(t *testing.T) { @@ -96,7 +85,5 @@ func TestRunList_RejectsArgs(t *testing.T) { var stdout, stderr bytes.Buffer code := run([]string{"list", "extra"}, &stdout, &stderr) - if code != 2 { - t.Fatalf("code = %d, want 2", code) - } + require.Equal(t, 2, code) } diff --git a/grafana-alertcheck/cmd/grafana-alertcheck/main.go b/grafana-alertcheck/cmd/main.go similarity index 91% rename from grafana-alertcheck/cmd/grafana-alertcheck/main.go rename to grafana-alertcheck/cmd/main.go index 2672de1a3..7ab5d6397 100644 --- a/grafana-alertcheck/cmd/grafana-alertcheck/main.go +++ b/grafana-alertcheck/cmd/main.go @@ -12,7 +12,7 @@ func main() { os.Exit(run(os.Args[1:], os.Stdout, os.Stderr)) } -const usage = "usage: grafana-alertcheck " +const usage = "usage: grafana-alertcheck " // run is the whole of main's testable surface: parse the subcommand, dispatch, // return the process exit code. Exit codes below 2 (pass/violations) belong to @@ -37,8 +37,6 @@ func run(args []string, stdout, stderr io.Writer) int { return runWatch(args[1:], os.Stdin, stdout, stderr) case "check": return runCheck(args[1:], os.Stdin, stdout, stderr) - case "version": - return runVersion(args[1:], stdout, stderr) case "-h", "-help", "--help": fmt.Fprintln(stdout, usage) return 0 diff --git a/grafana-alertcheck/cmd/main_test.go b/grafana-alertcheck/cmd/main_test.go new file mode 100644 index 000000000..34dd3b2d2 --- /dev/null +++ b/grafana-alertcheck/cmd/main_test.go @@ -0,0 +1,43 @@ +package main + +import ( + "bytes" + "testing" + + "github.com/stretchr/testify/require" +) + +func TestRun_NoArgs(t *testing.T) { + var stdout, stderr bytes.Buffer + code := run(nil, &stdout, &stderr) + require.Equal(t, 2, code) + require.Contains(t, stderr.String(), "usage") +} + +func TestRun_Help(t *testing.T) { + for _, flag := range []string{"-h", "-help", "--help"} { + t.Run(flag, func(t *testing.T) { + var stdout, stderr bytes.Buffer + code := run([]string{flag}, &stdout, &stderr) + require.Equal(t, 0, code, "requested help is not a could-not-check condition") + require.Contains(t, stdout.String(), "usage") + require.Empty(t, stderr.String(), "help goes to stdout") + }) + } +} + +func TestRun_UnknownSubcommand(t *testing.T) { + var stdout, stderr bytes.Buffer + code := run([]string{"bogus"}, &stdout, &stderr) + require.Equal(t, 2, code) + require.Contains(t, stderr.String(), `"bogus"`) +} + +func TestRun_List_MissingEnv(t *testing.T) { + t.Setenv("GRAFANA_URL", "") + t.Setenv("GRAFANA_TOKEN", "") + var stdout, stderr bytes.Buffer + code := run([]string{"list"}, &stdout, &stderr) + require.Equal(t, 2, code) + require.Contains(t, stderr.String(), "GRAFANA_URL") +} diff --git a/grafana-alertcheck/cmd/style.go b/grafana-alertcheck/cmd/style.go new file mode 100644 index 000000000..bb46fc2ec --- /dev/null +++ b/grafana-alertcheck/cmd/style.go @@ -0,0 +1,125 @@ +package main + +import ( + "bytes" + "io" + "os" + "strings" +) + +// ANSI SGR codes for the human-facing notes and table footer. The colours are +// applied only when the destination is a terminal (see colorEnabled); a pipe, +// file or CI log gets plain text, so stdout stays reserved for --output json +// and no machine reader ever sees escape sequences. +const ( + ansiReset = "\x1b[0m" + ansiRed = "\x1b[31m" + ansiGreen = "\x1b[32m" + ansiYellow = "\x1b[33m" + ansiCyan = "\x1b[36m" + // Orange has no entry in the base-16 palette; 256-colour 208 is a legible + // orange used for warnings, distinct from the yellow used for notes. + ansiOrange = "\x1b[38;5;208m" +) + +// colorEnabled reports whether ANSI colour should be written to w. Colour is +// written only when three things hold: NO_COLOR is unset, w is a real *os.File +// (so text/tabwriter buffers, strings.Builder and bytes.Buffer tests all stay +// plain), and that file is a character device (a terminal, not a redirect). +func colorEnabled(w io.Writer) bool { + if os.Getenv("NO_COLOR") != "" { + return false + } + f, ok := w.(*os.File) + if !ok { + return false + } + fi, err := f.Stat() + if err != nil { + return false + } + return fi.Mode()&os.ModeCharDevice != 0 +} + +// styleLine applies the note vocabulary's colour to one line when enabled. The +// colour wraps the text only; the terminating newline is written uncoloured so +// the terminal's line discipline is never inside the escape sequence. +func styleLine(line string, enabled bool) string { + if !enabled { + return line + } + content := strings.TrimRight(line, "\n") + var color string + switch { + case strings.HasPrefix(content, "warning:"): + color = ansiOrange + case strings.HasPrefix(content, "note:"): + color = ansiYellow + case strings.HasPrefix(content, "drain wait:"): + color = ansiCyan + } + if color == "" { + return line + } + return color + content + ansiReset + "\n" +} + +// noteStyler wraps the gate package's Notes stream — a presentation seam that +// keeps colour out of the library. It colourises each line by its known prefix +// and separates the collection countdown from the setup phase with a single +// blank line before the first "collecting:" line. The gate keeps emitting plain +// prose; only the CLI lays it out. +type noteStyler struct { + w io.Writer + enabled bool + pending []byte + sawCollecting bool +} + +func newNoteStyler(w io.Writer) *noteStyler { + return ¬eStyler{w: w, enabled: colorEnabled(w)} +} + +// startsSection reports whether a line opens a new phase of the stream and so +// deserves a blank line above it. "collecting:" opens the countdown (once — +// later countdown lines follow on from the first), and "drain wait:" opens the +// drain phase. The setup lines (planned run time, warning, min-observed, notes) +// are one contiguous block and are not separated from each other. +func (s *noteStyler) startsSection(line string) bool { + switch { + case strings.HasPrefix(line, "warning:"): + return true + case strings.HasPrefix(line, "drain wait:"): + return true + case strings.HasPrefix(line, "collecting:"): + if s.sawCollecting { + return false + } + s.sawCollecting = true + return true + } + return false +} + +func (s *noteStyler) Write(p []byte) (int, error) { + n := len(p) + s.pending = append(s.pending, p...) + for { + i := bytes.IndexByte(s.pending, '\n') + if i < 0 { + break + } + line := string(s.pending[:i+1]) + s.pending = s.pending[i+1:] + + if s.startsSection(line) { + if _, err := io.WriteString(s.w, "\n"); err != nil { + return n, err + } + } + if _, err := io.WriteString(s.w, styleLine(line, s.enabled)); err != nil { + return n, err + } + } + return n, nil +} diff --git a/grafana-alertcheck/cmd/grafana-alertcheck/table.go b/grafana-alertcheck/cmd/table.go similarity index 55% rename from grafana-alertcheck/cmd/grafana-alertcheck/table.go rename to grafana-alertcheck/cmd/table.go index 11b72a074..39ec75cbc 100644 --- a/grafana-alertcheck/cmd/grafana-alertcheck/table.go +++ b/grafana-alertcheck/cmd/table.go @@ -14,29 +14,39 @@ import ( // which the caller (runCheck) always points at stderr — stdout is reserved for // the machine-readable --output json. // -// Three sections, in order: +// Three titled tables, in order (the name column is RULE in all of them — one +// row is one resolved alert rule, never a firing instance): // -// 1. one line per rule: outcome, BadFor, pollEvery, proved-or-not with the -// largest gap; -// 2. one line per Violation: a rule's worst-of outcome does not carry the -// State/Health of the instance that actually caused it — Violation does — -// so this is also where those two columns appear, sorted after the rule -// table rather than folded into it, and it is the only place an operator -// running WITHOUT --output json sees the --allow-paused hint that -// Violation.Note already carries (classify.go); -// 3. a footer with the numbers that answer "why" on exit 2: each non-skipped -// rule's maxGap/healthGrace/evalStaleAfter, the global transitionGrace and -// drainTimeout, and the largest measured clock skew alongside its own -// error bound (RTT/2) — SkewHardLimit is a separate, fixed input threshold -// and is reported next to it, never as if it were that bound. +// 1. RESULTS, one line per rule: outcome, BadFor, pollEvery, proved-or-not +// with the largest gap; +// 2. VIOLATIONS, one line per Violation (only when any): a rule's worst-of +// outcome does not carry the State/Health of the instance that actually +// caused it — Violation does — so this is also where those two columns +// appear, sorted after the result table rather than folded into it, and it +// is the only place an operator running WITHOUT --output json sees the +// --allow-paused hint that Violation.Note already carries (classify.go); +// 3. THRESHOLDS, the numbers that answer "why" on exit 2: each non-skipped +// rule's maxGap/healthGrace/evalStaleAfter, followed by the global +// transitionGrace and drainTimeout, and the largest measured clock skew +// alongside its own error bound (RTT/2) — SkewHardLimit is a separate, +// fixed input threshold and is reported next to it, never as if it were +// that bound. func renderTable(w io.Writer, res gate.Result) error { alertOf := make(map[string]string, len(res.Verdicts)) for _, v := range res.Verdicts { alertOf[v.RuleUID] = v.Alert } + enabled := colorEnabled(w) + // A blank line separates the result table from the notes the gate streamed + // before it (planned run time, warning, min-observed, collecting, drain + // wait), so the verdict reads as its own section rather than the tail of a + // wall of progress text. + fmt.Fprintln(w) + + fmt.Fprintln(w, "RESULTS") tw := tabwriter.NewWriter(w, 0, 4, 2, ' ', 0) - fmt.Fprintln(tw, "ALERT\tOUTCOME\tBADFOR\tPOLLEVERY\tPROVED\tNOTE") + fmt.Fprintln(tw, "RULE\tOUTCOME\tBADFOR\tPOLLEVERY\tPROVED\tNOTE") for _, v := range sortedVerdicts(res.Verdicts) { fmt.Fprintf(tw, "%s\t%s\t%s\t%s\t%s\t%s\n", v.Alert, v.Outcome, v.BadFor.Round(time.Second), v.PollEvery.Round(time.Second), @@ -49,7 +59,7 @@ func renderTable(w io.Writer, res gate.Result) error { if len(res.Violations) > 0 { fmt.Fprintln(w, "\nVIOLATIONS") vtw := tabwriter.NewWriter(w, 0, 4, 2, ' ', 0) - fmt.Fprintln(vtw, "ALERT\tOUTCOME\tSTATE\tHEALTH\tNOTE") + fmt.Fprintln(vtw, "RULE\tOUTCOME\tSTATE\tHEALTH\tNOTE") for _, v := range sortedViolations(res.Violations) { fmt.Fprintf(vtw, "%s\t%s\t%s\t%s\t%s\n", alertLabel(v, alertOf), v.Outcome, v.State, v.Health, v.Note) } @@ -58,20 +68,49 @@ func renderTable(w io.Writer, res gate.Result) error { } } + // The per-rule thresholds answer "why" on exit 2: a table, not the prose + // "rule NAME: maxGap=... healthGrace=... evalStaleAfter=..." that repeated + // the rule name a fourth time. It is separated from the result above by a + // blank line. fmt.Fprintln(w) + fmt.Fprintln(w, "THRESHOLDS") + ttw := tabwriter.NewWriter(w, 0, 4, 2, ' ', 0) + fmt.Fprintln(ttw, "RULE\tMAXGAP\tHEALTHGRACE\tEVALSTALEAFTER") for _, uid := range sortedThresholdUIDs(res.Thresholds, alertOf) { t := res.Thresholds[uid] - fmt.Fprintf(w, "rule %s: maxGap=%s healthGrace=%s evalStaleAfter=%s\n", + fmt.Fprintf(ttw, "%s\t%s\t%s\t%s\n", alertOr(uid, alertOf), t.MaxGap, t.HealthGrace, t.EvalStaleAfter) } + if err := ttw.Flush(); err != nil { + return fmt.Errorf("render table: %w", err) + } + + fmt.Fprintln(w) fmt.Fprintf(w, "global: transitionGrace=%s (source: %s) drainTimeout=%s\n", res.Global.TransitionGrace, res.Global.GraceSource, res.Global.DrainTimeout) - fmt.Fprintf(w, "violations: %d, largest measured clock skew: %s (bound ±%s, hard limit %s), grafana %s\n", - len(res.Violations), res.ClockSkew.Round(time.Millisecond), res.ClockSkewBound.Round(time.Millisecond), + fmt.Fprintf(w, "largest measured clock skew: %s (bound ±%s, hard limit %s), grafana %s\n", + res.ClockSkew.Round(time.Millisecond), res.ClockSkewBound.Round(time.Millisecond), gate.SkewHardLimit, res.GrafanaVersion) + // The verdict — the single number a terminal operator reads last — sits on + // its own line at the very bottom, separated from the diagnostics above and + // from the shell prompt below. + fmt.Fprintf(w, "\n%s\n\n", violationsLabel(len(res.Violations), enabled)) return nil } +// violationsLabel colours the "violations: N" prefix of the footer: green for a +// clean run, red otherwise. The rest of the line is written uncoloured. +func violationsLabel(n int, enabled bool) string { + s := fmt.Sprintf("violations: %d", n) + if !enabled { + return s + } + if n == 0 { + return ansiGreen + s + ansiReset + } + return ansiRed + s + ansiReset +} + // provedLabel is the table's PROVED column: "yes" for a clean coverage // proof, "no" with the reason and largest gap for an unobservable rule, and // "-" for a rule decide never asked proveCoverage about at all (skipped — diff --git a/grafana-alertcheck/cmd/grafana-alertcheck/table_test.go b/grafana-alertcheck/cmd/table_test.go similarity index 52% rename from grafana-alertcheck/cmd/grafana-alertcheck/table_test.go rename to grafana-alertcheck/cmd/table_test.go index e585e9b61..86c80a19b 100644 --- a/grafana-alertcheck/cmd/grafana-alertcheck/table_test.go +++ b/grafana-alertcheck/cmd/table_test.go @@ -2,11 +2,11 @@ package main import ( "bytes" - "strings" "testing" "time" "github.com/smartcontractkit/chainlink-testing-framework/grafana-alertcheck/internal/gate" + "github.com/stretchr/testify/require" ) // The golden table test: a fixed Result renders a deterministic, ordered rule @@ -46,68 +46,43 @@ func TestRenderTable(t *testing.T) { } var buf bytes.Buffer - if err := renderTable(&buf, res); err != nil { - t.Fatalf("renderTable: %v", err) - } + require.NoError(t, renderTable(&buf, res)) out := buf.String() // Rule table: Ape sorts before Zebra sorts before... Paused is skipped and // carries no coverage entry, so it renders "-" for PROVED. - if !strings.Contains(out, "Ape Alert") || !strings.Contains(out, "unobservable") { - t.Fatalf("out = %q, want Ape's unobservable row", out) - } - if !strings.Contains(out, "heartbeat_gap") || !strings.Contains(out, "largest gap 5m0s") { - t.Fatalf("out = %q, want the coverage reason and largest gap", out) - } - if !strings.Contains(out, "Zebra Alert") || !strings.Contains(out, "clean") { - t.Fatalf("out = %q, want Zebra's clean row", out) - } + require.Contains(t, out, "Ape Alert") + require.Contains(t, out, "unobservable") + require.Contains(t, out, "heartbeat_gap") + require.Contains(t, out, "largest gap 5m0s") + require.Contains(t, out, "Zebra Alert") + require.Contains(t, out, "clean") // The violations section must show up even without --output json, and must // carry the --allow-paused hint text verbatim. - if !strings.Contains(out, "VIOLATIONS") { - t.Fatalf("out = %q, want a VIOLATIONS section", out) - } - if !strings.Contains(out, "--allow-paused") { - t.Fatalf("out = %q, want the --allow-paused hint in the human table", out) - } - if !strings.Contains(out, "STATE") || !strings.Contains(out, "HEALTH") { - t.Fatalf("out = %q, want the violations table to have STATE and HEALTH columns", out) - } - if !strings.Contains(out, string(gate.StateFiring)) || !strings.Contains(out, "error") { - t.Fatalf("out = %q, want Ape's violation State/Health", out) - } + require.Contains(t, out, "VIOLATIONS") + require.Contains(t, out, "--allow-paused") + require.Contains(t, out, "STATE") + require.Contains(t, out, "HEALTH") + require.Contains(t, out, string(gate.StateFiring)) + require.Contains(t, out, "error") - // The footer: per-rule thresholds, global thresholds, and skew with its own - // bound rather than the fixed hard limit. - if !strings.Contains(out, "Ape Alert: maxGap=1m0s healthGrace=2m0s evalStaleAfter=1m0s") { - t.Fatalf("out = %q, want Ape's per-rule thresholds", out) - } - if !strings.Contains(out, "Zebra Alert: maxGap=1m0s healthGrace=1m0s evalStaleAfter=1m0s") { - t.Fatalf("out = %q, want Zebra's per-rule thresholds", out) - } - if strings.Contains(out, "Paused Alert: maxGap") { - t.Fatalf("out = %q, a skipped rule must not report thresholds it never had", out) - } - if !strings.Contains(out, "global: transitionGrace=5m0s (source: Ape Alert (for=5m)) drainTimeout=2m0s") { - t.Fatalf("out = %q, want the global thresholds line", out) - } - if !strings.Contains(out, "largest measured clock skew: 1.5s (bound ±250ms, hard limit 1m0s)") { - t.Fatalf("out = %q, want the skew and its own bound, not the hard limit misused as one", out) - } - if !strings.Contains(out, "violations: 2") { - t.Fatalf("out = %q, want the violation count", out) - } - if !strings.Contains(out, "13.1.0") { - t.Fatalf("out = %q, want the grafana version", out) - } + // The footer: per-rule thresholds are a table (RULE/MAXGAP/HEALTHGRACE/ + // EVALSTALEAFTER) rather than prose, followed by the global thresholds and + // the violations count with the skew and its own bound rather than the + // fixed hard limit. + require.Contains(t, out, "MAXGAP") + require.Contains(t, out, "HEALTHGRACE") + require.Contains(t, out, "EVALSTALEAFTER") + require.Contains(t, out, "global: transitionGrace=5m0s (source: Ape Alert (for=5m)) drainTimeout=2m0s") + require.Contains(t, out, "largest measured clock skew: 1.5s (bound ±250ms, hard limit 1m0s)") + require.Contains(t, out, "violations: 2") + require.Contains(t, out, "13.1.0") } // The "-" case: a rule decide never asked proveCoverage about (paused before // the window opened) has an empty CoverageResult and must not be reported as // either proved or unobservable. func TestProvedLabel_Skipped(t *testing.T) { - if got := provedLabel(gate.CoverageResult{}); got != "-" { - t.Fatalf("provedLabel(zero value) = %q, want \"-\"", got) - } + require.Equal(t, "-", provedLabel(gate.CoverageResult{})) } diff --git a/grafana-alertcheck/cmd/grafana-alertcheck/watch.go b/grafana-alertcheck/cmd/watch.go similarity index 95% rename from grafana-alertcheck/cmd/grafana-alertcheck/watch.go rename to grafana-alertcheck/cmd/watch.go index df2c50029..69b585298 100644 --- a/grafana-alertcheck/cmd/grafana-alertcheck/watch.go +++ b/grafana-alertcheck/cmd/watch.go @@ -2,6 +2,7 @@ package main import ( "context" + "errors" "flag" "fmt" "io" @@ -49,6 +50,13 @@ func runWatch(args []string, stdin io.Reader, stdout, stderr io.Writer) int { readyFD := fs.Int(gate.ReadyFDFlag[2:], 0, "") if err := fs.Parse(args); err != nil { + if errors.Is(err, flag.ErrHelp) { + return 0 + } + return 2 + } + if fs.NArg() != 0 { + fmt.Fprintf(stderr, "watch: unexpected arguments %v\n", fs.Args()) return 2 } @@ -78,7 +86,7 @@ func runWatch(args []string, stdin io.Reader, stdout, stderr io.Writer) int { DaemonLog: *daemonLog, Concurrency: *common.concurrency, Clock: gate.SystemClock{}, - Notes: stderr, + Notes: newNoteStyler(stderr), } if *until != "" { t, err := time.Parse(time.RFC3339, *until) diff --git a/grafana-alertcheck/cmd/grafana-alertcheck/watch_test.go b/grafana-alertcheck/cmd/watch_test.go similarity index 73% rename from grafana-alertcheck/cmd/grafana-alertcheck/watch_test.go rename to grafana-alertcheck/cmd/watch_test.go index 6d8c8c1a6..5c8f53c55 100644 --- a/grafana-alertcheck/cmd/grafana-alertcheck/watch_test.go +++ b/grafana-alertcheck/cmd/watch_test.go @@ -3,8 +3,9 @@ package main import ( "bytes" "os" - "strings" "testing" + + "github.com/stretchr/testify/require" ) // The record step's flag-validation matrix. Every case fails inside @@ -47,12 +48,8 @@ func TestRunWatch_FlagValidation(t *testing.T) { var stdout, stderr bytes.Buffer args := append([]string{"watch"}, tt.args(t)...) code := run(args, &stdout, &stderr) - if code != 2 { - t.Fatalf("code = %d, want 2; stderr = %q", code, stderr.String()) - } - if !strings.Contains(stderr.String(), tt.wantErr) { - t.Fatalf("stderr = %q, want it to contain %q", stderr.String(), tt.wantErr) - } + require.Equal(t, 2, code) + require.Contains(t, stderr.String(), tt.wantErr) }) } } @@ -68,18 +65,10 @@ func TestRunWatch_DaemonChildDispatch(t *testing.T) { // enough to prove dispatch happened without needing a real recording. missing := os.DevNull + ".missing" code := run([]string{"watch", "--daemon-child", "--out", missing, "--ready-fd", "0"}, &stdout, &stderr) - if code != 2 { - t.Fatalf("code = %d, want 2; stderr = %q", code, stderr.String()) - } - if !strings.Contains(stderr.String(), missing) { - t.Fatalf("stderr = %q, want RunDaemonChild's read failure naming %q", stderr.String(), missing) - } - if strings.Contains(watchUsage, "daemon-child") { - t.Fatalf("watchUsage = %q, must never name --daemon-child", watchUsage) - } - if strings.Contains(watchUsage, "ready-fd") { - t.Fatalf("watchUsage = %q, must never name --ready-fd", watchUsage) - } + require.Equal(t, 2, code) + require.Contains(t, stderr.String(), missing) + require.NotContains(t, watchUsage, "daemon-child") + require.NotContains(t, watchUsage, "ready-fd") } func TestRunWatch_DaemonChild_MissingEnv(t *testing.T) { @@ -88,10 +77,6 @@ func TestRunWatch_DaemonChild_MissingEnv(t *testing.T) { var stdout, stderr bytes.Buffer code := run([]string{"watch", "--daemon-child", "--out", "log.jsonl"}, &stdout, &stderr) - if code != 2 { - t.Fatalf("code = %d, want 2; stderr = %q", code, stderr.String()) - } - if !strings.Contains(stderr.String(), "GRAFANA_URL") { - t.Fatalf("stderr = %q, want it to name the missing env var", stderr.String()) - } + require.Equal(t, 2, code) + require.Contains(t, stderr.String(), "GRAFANA_URL") } diff --git a/grafana-alertcheck/docs/_category_.yaml b/grafana-alertcheck/docs/_category_.yaml new file mode 100644 index 000000000..3cbc431c4 --- /dev/null +++ b/grafana-alertcheck/docs/_category_.yaml @@ -0,0 +1,8 @@ +position: 1 +label: 'Grafana Alertcheck' +collapsible: true +collapsed: false +link: + type: generated-index + slug: /platform-services/devex/cicd/grafana-alertcheck/index + description: 'CD quality gate for Grafana alerts: watch, classify, and gate releases.' diff --git a/grafana-alertcheck/docs/advanced.md b/grafana-alertcheck/docs/advanced.md new file mode 100644 index 000000000..073292cd6 --- /dev/null +++ b/grafana-alertcheck/docs/advanced.md @@ -0,0 +1,41 @@ +--- +id: grafana-alertcheck-advanced +title: Check budget and scheduling +sidebar_label: Budget and scheduling +sidebar_position: 2 +description: Why grafana-alertcheck schedules per rule, how the request budget works, and why it never queries state history. +--- + +# Check budget and scheduling + +## Per-rule schedules, never a global cycle + +Each rule polls at its **own** cadence, `--poll-interval` (default: half the rule's own evaluation interval). There is deliberately no single global minimum-interval cycle. + +One rule at `intervalSeconds=10` beside twenty at `300` keeps a 5 s cadence for itself and 150 s for the other twenty — not a 5 s cycle for all of them, which would be a 60× request bloat at ~1.8 s per request and would fail to start on a reasonable fleet. + +The scheduler staggers each rule's initial next-due time across its cadence, and serves due rules **earliest-due-first**, so a tight rule never queues behind slack ones. + +## The check budget + +The gate records one observation of every rule up front and checks the schedule against those **measured** latencies (payload sizes vary ~230× across rules, so a fixed estimate is meaningless). It errors at start — before waiting — if any of three conditions hold: + +- **Utilization** — total request rate exceeds `--concurrency`. +- **Per-rule** — one rule's request can't fit its own cadence. +- **Burst bound** — the slowest request exceeds the fleet's tightest cadence, which can open a mid-run gap. + +The error names the three levers only: raise `--concurrency`, raise `--poll-interval`, or watch fewer alerts. It never prescribes a single interval. + +## Why the gate never queries state history + +Querying Grafana's alert state history after the fact fails closed *in the wrong direction* — it returns "pass" when the truth is unknown: + +- History stores **transitions**, not states. An alert firing through the whole window has its only record *before* the window. +- The annotations API **does not serve Loki-backed history** at all. +- An empty result is indistinguishable from a healthy one: no alert fired, the backend differs, retention removed data, the token lacked permission — all look identical. +- There is **no coverage signal** — nothing proves the history is complete to time T. +- Artifact transitions (`Paused`, `RuleDeleted`, `Updated`, `MissingSeries`) look like recoveries. + +Instead, `watch` records its own evidence live and the log becomes the source of truth. The trade-off: the gate can miss an episode shorter than a rule's poll interval, though `activeAt` still surfaces sub-interval onsets for instances still active at a poll. + +A corollary of recording fresh: there is no replay. Re-running a failed job is a new deploy with a new `from` and a new recording — never a re-classification of old evidence. diff --git a/grafana-alertcheck/docs/architecture.md b/grafana-alertcheck/docs/architecture.md new file mode 100644 index 000000000..ca493220e --- /dev/null +++ b/grafana-alertcheck/docs/architecture.md @@ -0,0 +1,68 @@ +--- +id: grafana-alertcheck-architecture +title: Architecture +sidebar_label: Architecture +sidebar_position: 3 +description: The design invariants, pure-function seam, and recorder lifecycle of grafana-alertcheck, for maintainers. +--- + +# Architecture + +This page documents the invariants and seams a maintainer must not break. It exists because most of them are the difference between a gate that fails closed and one that silently passes broken windows. + +## Fail-closed invariants + +The gate must stop the release if it cannot get an answer. Every rule below is a specific instance of that: + +- **An error is never a pass.** A pass is exactly `len(Violations) == 0 && err == nil`. Every error path leaves `err` non-nil, and the CLI maps that to exit `2` unconditionally. +- **Inability beats violation.** Any `unobservable` rule is exit `2`, even alongside a real violation found first. +- **Absent never means normal.** An instance that leaves the bad set is looked up in the *same* response: present as `normal` → cleared; absent (or `MissingSeries`) → vanished (a discontinuity, not a recovery). +- **Staleness is absolute.** `grafana_now − lastEvaluation` is compared against a threshold, never "did it increase since the last poll" — a delta check reports stale on ~half the polls of a healthy rule. +- **`grafana_now` is the response `Date` header.** Never the runner clock, in any comparison against a Grafana timestamp. +- **No early exit.** `check` collects to `to + transitionGrace` before classifying once. +- **No replay.** No run-id key, no artifact download, no state between attempts. A retry is a new deploy. + +## The pure-function seam + +All correctness lives in two phases written as **pure functions** over a flat list of polls — no HTTP, no files, no clock, no goroutines: + +``` +HTTP ──> Source ──> []StateRule ──> reduce ──> []Poll ──> proveCoverage ──> decide ──> Result + │ + JSONL log ──> ReadLog ──┘ +``` + +- `proveCoverage` (the nine coverage checks) and `decide` (the instance timelines and outcomes) are pure; tests drive them with `[]Poll` literals and a fake `Clock`, with no sleeping or fixture server. +- `Check`/`Watch` are I/O shells: HTTP, signals, the pidfile, file reads, the countdown print. The only test doubles needed are the `Source` and `Clock` interfaces. +- `Policy` is the narrowed view of `Config` that reaches the pure layer — classification knobs and the window, no URL and no token. The token must never cross that line, which is the cheapest guarantee it never lands in an error string or a result. + +## Strict parsing as the version guard + +Both API responses are parsed strictly: a missing or unparseable **required** field (`health`, `state`, `lastEvaluation`, `interval`) is an error, never a zero value. Optional keys (`alerts`, `totals`, `labels`, `keepFiringFor`) are absent-tolerant, and unknown keys are ignored — so Grafana can add fields without breaking the parser, but removing one fails loudly. + +This, plus the declared supported range (Grafana >= 13.0.0, < 14.0.0), is how a deprecation or schema change is caught instead of silently misread. + +## The recorder lifecycle + +`watch` detaches a background recorder so observation survives the step boundary: + +1. Parent resolves names, writes the header, observes every non-paused rule once, checks the budget. +2. Parent re-execs itself as the child (`--daemon-child`) under a new session/process group, stdout/stderr to the daemon log. +3. Child re-reads the header, reopens the log `O_APPEND`, takes the exclusive `flock`, and writes one readiness byte on `--ready-fd`. +4. Parent writes the pidfile **after** the readiness report, then returns. + +Two authorities, only one of which is evidence: + +- The **pidfile** says a recording ever started (written only after ready, removed on failure). It can go stale — a pid gets reused. +- The **flock** says a writer exists *now*. The kernel drops it on exit, so the lock is always authoritative. + +On a clean stop (SIGTERM/SIGINT/`--until`) the child finishes the in-flight write, appends the `stopped` sentinel, fsyncs, and exits. A hard error writes no sentinel — so a recorder that died reads exactly like a coverage gap, because it is one. + +`check` signals via the pidfile, waits for the **lock** to release (never the pid), and only then reads the log once. Reading while a writer can still append can only produce a shorter window than was recorded. + +## The log is the source of truth + +`watch` records raw evidence, so nothing trusts a state that could become unreachable. Two consequences a maintainer must preserve: + +- The **header is authoritative for recording facts** (the cadence actually used, the URL, the alert set); the ruler API is authoritative for **rule facts** (`for`, `intervalSeconds`, kind). `check` always re-resolves definitions fresh and never reconstructs them from the header — the header duplicates `for`/`interval` only so the uploaded artifact is self-describing. +- The **cadence authority** is the header's `poll_every_seconds`, not the definitions. Re-deriving it would compare gaps recorded at an override cadence against default-cadence thresholds — fail-open in the faster-override direction. diff --git a/grafana-alertcheck/docs/how-alerts-are-evaluated.md b/grafana-alertcheck/docs/how-alerts-are-evaluated.md new file mode 100644 index 000000000..7c9cb3b6a --- /dev/null +++ b/grafana-alertcheck/docs/how-alerts-are-evaluated.md @@ -0,0 +1,85 @@ +--- +id: grafana-alertcheck-evaluation +title: How alerts are evaluated +sidebar_label: How alerts are evaluated +sidebar_position: 1 +description: The verdict model, instance timelines, and coverage proof behind grafana-alertcheck. +--- + +# How alerts are evaluated + +Both `watch`+`check` (recorder mode) and `check` alone (single-step mode) converge on the same input: a flat list of polls. Everything below runs over that list; the mode only changes where the polls came from. + +## Instance states + +Grafana reports instance states in two vocabularies (`Alerting`/`Normal` at instance level, `firing`/`inactive` at rule level). The gate normalizes every instance to one canonical set: + +| Canonical | Meaning | +| --------- | ------- | +| `normal` | Healthy | +| `firing` | The condition is true and `for` has elapsed | +| `pending` | The condition is true, `for` has not elapsed | +| `nodata` | The query returned no series (synthetic instance) | +| `error` | The query failed (synthetic instance) | + +A rule's **rule-level** `state` and `health` are kept verbatim and only reported — they are never classified. The **instance** state is what the classifier reasons about. + +A "bad" instance is one whose canonical state is in `--states` (default `firing`). `pending` and `nodata` are excluded by default. + +## Verdict model + +For each instance the gate builds a timeline of bad spans over `[from, to]`, then takes the worst outcome across a rule's instances as the rule's outcome. + +| Outcome | Shape | Exit | +| ------- | ----- | ---- | +| `clean` | Good throughout, observed throughout | → 0 | +| `newly_bad` | Entered a bad state **inside** the window | → 1 | +| `persistently_bad` | Bad at `from`, still bad at `to` | → 1 | +| `recovered` | Bad at `from`, cleared before `to`, stayed clear | → 0 | +| `flapping` | Cleared, then became bad again | → 1 | +| `skipped` | Paused **before** the window opened | reported, not observable | +| `unobservable` | Coverage gap / sustained `health=error` / stale / absent | → 2 | + +`recovered` has **no deadline** — an alert that clears at minute 58 of a 60-minute window still passes. The total bad time is reported as `BadFor`; the removed deadline is replaced by that measured value rather than a derived limit. + +### Preexisting policy + +For an instance already bad when `from` opened, `--preexisting` decides: + +- `fail-unless-recovered` (default) — clears and stays clear → pass; never clears → fail. +- `fail` — any preexisting instance fails, recovered or not. +- `ignore` — preexisting instances are disregarded; only new episodes fail. + +## Cleared vs vanished + +When an instance leaves the bad set, the gate looks it up **in the same response**: + +- Present as `normal` → `cleared` (a real recovery). +- Absent, or present as `normal (MissingSeries)` → `vanished` (a discontinuity, **not** a recovery). + +A vanished instance that was bad stays `persistently_bad`. A metric that stops being emitted is not evidence of health — this is deliberate and can surprise users whose fix is to remove a metric rather than drive it to a good value. + +## Coverage proof + +Before classifying, `check` must **prove** continuous coverage of `[from, to]` for each alert. Nine checks run; any failure makes the rule `unobservable`: + +1. **Sentinel** — a clean recorder stop, timestamped at or after `to + transitionGrace`. A recorder that died mid-window looks exactly like a coverage gap and is one. +2. **`from` bounds** — `from` earlier than the recording start is unprovable. +3. **Heartbeat gap** — any gap larger than `maxGap` (= 2 × poll cadence) inside the window. Data at both ends with a hole between is not enough. +4. **`health=error`** — a contiguous run longer than `healthGrace` consumes coverage; a short blip is a note. +5. **`health=nodata`** — a note, never fatal (unless `--nodata-is-unobservable`). +6. **Liveness** — `grafana_now − lastEvaluation` must not exceed `evalStaleAfter`. This is an **absolute** check, never a "did it increase since the last poll" delta. +7. **In-window pause** — a poll reporting `isPaused` mid-window is `unobservable` (the primary pause detector). +8. **Rule absent** — an authoritative `2xx` with no matching rule. +9. **`KeepLast`** — a note naming a stale-state blind spot. + +## Health: `error` vs `nodata` + +- `health=error` means the query **failed** — a malfunction. Sustained past `healthGrace`, it makes the rule `unobservable`. +- `health=nodata` means the query **ran and returned no series** — indistinguishable from a quiet system. It is not fatal by default; most of a fleet runs `no_data_state: OK`. + +## The drain wait and `transitionGrace` + +A condition that arises just before `to` becomes `firing` only at the first evaluation after its `for` elapses. `transitionGrace` (derived from the watched rules' `for` values) extends the classification bound past `to` so such a surfacing condition is caught. After collection, a **drain wait** polls until each rule has evaluated through `to + transitionGrace` (bounded by `drainTimeout`); a rule that never does is `unobservable`. + +Run time = `(to − from) + transitionGrace + drainTimeout`. This is printed at start, and the grace is warned about when it exceeds a quarter of the window — the window may be too short for the alert's `for`. diff --git a/grafana-alertcheck/docs/index.md b/grafana-alertcheck/docs/index.md new file mode 100644 index 000000000..95d413772 --- /dev/null +++ b/grafana-alertcheck/docs/index.md @@ -0,0 +1,87 @@ +--- +id: grafana-alertcheck-index +title: Grafana Alertcheck +sidebar_label: Overview +sidebar_position: 0 +description: A CD quality gate that bookends a release with alert-state observation and answers whether any watched Grafana alert was bad during the release window. +--- + +# Grafana Alertcheck + +`grafana-alertcheck` is a CD quality gate for Grafana alerts. It bookends a release with two commands — `watch` (record) and `check` (classify) — and answers one question: + +> A release finished at time T. Was any of these Grafana alerts in a bad state during the next N minutes? + +The contract is `watch → your work → check`. Between the two you run whatever you want (deploy, tests, migration); the gate only observes, then classifies. + +It **fails closed**: if it cannot get an answer, it stops the release. It never passes an unproven window. + +## How it works, in one paragraph + +`watch` starts a background recorder that polls each named alert and appends snapshots to a JSONL log. Your work then emits two RFC3339 timestamps — `from` (when the change landed) and `to` (when the work ended). `check` proves continuous coverage of `[from, to]`, builds a state timeline per alert, classifies it, and exits `0`, `1`, or `2`. + +## Install + +```bash +go install github.com/smartcontractkit/chainlink-testing-framework/grafana-alertcheck/cmd@latest +``` + +Connection details come from the environment — the token is env-only, never a flag: + +```bash +export GRAFANA_URL=https://grafana.example.com +export GRAFANA_TOKEN=… +``` + +Requires Grafana >= 13.0.0 and < 14.0.0. Outside that range the gate exits `2`. + +## Quickstart — recorder mode + +```bash +grafana-alertcheck watch --out /tmp/run.jsonl --alerts alerts.txt +./deploy.sh # emits deployed_at= when the rollout is stable +./verify.sh # emits finished_at= when the work is done +grafana-alertcheck check --in /tmp/run.jsonl --from "$deployed_at" --to "$finished_at" +``` + +`alerts.txt` holds one alert name per line. See [Naming alerts](./reference/cli#naming-alerts). + +`watch` returns only after the recorder has observed every named, non-paused alert once and reported ready — so auth, name-resolution, and parse failures surface **before** your deploy runs. + +## Quickstart — single-step mode + +Skip the recorder and observe the window inline, from inside `check` itself: + +```bash +grafana-alertcheck check --alerts alerts.txt --to "$finished_at" +``` + +In single-step mode the window starts at `check`'s first observation; if you give no `--from`, the interval before that first observation is declared as a blind spot with a warning (not an error). + +## Exit codes + +| Code | Meaning | +| ---- | ------- | +| `0` | Pass — no violations | +| `1` | Violations (including a paused-rule-only `--min-observed` shortfall) | +| `2` | The gate could not check — config, auth, resolution, a coverage gap, health/staleness, the drain limit, transport, … | + +An error is never a pass: `2` wins over any violation found alongside it. + +## Common surprises + +- A **paused rule fails by default** — even one someone else paused. Use `--allow-paused`. +- A fix that **stops emitting a metric is not a recovery** — the instance vanishes, which is a discontinuity, not health. +- The gate checks alert **state and health**, not notification delivery — a silenced alert that still fires fails. +- `recovered` has **no deadline** — a bad-at-`from` alert that clears by `to` passes; set `--preexisting fail` to forbid it. +- A **retry is a new deploy**, not a replay — re-running the job re-records against a new `from`. +- `watch` and `check` must run in **one job, one runner, one filesystem** — nothing persists across jobs or attempts. +- The gate **never exits early** — a violation at minute 2 still holds the runner to `to + transitionGrace + drainTimeout`; size the job timeout to the planned run time the gate prints at start. + +## More + +- [How alerts are evaluated](./how-alerts-are-evaluated) — the verdict model and coverage proof +- [Check budget and scheduling](./advanced) — why the schedule and budget look the way they do, and why history isn't queried +- [Architecture](./architecture) — design invariants and the recorder lifecycle, for maintainers +- [CLI reference](./reference/cli) — every subcommand and flag +- [Log format](./reference/log-format) — the JSONL log schema, for debugging artifacts diff --git a/grafana-alertcheck/docs/reference/_category_.yaml b/grafana-alertcheck/docs/reference/_category_.yaml new file mode 100644 index 000000000..2e5d50946 --- /dev/null +++ b/grafana-alertcheck/docs/reference/_category_.yaml @@ -0,0 +1,8 @@ +position: 3 +label: Reference +collapsible: true +collapsed: false +link: + type: generated-index + slug: /platform-services/devex/cicd/grafana-alertcheck/reference + description: 'CLI reference for grafana-alertcheck.' diff --git a/grafana-alertcheck/docs/reference/cli.md b/grafana-alertcheck/docs/reference/cli.md new file mode 100644 index 000000000..1530c508d --- /dev/null +++ b/grafana-alertcheck/docs/reference/cli.md @@ -0,0 +1,91 @@ +--- +id: grafana-alertcheck-cli +title: CLI reference +sidebar_label: CLI reference +sidebar_position: 0 +description: Full reference for the grafana-alertcheck CLI: watch, check, list, environment, naming, and output. +--- + +# CLI reference + +``` +grafana-alertcheck +``` + +Connection details are always from the environment: `GRAFANA_URL` and `GRAFANA_TOKEN`. The token is never a flag and never logged. + +## `list` + +Lists every rule from the ruler endpoint — kind, folder, group, title, uid. Useful to check auth and to find `uid:` names. + +```bash +grafana-alertcheck list +``` + +## `watch` — record + +```bash +grafana-alertcheck watch --out [--pidfile F] [--daemon-log F] \ + --alerts [--folder F] [--poll-interval D] [--concurrency N] [--until RFC3339] +``` + +| Flag | Default | Meaning | +| ---- | ------- | ------- | +| `--out` | — | JSONL log path (required) | +| `--pidfile` | `.pid` | Where the recorder's pid is written | +| `--daemon-log` | `.daemon.log` | stdout/stderr sink for the detached recorder | +| `--alerts` | — | File of alert names, one per line, or `-` for stdin (required) | +| `--folder` | — | Default folder to scope unqualified names | +| `--poll-interval` | half the rule's interval | Override every rule's cadence (never clamped) | +| `--concurrency` | `1` | Max concurrent requests to Grafana | +| `--until` | run until signalled | Optional hard stop | + +`watch` writes the header, observes every non-paused rule once, checks the budget, then detaches a background recorder and returns. Recording is **unfiltered** — there is no `--states` here, so the same log can be re-classified later under different `--states` without re-recording. + +## `check` — classify + +```bash +grafana-alertcheck check [--in ] [--pidfile F] --from RFC3339 --to RFC3339 \ + [--alerts ...] [--folder F] [--states ...] [--preexisting ...] [--min-observed N] \ + [--allow-paused] [--nodata-is-unobservable] [--concurrency N] [--output json] +``` + +| Flag | Default | Meaning | +| ---- | ------- | ------- | +| `--in` | — | Log recorded by `watch`; empty selects single-step mode | +| `--pidfile` | `.pid` | Recorder to stop before reading `--in` | +| `--from` | see below | Moment the deploy finished | +| `--to` | — | End of the window (required) | +| `--alerts` | — | Required **without** `--in`; refused **with** `--in` | +| `--states` | `firing` | Comma-separated bad states: `firing,pending,nodata,error` | +| `--preexisting` | `fail-unless-recovered` | `fail-unless-recovered` \| `fail` \| `ignore` | +| `--min-observed` | every resolved rule | Minimum rules that must be observed | +| `--allow-paused` | `false` | Don't count pre-window-paused rules against `--min-observed` | +| `--nodata-is-unobservable` | `false` | Treat sustained `health=nodata` as unobservable | +| `--concurrency` | `1` | Max concurrent requests | +| `--output` | `table` | `json` also writes the machine-readable result to stdout | + +`--from` and `--to` are RFC3339 with an explicit offset and must come from your work — `from` from the deploy step, `to` from the step that finishes. In recorder mode an absent `--from` is a hard error; in single-step mode it falls back (with a warning) to the start of the step. + +## Naming alerts + +Alert names take one of four forms: + +| Form | Meaning | +| ---- | ------- | +| `HighErrorRate` | Title only, scoped by `--folder` | +| `Platform/HighErrorRate` | Folder + title | +| `Platform/api/HighErrorRate` | Folder + group + title (always unique) | +| `uid:abc123` | Exact uid (present on both endpoints) | + +Datasource-managed and recording rules are refused with a specific error. A name matching multiple rules errors listing every candidate with the copyable `Folder/Group/Title` and its `uid:` form. A no-match errors with case-insensitive substring suggestions and points at `list`. Duplicate names that resolve to the same uid collapse to one (a note, not an error). + +## Output and exit codes + +The human table goes to **stderr**: `RESULTS` (one row per rule), `VIOLATIONS` (one per violation), and `THRESHOLDS` (each rule's `maxGap`/`healthGrace`/`evalStaleAfter` plus global `transitionGrace`/`drainTimeout` and the largest measured clock skew). `--output json` writes the result to stdout. + +| Code | Meaning | +| ---- | ------- | +| `0` | Pass | +| `1` | Violations | +| `2` | Could not check — every library error, never a pass | diff --git a/grafana-alertcheck/docs/reference/log-format.md b/grafana-alertcheck/docs/reference/log-format.md new file mode 100644 index 000000000..66e071770 --- /dev/null +++ b/grafana-alertcheck/docs/reference/log-format.md @@ -0,0 +1,98 @@ +--- +id: grafana-alertcheck-log-format +title: Log format +sidebar_label: Log format +sidebar_position: 1 +description: The JSONL log schema written by watch and read by check, for debugging the forensic artifact. +--- + +# Log format + +`watch` records evidence to a JSONL log — one JSON object per line. A poll record *is* the heartbeat; there is no separate heartbeat type. + +## Record types + +Exactly three: + +| `type` | Meaning | +| ------ | ------- | +| `header` | Line 1 — identity and the alert set | +| `poll` | One reduced observation of one rule | +| `stopped` | The sentinel, written on a clean stop only | + +The header must be line 1, appear once, and carry `schema_version` `1` (any other value is a read error). Any unparseable line — including the last, or one after the sentinel — makes the log unreadable: a truncated log is evidence the recorder was killed, and must not pass. + +## Header + +```json +{ + "type": "header", + "schema_version": 1, + "url": "https://grafana.example.com", + "grafana_version": "13.1.0", + "started_at": "2026-09-07T10:00:00Z", + "rules": [ + { + "uid": "rule0000001", + "title": "HighErrorRate", + "folder": "Platform", + "group": "api", + "for_seconds": 300, + "interval_seconds": 60, + "is_paused": false, + "no_data_state": "OK", + "exec_err_state": "OK", + "poll_every_seconds": 30 + } + ] +} +``` + +- `url` and `rules` are the log's identity — `check` validates them against the current environment and a fresh ruler read. +- `is_paused` records the pause state at record start (the moment `skipped` means). +- `poll_every_seconds` is the cadence the recording **actually used** (after any `--poll-interval` override). `check` derives `maxGap` from it, never from `interval_seconds`. +- `for_seconds`, `interval_seconds`, `no_data_state`, `exec_err_state` are forensic only — `check` re-resolves definitions and never reads them back. + +## Poll + +```json +{ + "type": "poll", + "rule_uid": "rule0000001", + "grafana_now": "2026-09-07T10:00:30Z", + "skew_ms": 20, + "skew_bound_ms": 40, + "latency_ms": 123, + "found": true, + "state": "inactive", + "health": "ok", + "last_evaluation": "2026-09-07T10:00:28Z", + "is_paused": false, + "histogram": { "alerting": 0, "normal": 2004 }, + "reasons": { "NoData": 1091 }, + "abnormal": [ { "labels": { "env": "prod" }, "state": "firing", "active_at": "2026-09-07T09:50:00Z", "value": "1.5" } ], + "cleared": [ "env=prod\u0001..." ], + "vanished": [] +} +``` + +Field notes: + +- `grafana_now` is the response's `Date` header — never the runner clock. +- `skew_ms`/`skew_bound_ms` are the per-poll clock-skew estimate and its uncertainty (RTT/2), in milliseconds for compactness only. +- `found: false` is an authoritative `2xx` in which this rule was absent — a transport failure is retried and never becomes a poll. +- `state`, `health`, `last_error` are raw rule-level strings, reporting-only. +- `histogram` is a verbatim copy of the response `totals`; written, never analysed. +- `reasons` counts non-empty instance reasons (`NoData`, `Error`, `KeepLast`, …); composite states stay visible only here. +- `abnormal` holds only instances whose **canonical** state is not `normal`. +- `cleared`/`vanished` are instance keys that left the bad set, resolved against the same response: `cleared` = a real recovery; `vanished` = a discontinuity, never a recovery. + +Instance keys are a sorted `k=v\n` join of labels, so they correlate across polls without hashing. + +## Stopped + +```json +{ "type": "stopped", "at": "2026-09-07T10:10:30Z" } +``` + +`at` is the recorder's own stop time. `check` compares it against `to + transitionGrace`; absent or earlier is `unobservable` — never a pass. diff --git a/grafana-alertcheck/go.mod b/grafana-alertcheck/go.mod index b0c8511ce..d5e0c88be 100644 --- a/grafana-alertcheck/go.mod +++ b/grafana-alertcheck/go.mod @@ -1,3 +1,7 @@ module github.com/smartcontractkit/chainlink-testing-framework/grafana-alertcheck go 1.26.6 + +require github.com/stretchr/testify v1.12.1 + +require go.yaml.in/yaml/v3 v3.0.5 // indirect diff --git a/grafana-alertcheck/go.sum b/grafana-alertcheck/go.sum new file mode 100644 index 000000000..c2336837e --- /dev/null +++ b/grafana-alertcheck/go.sum @@ -0,0 +1,4 @@ +github.com/stretchr/testify v1.12.1 h1:EuwCh5fleGS7H32xRwO3wRGT7DxrDhLAT6FF8MpWDWE= +github.com/stretchr/testify v1.12.1/go.mod h1:MDEgiDPPsNp5cuIrHPPCyornHKgEVbtFUmoNlxoYthg= +go.yaml.in/yaml/v3 v3.0.5 h1:N6y/pJk8buWs9NY5ERU2HSMfm+IuD/OtfdAnq6kESPw= +go.yaml.in/yaml/v3 v3.0.5/go.mod h1:HVTZu1O7/Vkt2N+BFy8Zza+lnLsABggaTM2ZpNIGuKg= diff --git a/grafana-alertcheck/internal/gate/check.go b/grafana-alertcheck/internal/gate/check.go index eab5d931b..ef6543b71 100644 --- a/grafana-alertcheck/internal/gate/check.go +++ b/grafana-alertcheck/internal/gate/check.go @@ -237,6 +237,16 @@ func check(ctx context.Context, cfg Config, src Source) (Result, error) { } logHasHdr = true resolved, notes, err = resolveFromLog(allDefs, earlyHdr, cfg) + if err == nil { + // Fail fast on a bound violation that can't change: StartedAt is + // immutable (line 1), so check 2's backstop still catches any bad + // advisory read — fail closed, never false-pass. Recorder mode only; + // single-step warns-and-passes (see below). + if from.Before(earlyHdr.StartedAt) { + return Result{}, fmt.Errorf("check: `from` %s is before recording started at %s", + from.Format(time.RFC3339), earlyHdr.StartedAt.Format(time.RFC3339)) + } + } } else { resolved, notes, err = Resolve(allDefs, cfg.namedAlerts(), cfg.Folder) } @@ -271,6 +281,16 @@ func check(ctx context.Context, cfg Config, src Source) (Result, error) { } summary, warning := StartupSummary(from, cfg.To, gt) fmt.Fprintln(cfg.Notes, summary) + // MinObserved is printed with the plan, beside "planned run time", rather + // than after it: it is a fact about the run, not a diagnostic. Its default + // is the resolved rule count AFTER duplicate names collapse, which is + // len(resolved) by construction; decide defaults it identically, and it is + // resolved here rather than inferred from the verdict afterwards. + minObserved := cfg.MinObserved + if minObserved == 0 { + minObserved = len(resolved) + } + fmt.Fprintf(cfg.Notes, "min-observed: %d of %d resolved rule(s)\n", minObserved, len(resolved)) if warning != "" { fmt.Fprintf(cfg.Notes, "warning: %s\n", warning) } @@ -321,17 +341,6 @@ func check(ctx context.Context, cfg Config, src Source) (Result, error) { } } - // ---- Apply MinObserved. ----------------------------------------------- - // Its default is the resolved rule count AFTER duplicate names collapse, - // which is len(resolved) by construction. decide defaults it identically; - // it is resolved here as well so the value the run will judge against is - // printed before the wait rather than inferred from the verdict afterwards. - minObserved := cfg.MinObserved - if minObserved == 0 { - minObserved = len(resolved) - } - fmt.Fprintf(cfg.Notes, "min-observed: %d of %d resolved rule(s)\n", minObserved, len(resolved)) - // ---- Collect the evidence. -------------------------------------------- // Collect ONLY. No classification happens here and there is no early exit, // even once a violation is certain: the loop always runs to diff --git a/grafana-alertcheck/internal/gate/check_test.go b/grafana-alertcheck/internal/gate/check_test.go index d13914a8e..780184bd3 100644 --- a/grafana-alertcheck/internal/gate/check_test.go +++ b/grafana-alertcheck/internal/gate/check_test.go @@ -14,6 +14,8 @@ import ( "syscall" "testing" "time" + + "github.com/stretchr/testify/require" ) // The one rule every test in this file watches, unless it says otherwise: a @@ -228,12 +230,8 @@ func TestCheckValidateRejectsBadConfigurations(t *testing.T) { cfg := base() tc.mutate(&cfg) err := cfg.withDefaults().validate() - if err == nil { - t.Fatalf("validate() = nil, want an error containing %q", tc.wantErr) - } - if !strings.Contains(err.Error(), tc.wantErr) { - t.Fatalf("validate() = %q, want it to contain %q", err, tc.wantErr) - } + require.Errorf(t, err, "validate()") + require.Contains(t, err.Error(), tc.wantErr) }) } } @@ -250,12 +248,8 @@ func TestCheckValidateAcceptsAPastToWithALog(t *testing.T) { Clock: newFakeClock(testNow), }.withDefaults() - if err := cfg.validate(); err != nil { - t.Fatalf("validate() = %v, want nil", err) - } - if cfg.PidFile != "log.jsonl.pid" { - t.Errorf("PidFile = %q, want the .pid default", cfg.PidFile) - } + require.NoError(t, cfg.validate()) + require.Equal(t, "log.jsonl.pid", cfg.PidFile) } // --------------------------------------------------------------------------- @@ -270,35 +264,24 @@ func TestCheckSingleStepCleanWindowPasses(t *testing.T) { }) res, err := check(context.Background(), cfg, src) - if err != nil { - t.Fatalf("check() = %v, want nil\nnotes:\n%s", err, notesOf(cfg)) - } + require.NoError(t, err) // A pass is exactly this shape. - if len(res.Violations) != 0 { - t.Fatalf("Violations = %+v, want none", res.Violations) - } - if len(res.Verdicts) != 1 || res.Verdicts[0].Outcome != OutcomeClean { - t.Fatalf("Verdicts = %+v, want one clean verdict", res.Verdicts) - } - if cov := res.Coverage[checkUID]; !cov.Proved || cov.Unobservable { - t.Fatalf("Coverage = %+v, want proved", cov) - } + require.Empty(t, res.Violations) + require.Len(t, res.Verdicts, 1) + require.Equal(t, OutcomeClean, res.Verdicts[0].Outcome) + cov := res.Coverage[checkUID] + require.True(t, cov.Proved) + require.False(t, cov.Unobservable) // The collection loop ran to to+transitionGrace and no further. windowEnd := cfg.To.Add(checkGrace) - if clock.Now().Before(windowEnd) { - t.Errorf("stopped collecting at %s, before to+grace %s", clock.Now(), windowEnd) - } + require.False(t, clock.Now().Before(windowEnd)) // One measurement-pass poll plus one every 30s across the 6-minute // collection, plus the drain wait's own polls. The exact count depends on // the scheduler's random stagger, so assert the order of magnitude a full // window implies rather than an exact number. - if got := src.callCount(checkTitle); got < 12 { - t.Errorf("polled %d times, want at least the ~13 a full 6-minute window at 30s implies", got) - } - if notes := notesOf(cfg); !strings.Contains(notes, "planned run time") { - t.Errorf("the planned run time must be printed at start; notes were:\n%s", notes) - } + require.GreaterOrEqual(t, src.callCount(checkTitle), 12) + require.Contains(t, notesOf(cfg), "planned run time") } // resolve_test.go proves the collapse-note-plus-satisfied-MinObserved path at @@ -315,18 +298,10 @@ func TestCheckSingleStepDuplicateAlertNamesCollapseWithNote(t *testing.T) { }) res, err := check(context.Background(), cfg, src) - if err != nil { - t.Fatalf("check() = %v, want nil\nnotes:\n%s", err, notesOf(cfg)) - } - if len(res.Verdicts) != 1 { - t.Fatalf("Verdicts = %+v, want exactly one — the duplicate must collapse to a single rule", res.Verdicts) - } - if len(res.Violations) != 0 { - t.Fatalf("Violations = %+v, want none: MinObserved must be satisfied by the post-collapse count of 1", res.Violations) - } - if notes := notesOf(cfg); !strings.Contains(notes, "counted once") { - t.Errorf("want the collapse note in the run's own notes; got:\n%s", notes) - } + require.NoError(t, err) + require.Len(t, res.Verdicts, 1, "the duplicate must collapse to a single rule") + require.Empty(t, res.Violations, "MinObserved must be satisfied by the post-collapse count of 1") + require.Contains(t, notesOf(cfg), "counted once") } // A rule with health=error for the whole window is unobservable, exit 2 — @@ -337,9 +312,7 @@ func TestCheckSingleStepDuplicateAlertNamesCollapseWithNote(t *testing.T) { func TestCheckSingleStepContinuousHealthErrorIsUnobservable(t *testing.T) { body := readFixture(t, "state_health_error.json") rules, err := ParseState(body) - if err != nil { - t.Fatalf("ParseState: %v", err) - } + require.NoError(t, err) base := rules[0] def := Definition{ UID: base.UID, Title: base.Title, Folder: base.Folder, Group: base.Group, @@ -365,15 +338,10 @@ func TestCheckSingleStepContinuousHealthErrorIsUnobservable(t *testing.T) { src.defs = []Definition{def} res, err := check(context.Background(), cfg, src) - if err == nil { - t.Fatalf("check() = nil, want an error: continuous health=error must be unobservable\nnotes:\n%s", notesOf(cfg)) - } - if len(res.Verdicts) != 1 || res.Verdicts[0].Outcome != OutcomeUnobservable { - t.Fatalf("Verdicts = %+v, want one unobservable verdict", res.Verdicts) - } - if cov := res.Coverage[def.UID]; cov.Reason != ReasonHealthError { - t.Fatalf("Coverage[%s].Reason = %q, want %q", def.UID, cov.Reason, ReasonHealthError) - } + require.Error(t, err, "continuous health=error must be unobservable") + require.Len(t, res.Verdicts, 1) + require.Equal(t, OutcomeUnobservable, res.Verdicts[0].Outcome) + require.Equal(t, ReasonHealthError, res.Coverage[def.UID].Reason) } // A certain violation does not release the runner early, and it does not stop @@ -391,18 +359,10 @@ func TestCheckSingleStepFiringInstanceReportsWithoutExitingEarly(t *testing.T) { }) res, err := check(context.Background(), cfg, src) - if err != nil { - t.Fatalf("check() = %v, want nil (a violation is exit 1, not an error)", err) - } - if len(res.Violations) != 1 { - t.Fatalf("Violations = %+v, want exactly one", res.Violations) - } - if got := res.Violations[0].Outcome; got != OutcomePersistentlyBad { - t.Errorf("Outcome = %q, want %q", got, OutcomePersistentlyBad) - } - if windowEnd := cfg.To.Add(checkGrace); clock.Now().Before(windowEnd) { - t.Errorf("exited early at %s; collection must run to %s", clock.Now(), windowEnd) - } + require.NoError(t, err, "a violation is exit 1, not an error") + require.Len(t, res.Violations, 1) + require.Equal(t, OutcomePersistentlyBad, res.Violations[0].Outcome) + require.False(t, clock.Now().Before(cfg.To.Add(checkGrace)), "exited early; collection must run to to+grace") } // A newly_bad instance at from+30s gives exit 1, but ONLY after @@ -427,15 +387,11 @@ func TestCheckSingleStepNewOnsetDoesNotExitEarly(t *testing.T) { }) res, err := check(context.Background(), cfg, src) - if err != nil { - t.Fatalf("check() = %v, want nil (a violation is exit 1, not an error)", err) - } - if len(res.Violations) != 1 || res.Violations[0].Outcome != OutcomeNewlyBad { - t.Fatalf("Violations = %+v, want exactly one newly_bad", res.Violations) - } - if windowEnd := cfg.To.Add(checkGrace); clock.Now().Before(windowEnd) { - t.Errorf("exited early at %s; collection must run to %s even for a fresh onset at from+30s", clock.Now(), windowEnd) - } + require.NoError(t, err) + require.Len(t, res.Violations, 1) + require.Equal(t, OutcomeNewlyBad, res.Violations[0].Outcome) + require.False(t, clock.Now().Before(cfg.To.Add(checkGrace)), + "exited early; collection must run to to+grace even for a fresh onset at from+30s") } // An ABSENT `from` in single-step mode (as opposed to recorder mode, which @@ -451,16 +407,9 @@ func TestCheckSingleStepAbsentFromFallsBackToStepStart(t *testing.T) { }) res, err := check(context.Background(), cfg, src) - if err != nil { - t.Fatalf("check() = %v, want nil: an absent `from` in single-step mode is a fallback, not an error\nnotes:\n%s", err, notesOf(cfg)) - } - notes := notesOf(cfg) - if !strings.Contains(notes, "no `from` given") { - t.Errorf("want the step-start fallback note; notes were:\n%s", notes) - } - if !res.From.Equal(testNow) { - t.Errorf("Result.From = %s, want the step-start fallback %s", res.From, testNow) - } + require.NoError(t, err, "an absent `from` in single-step mode is a fallback, not an error") + require.Contains(t, notesOf(cfg), "no `from` given") + require.True(t, res.From.Equal(testNow)) } // In single-step mode an explicit `from` earlier than the first observation is @@ -475,18 +424,13 @@ func TestCheckSingleStepFromBeforeFirstObservationWarnsAndPasses(t *testing.T) { }) res, err := check(context.Background(), cfg, src) - if err != nil { - t.Fatalf("check() = %v, want a pass with a warning\nnotes:\n%s", err, notesOf(cfg)) - } + require.NoError(t, err) notes := notesOf(cfg) - if !strings.Contains(notes, "cannot see [") || !strings.Contains(notes, testNow.Format(time.RFC3339)) { - t.Errorf("want a warning naming the unseen interval; notes were:\n%s", notes) - } + require.Contains(t, notes, "cannot see [") + require.Contains(t, notes, testNow.Format(time.RFC3339)) // The classified window is the clamped one, and Result says so rather than // reporting a window the run never proved. - if !res.From.Equal(testNow) { - t.Errorf("Result.From = %s, want the clamped %s", res.From, testNow) - } + require.True(t, res.From.Equal(testNow)) } // The failure limit was exceeded. The measurement pass succeeds and the @@ -503,15 +447,9 @@ func TestCheckFailClosedOnExhaustedRetries(t *testing.T) { }) res, err := check(context.Background(), cfg, src) - if err == nil { - t.Fatalf("check() = nil, want the collection failure to fail closed") - } - if !strings.Contains(err.Error(), "collect evidence") { - t.Errorf("err = %q, want it to name the collection step", err) - } - if len(res.Violations) != 0 { - t.Errorf("Violations = %+v; an error must never be reported as a verdict", res.Violations) - } + require.Error(t, err, "the collection failure to fail closed") + require.Contains(t, err.Error(), "collect evidence") + require.Empty(t, res.Violations, "an error must never be reported as a verdict") } // The resolution of the definitions failed. Both shapes — the ruler read @@ -523,10 +461,9 @@ func TestCheckFailClosedOnDefinitionResolution(t *testing.T) { src := newCheckSource(nil) src.defsErr = errors.New("502 bad gateway") - if _, err := check(context.Background(), cfg, src); err == nil || - !strings.Contains(err.Error(), "read rule definitions") { - t.Fatalf("check() = %v, want a definitions-read failure", err) - } + _, err := check(context.Background(), cfg, src) + require.Error(t, err) + require.Contains(t, err.Error(), "read rule definitions") }) t.Run("unknown alert name", func(t *testing.T) { @@ -535,10 +472,9 @@ func TestCheckFailClosedOnDefinitionResolution(t *testing.T) { cfg.Alerts = []string{"No Such Rule"} src := newCheckSource(nil) - if _, err := check(context.Background(), cfg, src); err == nil || - !strings.Contains(err.Error(), "no rule matched") { - t.Fatalf("check() = %v, want a no-match failure", err) - } + _, err := check(context.Background(), cfg, src) + require.Error(t, err) + require.Contains(t, err.Error(), "no rule matched") }) } @@ -550,10 +486,9 @@ func TestCheckRefusesUnsupportedGrafanaVersion(t *testing.T) { src := newCheckSource(nil) src.version = "12.4.0" - if _, err := check(context.Background(), cfg, src); err == nil || - !strings.Contains(err.Error(), "unsupported grafana version") { - t.Fatalf("check() = %v, want the version gate to refuse 12.4.0", err) - } + _, err := check(context.Background(), cfg, src) + require.Error(t, err) + require.Contains(t, err.Error(), "unsupported grafana version") } // The budget is checked against the latencies the measurement pass actually @@ -569,13 +504,9 @@ func TestCheckSingleStepRefusesAScheduleThatDoesNotFit(t *testing.T) { }) _, err := check(context.Background(), cfg, src) - if err == nil { - t.Fatalf("check() = nil, want the budget check to refuse the schedule") - } + require.Error(t, err, "the budget check to refuse the schedule") for _, want := range []string{"raising concurrency", "raising poll-interval", "watching fewer alerts"} { - if !strings.Contains(err.Error(), want) { - t.Errorf("err = %q, want it to name the control %q", err, want) - } + require.Contains(t, err.Error(), want) } } @@ -592,9 +523,7 @@ func recordedLog(t *testing.T, dir string, url string, startedAt, start, end, se path := filepath.Join(dir, "log.jsonl") clock := newFakeClock(sentinelAt) w, err := NewWriter(path, clock) - if err != nil { - t.Fatalf("NewWriter: %v", err) - } + require.NoError(t, err) header := Header{ URL: url, GrafanaVersion: "13.1.0", @@ -605,20 +534,14 @@ func recordedLog(t *testing.T, dir string, url string, startedAt, start, end, se PollEverySeconds: checkPollEvery.Seconds(), }}, } - if err := w.WriteHeader(header); err != nil { - t.Fatalf("WriteHeader: %v", err) - } + require.NoError(t, w.WriteHeader(header)) for at := start; !at.After(end); at = at.Add(checkPollEvery) { - if err := w.WritePoll(Poll{ + require.NoError(t, w.WritePoll(Poll{ RuleUID: checkUID, GrafanaNow: at, Found: true, State: "inactive", Health: "ok", LastEvaluation: at.Add(-lastEvalLag), - }); err != nil { - t.Fatalf("WritePoll: %v", err) - } - } - if err := w.Stop(); err != nil { - t.Fatalf("Stop: %v", err) + })) } + require.NoError(t, w.Stop()) return path } @@ -628,21 +551,15 @@ func recordedLog(t *testing.T, dir string, url string, startedAt, start, end, se func deadPid(t *testing.T) int { t.Helper() cmd := exec.Command("/bin/sh", "-c", "exit 0") - if err := cmd.Start(); err != nil { - t.Fatalf("start a throwaway process: %v", err) - } + require.NoError(t, cmd.Start()) pid := cmd.Process.Pid - if err := cmd.Wait(); err != nil { - t.Fatalf("wait for the throwaway process: %v", err) - } + require.NoError(t, cmd.Wait()) return pid } func writePid(t *testing.T, path, contents string) { t.Helper() - if err := os.WriteFile(path, []byte(contents), 0o644); err != nil { - t.Fatalf("write pidfile: %v", err) - } + require.NoError(t, os.WriteFile(path, []byte(contents), 0o644)) } // recorderConfig points check at a recording of [testNow-1m, windowEnd+30s] @@ -671,28 +588,19 @@ func TestCheckRecorderModeCleanWindowPasses(t *testing.T) { // The drain wait is satisfied from the log's own evidence, so the source // must never be asked for a state — asserted by the nil responder. src := newCheckSource(func(title string, _ int) (Observation, error) { - t.Errorf("the drain wait polled %q although the log already proves the evaluations", title) + require.Fail(t, fmt.Sprintf("the drain wait polled %q although the log already proves the evaluations", title)) return Observation{}, errors.New("unexpected poll") }) res, err := check(context.Background(), cfg, src) - if err != nil { - t.Fatalf("check() = %v, want nil\nnotes:\n%s", err, notesOf(cfg)) - } - if len(res.Violations) != 0 { - t.Fatalf("Violations = %+v, want none", res.Violations) - } - if len(res.Verdicts) != 1 || res.Verdicts[0].Outcome != OutcomeClean { - t.Fatalf("Verdicts = %+v, want one clean verdict", res.Verdicts) - } - if res.GrafanaVersion != "13.1.0" { - t.Errorf("GrafanaVersion = %q, want the recorded one", res.GrafanaVersion) - } + require.NoError(t, err) + require.Empty(t, res.Violations) + require.Len(t, res.Verdicts, 1) + require.Equal(t, OutcomeClean, res.Verdicts[0].Outcome) + require.Equal(t, "13.1.0", res.GrafanaVersion) // The collection loop still waited out to+transitionGrace even though the // recorder had already finished. - if clock.Now().Before(windowEnd) { - t.Errorf("returned at %s, before to+grace %s", clock.Now(), windowEnd) - } + require.False(t, clock.Now().Before(windowEnd)) } // The identity of the log is not correct. The check runs against the header @@ -707,12 +615,9 @@ func TestCheckFailClosedOnWrongLogIdentity(t *testing.T) { clock := newVirtualClock(testNow) cfg := recorderConfig(t, clock, logPath) _, err := check(context.Background(), cfg, newCheckSource(nil)) - if err == nil || !strings.Contains(err.Error(), "log identity") { - t.Fatalf("check() = %v, want a log-identity failure", err) - } - if !clock.Now().Equal(testNow) { - t.Errorf("the identity check waited out the window (now %s); it must fail before the wait", clock.Now()) - } + require.Error(t, err) + require.Contains(t, err.Error(), "log identity") + require.True(t, clock.Now().Equal(testNow), "it must fail before the wait") }) t.Run("rule no longer resolves", func(t *testing.T) { @@ -726,10 +631,56 @@ func TestCheckFailClosedOnWrongLogIdentity(t *testing.T) { src.defs = []Definition{{UID: "somebody-else", Title: "Other", Kind: KindGrafanaManaged, IntervalSeconds: 60}} _, err := check(context.Background(), cfg, src) - if err == nil || !strings.Contains(err.Error(), "log identity") { - t.Fatalf("check() = %v, want a log-identity failure", err) - } + require.Error(t, err) + require.Contains(t, err.Error(), "log identity") + }) +} + +// `from` before the recording's StartedAt is statically knowable from the +// header (immutable line 1), so check fails closed on it BEFORE the window's +// wait — exactly like the identity check above — rather than surfacing a +// from_before_record verdict only after the drain. +func TestCheckFailFastWhenFromPrecedesRecordStart(t *testing.T) { + dir := t.TempDir() + startedAt := testNow.Add(time.Minute) // the recording opened a minute AFTER `from` + windowEnd := testNow.Add(5*time.Minute + checkGrace) + // The poll range is irrelevant to the assertion: the fail-fast reads + // StartedAt from the header alone, before any polling would matter. + logPath := recordedLog(t, dir, "https://grafana.example.com", + startedAt, startedAt, windowEnd.Add(30*time.Second), windowEnd.Add(30*time.Second), 0) + + clock := newVirtualClock(testNow) + cfg := recorderConfig(t, clock, logPath) // From = testNow, before StartedAt + _, err := check(context.Background(), cfg, newCheckSource(nil)) + require.Error(t, err) + require.Contains(t, err.Error(), "before recording started") + require.True(t, clock.Now().Equal(testNow), "it must fail before the wait") +} + +// A whole-second `from` in the same second as the recording's sub-second +// StartedAt is not a blind interval: the whole-second comparison lets the run +// proceed to a clean pass instead of the fail-fast above. +func TestCheckRecorderModeFromSameSecondAsStartedAtPasses(t *testing.T) { + dir := t.TempDir() + windowEnd := testNow.Add(5*time.Minute + checkGrace) + // StartedAt is 500ms after `from` (testNow via recorderConfig) — the same + // whole second. Polls still cover the whole window. + logPath := recordedLog(t, dir, "https://grafana.example.com", + testNow.Add(500*time.Millisecond), testNow.Add(-time.Minute), windowEnd.Add(30*time.Second), windowEnd.Add(30*time.Second), 0) + writePid(t, logPath+".pid", fmt.Sprintf("%d\n", deadPid(t))) + + clock := newVirtualClock(testNow.Add(time.Minute)) + cfg := recorderConfig(t, clock, logPath) + src := newCheckSource(func(title string, _ int) (Observation, error) { + require.Fail(t, fmt.Sprintf("the drain wait polled %q although the log already proves the evaluations", title)) + return Observation{}, errors.New("unexpected poll") }) + + res, err := check(context.Background(), cfg, src) + require.NoError(t, err) + require.Empty(t, res.Violations) + require.Len(t, res.Verdicts, 1) + require.Equal(t, OutcomeClean, res.Verdicts[0].Outcome) } // The coverage proof failed: a hole in the middle of the recording is not @@ -740,42 +691,28 @@ func TestCheckFailClosedOnCoverageGap(t *testing.T) { path := filepath.Join(dir, "log.jsonl") clock := newFakeClock(windowEnd.Add(30 * time.Second)) w, err := NewWriter(path, clock) - if err != nil { - t.Fatalf("NewWriter: %v", err) - } - if err := w.WriteHeader(Header{ + require.NoError(t, err) + require.NoError(t, w.WriteHeader(Header{ URL: "https://grafana.example.com", GrafanaVersion: "13.1.0", StartedAt: testNow.Add(-time.Minute), Rules: []LoggedRule{{UID: checkUID, Title: checkTitle, IntervalSeconds: 60, PollEverySeconds: checkPollEvery.Seconds()}}, - }); err != nil { - t.Fatalf("WriteHeader: %v", err) - } + })) for at := testNow.Add(-time.Minute); !at.After(windowEnd.Add(30 * time.Second)); at = at.Add(checkPollEvery) { // A three-minute hole in the middle of the window. if at.After(testNow.Add(time.Minute)) && at.Before(testNow.Add(4*time.Minute)) { continue } - if err := w.WritePoll(Poll{ + require.NoError(t, w.WritePoll(Poll{ RuleUID: checkUID, GrafanaNow: at, Found: true, State: "inactive", Health: "ok", LastEvaluation: at, - }); err != nil { - t.Fatalf("WritePoll: %v", err) - } - } - if err := w.Stop(); err != nil { - t.Fatalf("Stop: %v", err) + })) } + require.NoError(t, w.Stop()) writePid(t, path+".pid", fmt.Sprintf("%d\n", deadPid(t))) cfg := recorderConfig(t, newVirtualClock(testNow), path) res, err := check(context.Background(), cfg, newCheckSource(nil)) - if err == nil { - t.Fatalf("check() = nil, want the coverage gap to fail closed") - } - if got := res.Coverage[checkUID].Reason; got != ReasonHeartbeatGap { - t.Errorf("Reason = %q, want %q", got, ReasonHeartbeatGap) - } - if got := res.Verdicts[0].Outcome; got != OutcomeUnobservable { - t.Errorf("Outcome = %q, want %q", got, OutcomeUnobservable) - } + require.Error(t, err, "the coverage gap to fail closed") + require.Equal(t, ReasonHeartbeatGap, res.Coverage[checkUID].Reason) + require.Equal(t, OutcomeUnobservable, res.Verdicts[0].Outcome) } // An episode fully between the deploy and the start of the check: recorder @@ -790,36 +727,25 @@ func TestCheckRecorderModeFindsAGapImmediatelyAfterTheDeploy(t *testing.T) { path := filepath.Join(dir, "log.jsonl") clock := newFakeClock(windowEnd.Add(30 * time.Second)) w, err := NewWriter(path, clock) - if err != nil { - t.Fatalf("NewWriter: %v", err) - } - if err := w.WriteHeader(Header{ + require.NoError(t, err) + require.NoError(t, w.WriteHeader(Header{ URL: "https://grafana.example.com", GrafanaVersion: "13.1.0", StartedAt: testNow.Add(-time.Minute), Rules: []LoggedRule{{UID: checkUID, Title: checkTitle, IntervalSeconds: 60, PollEverySeconds: checkPollEvery.Seconds()}}, - }); err != nil { - t.Fatalf("WriteHeader: %v", err) - } + })) gapEnd := testNow.Add(3 * time.Minute) // nothing recorded from `from` (testNow) to here for at := gapEnd; !at.After(windowEnd.Add(30 * time.Second)); at = at.Add(checkPollEvery) { - if err := w.WritePoll(Poll{ + require.NoError(t, w.WritePoll(Poll{ RuleUID: checkUID, GrafanaNow: at, Found: true, State: "inactive", Health: "ok", LastEvaluation: at, - }); err != nil { - t.Fatalf("WritePoll: %v", err) - } - } - if err := w.Stop(); err != nil { - t.Fatalf("Stop: %v", err) + })) } + require.NoError(t, w.Stop()) writePid(t, path+".pid", fmt.Sprintf("%d\n", deadPid(t))) cfg := recorderConfig(t, newVirtualClock(testNow), path) res, err := check(context.Background(), cfg, newCheckSource(nil)) - if err == nil { - t.Fatalf("check() = nil, want exit 2: a hole right after the deploy hides whatever happened there just as much as one in the middle") - } - if len(res.Verdicts) != 1 || res.Verdicts[0].Outcome != OutcomeUnobservable { - t.Fatalf("Verdicts = %+v, want one unobservable verdict, never clean", res.Verdicts) - } + require.Error(t, err, "a hole right after the deploy hides whatever happened there") + require.Len(t, res.Verdicts, 1) + require.Equal(t, OutcomeUnobservable, res.Verdicts[0].Outcome, "never clean") } // The drain limit passed. The recording itself is clean, so this isolates the @@ -846,21 +772,12 @@ func TestCheckFailClosedOnDrainTimeout(t *testing.T) { }) res, err := check(context.Background(), cfg, src) - if err == nil { - t.Fatalf("check() = nil, want the drain limit to fail closed") - } - if got := res.Coverage[checkUID].Reason; got != ReasonDrainTimeout { - t.Errorf("Reason = %q, want %q", got, ReasonDrainTimeout) - } - if got := res.Verdicts[0].Outcome; got != OutcomeUnobservable { - t.Errorf("Outcome = %q, want %q", got, OutcomeUnobservable) - } - if !strings.Contains(res.Verdicts[0].Note, "drain limit") { - t.Errorf("Note = %q, want it to explain the drain limit", res.Verdicts[0].Note) - } - if waited := clock.Now().Sub(windowEnd); waited < checkDrainLimit { - t.Errorf("gave up after %s of drain wait, want the full %s", waited, checkDrainLimit) - } + require.Error(t, err, "the drain limit to fail closed") + require.Equal(t, ReasonDrainTimeout, res.Coverage[checkUID].Reason) + require.Equal(t, OutcomeUnobservable, res.Verdicts[0].Outcome) + require.Contains(t, res.Verdicts[0].Note, "drain limit") + require.GreaterOrEqual(t, clock.Now().Sub(windowEnd), checkDrainLimit, + "the rule never evaluates through the window, so the drain wait must run its full limit") } // A rule the state endpoint no longer serves is knowable on the FIRST drain @@ -884,18 +801,10 @@ func TestCheckDrainWaitNamesADeletedRuleAtOnce(t *testing.T) { }) res, err := check(context.Background(), cfg, src) - if err == nil { - t.Fatalf("check() = nil, want a deleted rule to fail closed") - } - if got := res.Coverage[checkUID].Reason; got != ReasonRuleAbsent { - t.Errorf("Reason = %q, want %q — the fault, not the wait", got, ReasonRuleAbsent) - } - if got := src.callCount(checkTitle); got != 1 { - t.Errorf("polled %d times, want exactly 1: the absence is knowable on the first poll", got) - } - if waited := clock.Now().Sub(windowEnd); waited >= checkDrainLimit { - t.Errorf("spent %s in the drain wait, want it to conclude at once", waited) - } + require.Error(t, err, "a deleted rule to fail closed") + require.Equal(t, ReasonRuleAbsent, res.Coverage[checkUID].Reason, "the fault, not the wait") + require.Equal(t, 1, src.callCount(checkTitle), "the absence is knowable on the first poll") + require.Less(t, clock.Now().Sub(windowEnd), checkDrainLimit) } // --------------------------------------------------------------------------- @@ -909,18 +818,14 @@ func pausedAfterWindowLog(t *testing.T, dir string, firesAt time.Time, end, sent t.Helper() path := filepath.Join(dir, "log.jsonl") w, err := NewWriter(path, newFakeClock(sentinelAt)) - if err != nil { - t.Fatalf("NewWriter: %v", err) - } - if err := w.WriteHeader(Header{ + require.NoError(t, err) + require.NoError(t, w.WriteHeader(Header{ URL: "https://grafana.example.com", GrafanaVersion: "13.1.0", StartedAt: testNow.Add(-time.Minute), Rules: []LoggedRule{{ UID: checkUID, Title: checkTitle, IntervalSeconds: 60, IsPaused: false, PollEverySeconds: checkPollEvery.Seconds(), }}, - }); err != nil { - t.Fatalf("WriteHeader: %v", err) - } + })) firing := Instance{ Labels: map[string]string{"alertname": checkTitle, "instance": "a"}, State: StateFiring, @@ -932,13 +837,9 @@ func pausedAfterWindowLog(t *testing.T, dir string, firesAt time.Time, end, sent p.State = "firing" p.Abnormal = []Instance{firing} } - if err := w.WritePoll(p); err != nil { - t.Fatalf("WritePoll: %v", err) - } - } - if err := w.Stop(); err != nil { - t.Fatalf("Stop: %v", err) + require.NoError(t, w.WritePoll(p)) } + require.NoError(t, w.Stop()) return path } @@ -970,19 +871,13 @@ func pausedAfterWindowCheck(t *testing.T, allowPaused bool) (Result, error, Conf // that was active at record start is classified, whatever its pause state is // by the time check resolves the definitions. func TestCheckPausingARuleAfterTheWindowDoesNotMakeItSkipped(t *testing.T) { - res, err, cfg := pausedAfterWindowCheck(t, false) - if err != nil { - t.Fatalf("check() = %v, want a classified verdict\nnotes:\n%s", err, notesOf(cfg)) - } - if got := res.Verdicts[0].Outcome; got != OutcomeNewlyBad { - t.Fatalf("Outcome = %q, want %q: the rule was active for the whole window and fired inside it", got, OutcomeNewlyBad) - } - if len(res.Violations) != 1 || res.Violations[0].Outcome != OutcomeNewlyBad { - t.Fatalf("Violations = %+v, want the firing reported", res.Violations) - } - if strings.Contains(res.Verdicts[0].Note, "paused before the window opened") { - t.Errorf("Note = %q, which the log's own polls contradict", res.Verdicts[0].Note) - } + res, err, _ := pausedAfterWindowCheck(t, false) + require.NoError(t, err) + require.Equal(t, OutcomeNewlyBad, res.Verdicts[0].Outcome, + "the rule was active for the whole window and fired inside it") + require.Len(t, res.Violations, 1) + require.Equal(t, OutcomeNewlyBad, res.Violations[0].Outcome) + require.NotContains(t, res.Verdicts[0].Note, "paused before the window opened") } // The regression pin for the loophole this fix closed. Reading skipped from @@ -990,13 +885,9 @@ func TestCheckPausingARuleAfterTheWindowDoesNotMakeItSkipped(t *testing.T) { // skipped free; and a window in which the alert fired reported exit 0. The // default message names --allow-paused, so an operator was led straight to it. func TestCheckAllowPausedCannotExcuseARulePausedAfterItFired(t *testing.T) { - res, err, cfg := pausedAfterWindowCheck(t, true) - if err != nil { - t.Fatalf("check() = %v, want a classified verdict\nnotes:\n%s", err, notesOf(cfg)) - } - if len(res.Violations) == 0 { - t.Fatalf("Violations = none with --allow-paused: the run passed over a window in which the alert fired") - } + res, err, _ := pausedAfterWindowCheck(t, true) + require.NoError(t, err) + require.NotEmpty(t, res.Violations, "the run passed over a window in which the alert fired") } // The other direction, unchanged: a rule the HEADER says was paused when the @@ -1007,23 +898,17 @@ func TestCheckHeaderPausedRuleStaysSkipped(t *testing.T) { windowEnd := testNow.Add(5*time.Minute + checkGrace) path := filepath.Join(dir, "log.jsonl") w, err := NewWriter(path, newFakeClock(windowEnd.Add(30*time.Second))) - if err != nil { - t.Fatalf("NewWriter: %v", err) - } + require.NoError(t, err) // Named in the header, is_paused true, and no poll records at all — the // shape watch writes for a rule paused before the window opened. - if err := w.WriteHeader(Header{ + require.NoError(t, w.WriteHeader(Header{ URL: "https://grafana.example.com", GrafanaVersion: "13.1.0", StartedAt: testNow.Add(-time.Minute), Rules: []LoggedRule{{ UID: checkUID, Title: checkTitle, IntervalSeconds: 60, IsPaused: true, PollEverySeconds: checkPollEvery.Seconds(), }}, - }); err != nil { - t.Fatalf("WriteHeader: %v", err) - } - if err := w.Stop(); err != nil { - t.Fatalf("Stop: %v", err) - } + })) + require.NoError(t, w.Stop()) writePid(t, path+".pid", fmt.Sprintf("%d\n", deadPid(t))) run := func(allowPaused bool) (Result, error) { @@ -1031,30 +916,22 @@ func TestCheckHeaderPausedRuleStaysSkipped(t *testing.T) { cfg.AllowPaused = allowPaused // The definition is unpaused now; the header still decides. src := newCheckSource(func(title string, _ int) (Observation, error) { - t.Errorf("the drain wait polled skipped rule %q", title) + require.Fail(t, fmt.Sprintf("the drain wait polled skipped rule %q", title)) return Observation{}, errors.New("unexpected poll") }) return check(context.Background(), cfg, src) } res, err := run(false) - if err != nil { - t.Fatalf("check() = %v, want exit-1 shape: a skipped rule is a known condition, not an inability", err) - } - if got := res.Verdicts[0].Outcome; got != OutcomeSkipped { - t.Fatalf("Outcome = %q, want %q", got, OutcomeSkipped) - } - if _, ok := res.Coverage[checkUID]; ok { - t.Errorf("Coverage[%s] present, want absent: a skipped rule has no coverage to prove", checkUID) - } - if len(res.Violations) != 1 { - t.Errorf("Violations = %+v, want the MinObserved shortfall", res.Violations) - } + require.NoError(t, err, "a skipped rule is a known condition, not an inability") + require.Equal(t, OutcomeSkipped, res.Verdicts[0].Outcome) + _, ok := res.Coverage[checkUID] + require.False(t, ok, "a skipped rule has no coverage to prove") + require.Len(t, res.Violations, 1, "the MinObserved shortfall") res, err = run(true) - if err != nil || len(res.Violations) != 0 { - t.Errorf("with --allow-paused: err = %v, Violations = %+v, want a pass", err, res.Violations) - } + require.NoError(t, err) + require.Empty(t, res.Violations, "with --allow-paused: want a pass") } // A paused rule does not evaluate, so it can never catch up: the drain wait @@ -1076,21 +953,12 @@ func TestCheckDrainWaitConcludesAtOnceOnAPausedRule(t *testing.T) { }) res, err := check(context.Background(), cfg, src) - if err == nil { - t.Fatalf("check() = nil, want a rule that stopped evaluating to fail closed") - } - if got := src.callCount(checkTitle); got != 1 { - t.Errorf("polled %d times, want exactly 1: a paused rule can never catch up", got) - } - if got := res.Coverage[checkUID].Reason; got != ReasonDrainTimeout { - t.Errorf("Reason = %q, want %q — the vocabulary is published, so the detail goes in the note", got, ReasonDrainTimeout) - } - if !strings.Contains(res.Verdicts[0].Note, "paused before it evaluated through") { - t.Errorf("Note = %q, want it to say the rule was paused", res.Verdicts[0].Note) - } - if waited := clock.Now().Sub(windowEnd); waited >= checkDrainLimit { - t.Errorf("spent %s in the drain wait, want it to conclude at once", waited) - } + require.Error(t, err, "a rule that stopped evaluating must fail closed") + require.Equal(t, 1, src.callCount(checkTitle), "a paused rule can never catch up") + require.Equal(t, ReasonDrainTimeout, res.Coverage[checkUID].Reason, + "the vocabulary is published, so the detail goes in the note") + require.Contains(t, res.Verdicts[0].Note, "paused before it evaluated through") + require.Less(t, clock.Now().Sub(windowEnd), checkDrainLimit) } // An absent or unparseable pidfile is never "there was nothing to stop". The @@ -1120,9 +988,8 @@ func TestCheckRefusesToReadALogItCannotStop(t *testing.T) { cfg := recorderConfig(t, newVirtualClock(testNow), logPath) _, err := check(context.Background(), cfg, newCheckSource(nil)) - if err == nil || !strings.Contains(err.Error(), "cannot stop the recorder") { - t.Fatalf("check() = %v, want a refusal to stop the recorder", err) - } + require.Error(t, err) + require.Contains(t, err.Error(), "cannot stop the recorder") }) } } @@ -1137,19 +1004,14 @@ func startLockHolder(t *testing.T, logPath string) int { cmd.Env = append(os.Environ(), lockHolderEnv+"="+logPath) cmd.Stderr = os.Stderr stdout, err := cmd.StdoutPipe() - if err != nil { - t.Fatalf("pipe: %v", err) - } - if err := cmd.Start(); err != nil { - t.Fatalf("start the lock holder: %v", err) - } + require.NoError(t, err) + require.NoError(t, cmd.Start()) t.Cleanup(func() { _ = cmd.Process.Kill() _ = cmd.Wait() }) - if _, err := bufio.NewReader(stdout).ReadString('\n'); err != nil { - t.Fatalf("the lock holder never reported holding the lock: %v", err) - } + _, err = bufio.NewReader(stdout).ReadString('\n') + require.NoError(t, err, "the lock holder never reported holding the lock") return cmd.Process.Pid } @@ -1165,9 +1027,8 @@ func TestCheckFailsWhenTheRecorderWillNotExit(t *testing.T) { cfg := recorderConfig(t, newVirtualClock(testNow), logPath) _, err := check(context.Background(), cfg, newCheckSource(nil)) - if err == nil || !strings.Contains(err.Error(), "still holds") { - t.Fatalf("check() = %v, want the stop wait to time out on the lock", err) - } + require.Error(t, err) + require.Contains(t, err.Error(), "still holds") } // The regression pin for a stray SIGTERM. Nothing removes the pidfile when a @@ -1185,9 +1046,7 @@ func TestCheckDoesNotSignalABystanderHoldingAReusedPid(t *testing.T) { // left behind. It does not hold the log's lock, because it is not a // recorder. bystander := exec.Command("sleep", "30") - if err := bystander.Start(); err != nil { - t.Fatalf("start the bystander: %v", err) - } + require.NoError(t, bystander.Start()) t.Cleanup(func() { _ = bystander.Process.Kill() _ = bystander.Wait() @@ -1195,12 +1054,10 @@ func TestCheckDoesNotSignalABystanderHoldingAReusedPid(t *testing.T) { writePid(t, logPath+".pid", fmt.Sprintf("%d\n", bystander.Process.Pid)) cfg := recorderConfig(t, newVirtualClock(testNow), logPath) - if _, err := check(context.Background(), cfg, newCheckSource(nil)); err != nil { - t.Fatalf("check() = %v, want nil\nnotes:\n%s", err, notesOf(cfg)) - } - if err := syscall.Kill(bystander.Process.Pid, 0); err != nil { - t.Fatalf("the bystander is gone (%v): check signalled a process that was not the recorder", err) - } + _, err := check(context.Background(), cfg, newCheckSource(nil)) + require.NoError(t, err) + require.NoError(t, syscall.Kill(bystander.Process.Pid, 0), + "check signalled a process that was not the recorder") } // A dead pidfile (the recorder process has already exited, holding no flock) @@ -1212,40 +1069,29 @@ func TestCheckDeadPidWithNoSentinelIsUnobservable(t *testing.T) { windowEnd := testNow.Add(5*time.Minute + checkGrace) logPath := filepath.Join(dir, "log.jsonl") w, err := NewWriter(logPath, newFakeClock(testNow)) - if err != nil { - t.Fatalf("NewWriter: %v", err) - } - if err := w.WriteHeader(Header{ + require.NoError(t, err) + require.NoError(t, w.WriteHeader(Header{ URL: "https://grafana.example.com", GrafanaVersion: "13.1.0", StartedAt: testNow.Add(-time.Minute), Rules: []LoggedRule{{ UID: checkUID, Title: checkTitle, Folder: "F", Group: "G", IntervalSeconds: 60, NoDataState: "OK", ExecErrState: "OK", PollEverySeconds: checkPollEvery.Seconds(), }}, - }); err != nil { - t.Fatalf("WriteHeader: %v", err) - } + })) // Healthy heartbeats all the way past windowEnd — evaluatedThrough is // satisfied, so the drain wait needs no live re-poll — but no sentinel is // ever written: the recorder died before it could call Stop. for at := testNow.Add(-time.Minute); !at.After(windowEnd.Add(30 * time.Second)); at = at.Add(checkPollEvery) { - if err := w.WritePoll(Poll{RuleUID: checkUID, GrafanaNow: at, Found: true, State: "inactive", Health: "ok", LastEvaluation: at}); err != nil { - t.Fatalf("WritePoll: %v", err) - } - } - if err := w.Close(); err != nil { // no sentinel — a clean exit would call Stop - t.Fatalf("Close: %v", err) + require.NoError(t, w.WritePoll(Poll{RuleUID: checkUID, GrafanaNow: at, Found: true, State: "inactive", Health: "ok", LastEvaluation: at})) } + require.NoError(t, w.Close()) // no sentinel — a clean exit would call Stop writePid(t, logPath+".pid", fmt.Sprintf("%d\n", deadPid(t))) cfg := recorderConfig(t, newVirtualClock(testNow), logPath) res, err := check(context.Background(), cfg, newCheckSource(nil)) - if err == nil { - t.Fatalf("check() = nil, want an error: no sentinel means the recorder never proved it ran to the end") - } - if len(res.Verdicts) != 1 || res.Verdicts[0].Outcome != OutcomeUnobservable { - t.Fatalf("Verdicts = %+v, want one unobservable verdict", res.Verdicts) - } + require.Error(t, err, "no sentinel means the recorder never proved it ran to the end") + require.Len(t, res.Verdicts, 1) + require.Equal(t, OutcomeUnobservable, res.Verdicts[0].Outcome) } // An incomplete last line gives exit 2. log_test.go's TestReadLogRejectsBadLogs @@ -1263,25 +1109,19 @@ func TestCheckRecorderModeTruncatedLogFailsClosed(t *testing.T) { Rules: []LoggedRule{{UID: checkUID, Title: checkTitle, IntervalSeconds: 60, PollEverySeconds: checkPollEvery.Seconds()}}, } hb, err := json.Marshal(headerRecord{Type: RecordHeader, Header: h}) - if err != nil { - t.Fatalf("marshal header: %v", err) - } + require.NoError(t, err) pb, err := json.Marshal(pollRecord{Type: RecordPoll, Poll: Poll{ RuleUID: checkUID, GrafanaNow: testNow, Found: true, State: "inactive", Health: "ok", LastEvaluation: testNow, }}) - if err != nil { - t.Fatalf("marshal poll: %v", err) - } + require.NoError(t, err) content := string(hb) + "\n" + string(pb) + "\n" + `{"type":"poll","rule_ui` // torn mid-write - if err := os.WriteFile(path, []byte(content), 0o644); err != nil { - t.Fatalf("write log: %v", err) - } + require.NoError(t, os.WriteFile(path, []byte(content), 0o644)) writePid(t, path+".pid", fmt.Sprintf("%d\n", deadPid(t))) cfg := recorderConfig(t, newVirtualClock(testNow), path) - if _, err := check(context.Background(), cfg, newCheckSource(nil)); err == nil || !strings.Contains(err.Error(), "unparseable") { - t.Fatalf("check() = %v, want a refusal naming the unparseable tail", err) - } + _, err = check(context.Background(), cfg, newCheckSource(nil)) + require.Error(t, err) + require.Contains(t, err.Error(), "unparseable") } // One authority for the cadence, from check's side: maxGap comes from the @@ -1293,15 +1133,11 @@ func TestCheckDerivesMaxGapFromTheRecordedCadence(t *testing.T) { windowEnd := testNow.Add(5*time.Minute + checkGrace) path := filepath.Join(dir, "log.jsonl") w, err := NewWriter(path, newFakeClock(windowEnd.Add(30*time.Second))) - if err != nil { - t.Fatalf("NewWriter: %v", err) - } - if err := w.WriteHeader(Header{ + require.NoError(t, err) + require.NoError(t, w.WriteHeader(Header{ URL: "https://grafana.example.com", GrafanaVersion: "13.1.0", StartedAt: testNow.Add(-time.Minute), Rules: []LoggedRule{{UID: checkUID, Title: checkTitle, IntervalSeconds: 60, PollEverySeconds: 5}}, - }); err != nil { - t.Fatalf("WriteHeader: %v", err) - } + })) for at := testNow.Add(-time.Minute); !at.After(windowEnd.Add(30 * time.Second)); at = at.Add(5 * time.Second) { // A 20s hole: under the recorded 5s cadence maxGap is 10s and this // fails; under a cadence re-derived from intervalSeconds it would be @@ -1309,25 +1145,17 @@ func TestCheckDerivesMaxGapFromTheRecordedCadence(t *testing.T) { if at.After(testNow.Add(time.Minute)) && at.Before(testNow.Add(80*time.Second)) { continue } - if err := w.WritePoll(Poll{ + require.NoError(t, w.WritePoll(Poll{ RuleUID: checkUID, GrafanaNow: at, Found: true, State: "inactive", Health: "ok", LastEvaluation: at, - }); err != nil { - t.Fatalf("WritePoll: %v", err) - } - } - if err := w.Stop(); err != nil { - t.Fatalf("Stop: %v", err) + })) } + require.NoError(t, w.Stop()) writePid(t, path+".pid", fmt.Sprintf("%d\n", deadPid(t))) cfg := recorderConfig(t, newVirtualClock(testNow), path) res, err := check(context.Background(), cfg, newCheckSource(nil)) - if err == nil { - t.Fatalf("check() = nil; a 20s hole exceeds the 10s maxGap the recorded 5s cadence implies") - } - if got := res.Coverage[checkUID].Reason; got != ReasonHeartbeatGap { - t.Errorf("Reason = %q, want %q", got, ReasonHeartbeatGap) - } + require.Error(t, err, "a 20s hole exceeds the 10s maxGap the recorded 5s cadence implies") + require.Equal(t, ReasonHeartbeatGap, res.Coverage[checkUID].Reason) } // --------------------------------------------------------------------------- @@ -1367,9 +1195,7 @@ func TestEvaluatedThroughSpendsItsUncertaintyFailingClosed(t *testing.T) { for _, tc := range tests { t.Run(tc.name, func(t *testing.T) { - if got := evaluatedThrough(tc.lastEval, tc.skew, tc.bound, end); got != tc.wantSatisfied { - t.Errorf("evaluatedThrough() = %v, want %v", got, tc.wantSatisfied) - } + require.Equal(t, tc.wantSatisfied, evaluatedThrough(tc.lastEval, tc.skew, tc.bound, end)) }) } } @@ -1392,33 +1218,17 @@ func TestMergeDrainTimeoutsNamesEveryUnobservableRule(t *testing.T) { "a": {reason: ReasonDrainTimeout, note: "rule \"A\": did not evaluate through the end within the drain limit"}, "b": {reason: ReasonDrainTimeout, note: "rule \"B\": did not evaluate through the end within the drain limit"}, }) - if err == nil { - t.Fatal("mergeDrainTimeouts() = nil, want an error naming the newly unobservable rule") - } - // Its own shape: joined with decide's, two counts under one identical - // phrase would read as a contradiction rather than as two findings. - if !strings.Contains(err.Error(), "unobservable at the drain wait") { - t.Errorf("err = %q, want the drain wait's own error shape", err) - } + require.Error(t, err, "naming the newly unobservable rule") + require.Contains(t, err.Error(), "unobservable at the drain wait") // Only A is newly unobservable; B was already, so naming it twice would // only lengthen the message. - if !strings.Contains(err.Error(), "A ("+string(ReasonDrainTimeout)+")") { - t.Errorf("err = %q, want it to name A's drain timeout", err) - } - if strings.Contains(err.Error(), "B (") { - t.Errorf("err = %q, want it not to re-report B, which decide already reported", err) - } - if got := merged.Coverage["a"].Reason; got != ReasonDrainTimeout { - t.Errorf("Coverage[a].Reason = %q, want %q", got, ReasonDrainTimeout) - } + require.Contains(t, err.Error(), "A ("+string(ReasonDrainTimeout)+")") + require.NotContains(t, err.Error(), "B (") + require.Equal(t, ReasonDrainTimeout, merged.Coverage["a"].Reason) // B keeps the reason the coverage proof gave it — the FIRST reason wins, // as it does inside proveCoverage. - if got := merged.Coverage["b"].Reason; got != ReasonHeartbeatGap { - t.Errorf("Coverage[b].Reason = %q, want the earlier %q", got, ReasonHeartbeatGap) - } - if merged.Verdicts[0].Outcome != OutcomeUnobservable { - t.Errorf("Verdicts[0].Outcome = %q, want %q", merged.Verdicts[0].Outcome, OutcomeUnobservable) - } + require.Equal(t, ReasonHeartbeatGap, merged.Coverage["b"].Reason) + require.Equal(t, OutcomeUnobservable, merged.Verdicts[0].Outcome) } // ReadLogHeader is the one read of a log a writer may still hold, so its @@ -1429,45 +1239,30 @@ func TestReadLogHeader(t *testing.T) { t.Run("reads line 1 while the log keeps growing", func(t *testing.T) { path := filepath.Join(dir, "growing.jsonl") w, err := NewWriter(path, newFakeClock(testNow)) - if err != nil { - t.Fatalf("NewWriter: %v", err) - } + require.NoError(t, err) defer w.Close() - if err := w.WriteHeader(testHeader()); err != nil { - t.Fatalf("WriteHeader: %v", err) - } - if err := w.WritePoll(Poll{RuleUID: "rule1", GrafanaNow: testNow, Found: true}); err != nil { - t.Fatalf("WritePoll: %v", err) - } + require.NoError(t, w.WriteHeader(testHeader())) + require.NoError(t, w.WritePoll(Poll{RuleUID: "rule1", GrafanaNow: testNow, Found: true})) h, err := ReadLogHeader(path) - if err != nil { - t.Fatalf("ReadLogHeader: %v", err) - } - if h.URL != testHeader().URL || len(h.Rules) != 1 { - t.Errorf("header = %+v, want the written one", h) - } + require.NoError(t, err) + require.Equal(t, testHeader().URL, h.URL) + require.Len(t, h.Rules, 1) }) t.Run("a half-written header is not a header", func(t *testing.T) { path := filepath.Join(dir, "torn.jsonl") - if err := os.WriteFile(path, []byte(`{"type":"header","url":"htt`), 0o644); err != nil { - t.Fatalf("write: %v", err) - } - if _, err := ReadLogHeader(path); err == nil || - !strings.Contains(err.Error(), "no complete header") { - t.Fatalf("ReadLogHeader() = %v, want a refusal", err) - } + require.NoError(t, os.WriteFile(path, []byte(`{"type":"header","url":"htt`), 0o644)) + _, err := ReadLogHeader(path) + require.Error(t, err) + require.Contains(t, err.Error(), "no complete header") }) t.Run("a wrong schema version is refused", func(t *testing.T) { path := filepath.Join(dir, "old.jsonl") - if err := os.WriteFile(path, []byte(`{"type":"header","schema_version":99,"url":"u"}`+"\n"), 0o644); err != nil { - t.Fatalf("write: %v", err) - } - if _, err := ReadLogHeader(path); err == nil || - !strings.Contains(err.Error(), "schema version 99") { - t.Fatalf("ReadLogHeader() = %v, want a schema refusal", err) - } + require.NoError(t, os.WriteFile(path, []byte(`{"type":"header","schema_version":99,"url":"u"}`+"\n"), 0o644)) + _, err := ReadLogHeader(path) + require.Error(t, err) + require.Contains(t, err.Error(), "schema version 99") }) } diff --git a/grafana-alertcheck/internal/gate/classify.go b/grafana-alertcheck/internal/gate/classify.go index d5d9c8e12..d8d058a34 100644 --- a/grafana-alertcheck/internal/gate/classify.go +++ b/grafana-alertcheck/internal/gate/classify.go @@ -476,7 +476,7 @@ func decide(h Header, polls []Poll, sentinel *time.Time, defs []Definition, Thresholds: make(map[string]RuleThresholds), Global: GlobalThresholds{ TransitionGrace: gt.transitionGrace, - GraceSource: gt.graceSource, + GraceSource: graceSourceOrNone(gt.graceSource), DrainTimeout: gt.drainTimeout, }, } diff --git a/grafana-alertcheck/internal/gate/classify_test.go b/grafana-alertcheck/internal/gate/classify_test.go index 017d7ae6b..ad25a6ea0 100644 --- a/grafana-alertcheck/internal/gate/classify_test.go +++ b/grafana-alertcheck/internal/gate/classify_test.go @@ -3,6 +3,8 @@ package gate import ( "testing" "time" + + "github.com/stretchr/testify/require" ) func lbl(name string) map[string]string { return map[string]string{"instance": name} } @@ -55,9 +57,9 @@ func TestClassifyRule_NoEvidenceIsClean(t *testing.T) { polls := []Poll{quietPoll("r1", from), quietPoll("r1", to)} outcome, badFor, viols := classifyRule(def, polls, from, to, defaultBad, PreexistingFailUnlessRecovered) - if outcome != OutcomeClean || badFor != 0 || len(viols) != 0 { - t.Fatalf("outcome=%v badFor=%v viols=%v, want clean/0/none", outcome, badFor, viols) - } + require.Equal(t, OutcomeClean, outcome) + require.Zero(t, badFor) + require.Empty(t, viols) } func TestClassifyRule_NewOnsetInsideWindowIsNewlyBad(t *testing.T) { @@ -72,15 +74,10 @@ func TestClassifyRule_NewOnsetInsideWindowIsNewlyBad(t *testing.T) { abnormalPoll("r1", to, StateFiring, lbl("a"), onset), } outcome, badFor, viols := classifyRule(def, polls, from, to, defaultBad, PreexistingFailUnlessRecovered) - if outcome != OutcomeNewlyBad { - t.Fatalf("outcome = %v, want newly_bad", outcome) - } - if want := to.Sub(onset); badFor != want { - t.Fatalf("badFor = %v, want %v", badFor, want) - } - if len(viols) != 1 || viols[0].Outcome != OutcomeNewlyBad { - t.Fatalf("viols = %+v, want exactly one newly_bad violation", viols) - } + require.Equal(t, OutcomeNewlyBad, outcome) + require.Equal(t, to.Sub(onset), badFor) + require.Len(t, viols, 1) + require.Equal(t, OutcomeNewlyBad, viols[0].Outcome) } // A genuinely new bad episode fails even if it clears again before the window @@ -99,12 +96,8 @@ func TestClassifyRule_NewOnsetThatClearsStillFails(t *testing.T) { quietPoll("r1", to), } outcome, _, viols := classifyRule(def, polls, from, to, defaultBad, PreexistingFailUnlessRecovered) - if outcome != OutcomeNewlyBad { - t.Fatalf("outcome = %v, want newly_bad even though it cleared", outcome) - } - if len(viols) != 1 { - t.Fatalf("viols = %+v, want one violation", viols) - } + require.Equal(t, OutcomeNewlyBad, outcome, "even though it cleared") + require.Len(t, viols, 1) } // --- recovered / persistently_bad (preexisting) --- @@ -122,15 +115,9 @@ func TestClassifyRule_PreexistingThatRecoversIsRecoveredAndNotAViolation(t *test quietPoll("r1", to), } outcome, badFor, viols := classifyRule(def, polls, from, to, defaultBad, PreexistingFailUnlessRecovered) - if outcome != OutcomeRecovered { - t.Fatalf("outcome = %v, want recovered", outcome) - } - if want := clearAt.Sub(from); badFor != want { - t.Fatalf("badFor = %v, want %v", badFor, want) - } - if len(viols) != 0 { - t.Fatalf("viols = %+v, want none: default policy passes a recovered preexisting instance", viols) - } + require.Equal(t, OutcomeRecovered, outcome) + require.Equal(t, clearAt.Sub(from), badFor) + require.Empty(t, viols, "default policy passes a recovered preexisting instance") } // The late condition: bad for 58 of a 60-minute window, clear at minute 58, @@ -149,15 +136,9 @@ func TestClassifyRule_LateRecoveryPassesRegardlessOfHowLateItIs(t *testing.T) { quietPoll("r1", to), } outcome, badFor, viols := classifyRule(def, polls, from, to, defaultBad, PreexistingFailUnlessRecovered) - if outcome != OutcomeRecovered { - t.Fatalf("outcome = %v, want recovered even 58 minutes into a 60-minute window", outcome) - } - if want := clearAt.Sub(from); badFor != want { - t.Fatalf("badFor = %v, want the full %v bad duration, not a value clamped against a deadline", badFor, want) - } - if len(viols) != 0 { - t.Fatalf("viols = %+v, want none: there is no deadline a preexisting recovery must beat", viols) - } + require.Equal(t, OutcomeRecovered, outcome, "even 58 minutes into a 60-minute window") + require.Equal(t, clearAt.Sub(from), badFor, "not a value clamped against a deadline") + require.Empty(t, viols, "there is no deadline a preexisting recovery must beat") } func TestClassifyRule_PreexistingStillBadAtWindowEndIsPersistentlyBad(t *testing.T) { @@ -170,15 +151,10 @@ func TestClassifyRule_PreexistingStillBadAtWindowEndIsPersistentlyBad(t *testing abnormalPoll("r1", to, StateFiring, lbl("a"), from.Add(-time.Hour)), } outcome, badFor, viols := classifyRule(def, polls, from, to, defaultBad, PreexistingFailUnlessRecovered) - if outcome != OutcomePersistentlyBad { - t.Fatalf("outcome = %v, want persistently_bad", outcome) - } - if badFor != to.Sub(from) { - t.Fatalf("badFor = %v, want the full window %v", badFor, to.Sub(from)) - } - if len(viols) != 1 || viols[0].Outcome != OutcomePersistentlyBad { - t.Fatalf("viols = %+v, want one persistently_bad violation", viols) - } + require.Equal(t, OutcomePersistentlyBad, outcome) + require.Equal(t, to.Sub(from), badFor) + require.Len(t, viols, 1) + require.Equal(t, OutcomePersistentlyBad, viols[0].Outcome) } // --- flapping --- @@ -196,12 +172,9 @@ func TestClassifyRule_ClearThenBadAgainIsFlapping(t *testing.T) { quietPoll("r1", to), } outcome, _, viols := classifyRule(def, polls, from, to, defaultBad, PreexistingFailUnlessRecovered) - if outcome != OutcomeFlapping { - t.Fatalf("outcome = %v, want flapping", outcome) - } - if len(viols) != 1 || viols[0].Outcome != OutcomeFlapping { - t.Fatalf("viols = %+v, want one flapping violation, always a fail regardless of policy", viols) - } + require.Equal(t, OutcomeFlapping, outcome) + require.Len(t, viols, 1) + require.Equal(t, OutcomeFlapping, viols[0].Outcome, "always a fail regardless of policy") } // A clear and then a second bad state gives flapping, wherever the second bad @@ -233,12 +206,9 @@ func TestClassifyRule_FlappingAtEveryTimingOfTheSecondOnset(t *testing.T) { quietPoll("r1", to), } outcome, _, viols := classifyRule(def, polls, from, to, defaultBad, PreexistingFailUnlessRecovered) - if outcome != OutcomeFlapping { - t.Fatalf("outcome = %v, want flapping for a second onset at %s", outcome, tc.secondOnset) - } - if len(viols) != 1 || viols[0].Outcome != OutcomeFlapping { - t.Fatalf("viols = %+v, want one flapping violation", viols) - } + require.Equalf(t, OutcomeFlapping, outcome, "second onset at %s", tc.secondOnset) + require.Len(t, viols, 1) + require.Equal(t, OutcomeFlapping, viols[0].Outcome) }) } } @@ -257,15 +227,9 @@ func TestClassifyRule_VanishedWhileBadStaysPersistentlyBad(t *testing.T) { quietPoll("r1", to), } outcome, badFor, viols := classifyRule(def, polls, from, to, defaultBad, PreexistingFailUnlessRecovered) - if outcome != OutcomePersistentlyBad { - t.Fatalf("outcome = %v, want persistently_bad: a vanish must never read as a recovery", outcome) - } - if badFor != to.Sub(from) { - t.Fatalf("badFor = %v, want the full window %v: the freeze must hold the episode open to windowEnd", badFor, to.Sub(from)) - } - if len(viols) != 1 { - t.Fatalf("viols = %+v, want one violation", viols) - } + require.Equal(t, OutcomePersistentlyBad, outcome, "a vanish must never read as a recovery") + require.Equal(t, to.Sub(from), badFor, "the freeze must hold the episode open to windowEnd") + require.Len(t, viols, 1) } func TestClassifyRule_VanishedWhileNeverBadIsUninteresting(t *testing.T) { @@ -282,9 +246,9 @@ func TestClassifyRule_VanishedWhileNeverBadIsUninteresting(t *testing.T) { quietPoll("r1", to), } outcome, badFor, viols := classifyRule(def, polls, from, to, defaultBad, PreexistingFailUnlessRecovered) - if outcome != OutcomeClean || badFor != 0 || len(viols) != 0 { - t.Fatalf("outcome=%v badFor=%v viols=%v, want clean/0/none", outcome, badFor, viols) - } + require.Equal(t, OutcomeClean, outcome) + require.Zero(t, badFor) + require.Empty(t, viols) } // --- preexisting policy --- @@ -301,12 +265,10 @@ func TestClassifyRule_PreexistingPolicyFailFailsARecoveredInstance(t *testing.T) quietPoll("r1", to), } outcome, _, viols := classifyRule(def, polls, from, to, defaultBad, PreexistingFail) - if outcome != OutcomeRecovered { - t.Fatalf("outcome = %v, want recovered — the descriptive outcome does not change under policy=fail", outcome) - } - if len(viols) != 1 || viols[0].Outcome != OutcomeRecovered { - t.Fatalf("viols = %+v, want one violation: policy=fail gives no benefit of the doubt to a preexisting instance", viols) - } + require.Equal(t, OutcomeRecovered, outcome, "the descriptive outcome does not change under policy=fail") + require.Len(t, viols, 1) + require.Equal(t, OutcomeRecovered, viols[0].Outcome, + "policy=fail gives no benefit of the doubt to a preexisting instance") } func TestClassifyRule_PreexistingPolicyIgnoreForgivesPersistentlyBad(t *testing.T) { @@ -319,12 +281,8 @@ func TestClassifyRule_PreexistingPolicyIgnoreForgivesPersistentlyBad(t *testing. abnormalPoll("r1", to, StateFiring, lbl("a"), from.Add(-time.Hour)), } outcome, _, viols := classifyRule(def, polls, from, to, defaultBad, PreexistingIgnore) - if outcome != OutcomePersistentlyBad { - t.Fatalf("outcome = %v, want persistently_bad — the descriptive outcome does not change under policy=ignore", outcome) - } - if len(viols) != 0 { - t.Fatalf("viols = %+v, want none: policy=ignore disregards a preexisting instance even if it never recovers", viols) - } + require.Equal(t, OutcomePersistentlyBad, outcome, "the descriptive outcome does not change under policy=ignore") + require.Empty(t, viols, "policy=ignore disregards a preexisting instance even if it never recovers") } func TestClassifyRule_PreexistingPolicyIgnoreStillFailsANewOnset(t *testing.T) { @@ -339,9 +297,8 @@ func TestClassifyRule_PreexistingPolicyIgnoreStillFailsANewOnset(t *testing.T) { abnormalPoll("r1", to, StateFiring, lbl("a"), onset), } outcome, _, viols := classifyRule(def, polls, from, to, defaultBad, PreexistingIgnore) - if outcome != OutcomeNewlyBad || len(viols) != 1 { - t.Fatalf("outcome=%v viols=%v, want newly_bad/1: ignore only forgives PREEXISTING badness", outcome, viols) - } + require.Equal(t, OutcomeNewlyBad, outcome) + require.Len(t, viols, 1, "ignore only forgives PREEXISTING badness") } // --- worst-of across instances --- @@ -366,12 +323,9 @@ func TestClassifyRule_WorstOfMultipleInstancesWins(t *testing.T) { }, } outcome, _, viols := classifyRule(def, polls, from, to, defaultBad, PreexistingFailUnlessRecovered) - if outcome != OutcomePersistentlyBad { - t.Fatalf("outcome = %v, want persistently_bad: the worse of {recovered, persistently_bad}", outcome) - } - if len(viols) != 1 || viols[0].Outcome != OutcomePersistentlyBad { - t.Fatalf("viols = %+v, want exactly the persistently_bad instance's violation", viols) - } + require.Equal(t, OutcomePersistentlyBad, outcome, "the worse of {recovered, persistently_bad}") + require.Len(t, viols, 1) + require.Equal(t, OutcomePersistentlyBad, viols[0].Outcome) } // --- decide(): skipped rules, unobservable, MinObserved, exit mapping --- @@ -390,15 +344,11 @@ func TestDecide_SkippedRuleNeverReachesProveCoverage(t *testing.T) { // The HEADER is what says paused — decide reads skipped from there, not // from def.IsPaused, which is a post-window reading (Header.pausedAtStart). res, err := decide(pausedHeader(from.Add(-time.Hour), "r1"), nil, nil, defs, rt, gt, pol) - if err != nil { - t.Fatalf("err = %v, want nil: a rule paused before the window is skipped, not unobservable", err) - } - if len(res.Verdicts) != 1 || res.Verdicts[0].Outcome != OutcomeSkipped { - t.Fatalf("Verdicts = %+v, want exactly one skipped verdict", res.Verdicts) - } - if _, ok := res.Coverage["r1"]; ok { - t.Fatalf("Coverage[r1] present, want absent: a skipped rule has no coverage to prove") - } + require.NoError(t, err, "a rule paused before the window is skipped, not unobservable") + require.Len(t, res.Verdicts, 1) + require.Equal(t, OutcomeSkipped, res.Verdicts[0].Outcome) + _, ok := res.Coverage["r1"] + require.False(t, ok, "a skipped rule has no coverage to prove") } func TestDecide_UnobservableRuleAlwaysReturnsAnError(t *testing.T) { @@ -413,12 +363,9 @@ func TestDecide_UnobservableRuleAlwaysReturnsAnError(t *testing.T) { // No sentinel at all: check 1 fails, so the rule is unobservable // regardless of anything else. res, err := decide(Header{StartedAt: from.Add(-time.Hour)}, nil, nil, defs, rt, gt, pol) - if err == nil { - t.Fatalf("err = nil, want non-nil: an unobservable rule must always fail the run") - } - if len(res.Verdicts) != 1 || res.Verdicts[0].Outcome != OutcomeUnobservable { - t.Fatalf("Verdicts = %+v, want exactly one unobservable verdict", res.Verdicts) - } + require.Error(t, err, "an unobservable rule must always fail the run") + require.Len(t, res.Verdicts, 1) + require.Equal(t, OutcomeUnobservable, res.Verdicts[0].Outcome) } // Any unobservable rule means exit 2, with no exception — even alongside a @@ -450,9 +397,7 @@ func TestDecide_UnobservableWinsEvenAlongsideARealViolation(t *testing.T) { sentinel := to res, err := decide(Header{StartedAt: from.Add(-time.Hour)}, polls, &sentinel, defs, rt, gt, pol) - if err == nil { - t.Fatalf("err = nil, want non-nil: one rule is unobservable") - } + require.Error(t, err, "one rule is unobservable") var gotBroken, gotBad Outcome for _, v := range res.Verdicts { switch v.RuleUID { @@ -462,15 +407,11 @@ func TestDecide_UnobservableWinsEvenAlongsideARealViolation(t *testing.T) { gotBad = v.Outcome } } - if gotBroken != OutcomeUnobservable { - t.Fatalf("broken.Outcome = %v, want unobservable", gotBroken) - } - if gotBad != OutcomeNewlyBad { - t.Fatalf("bad.Outcome = %v, want newly_bad: classification still runs and is still visible in Verdicts", gotBad) - } - if len(res.Violations) == 0 { - t.Fatalf("Violations empty, want the newly_bad instance still reported even though the run fails on the unobservable rule") - } + require.Equal(t, OutcomeUnobservable, gotBroken) + require.Equal(t, OutcomeNewlyBad, gotBad, + "classification still runs and is still visible in Verdicts") + require.NotEmpty(t, res.Violations, + "the newly_bad instance still reported even though the run fails on the unobservable rule") } // A clean verdict with a coverage gap must never give exit 0, and recovered @@ -550,9 +491,7 @@ func TestDecide_UnobservableRuleWinsOverEveryFavorableOutcome(t *testing.T) { // so it is unobservable regardless of "good". sentinel := to res, err := decide(h, tc.goodPolls, &sentinel, defs, rt, gt, pol) - if err == nil { - t.Fatalf("err = nil, want non-nil: 'broken' is unobservable regardless of 'good' being %s", tc.name) - } + require.Errorf(t, err, "'broken' is unobservable regardless of 'good' being %s", tc.name) var gotGood, gotBroken Outcome for _, v := range res.Verdicts { switch v.RuleUID { @@ -562,12 +501,8 @@ func TestDecide_UnobservableRuleWinsOverEveryFavorableOutcome(t *testing.T) { gotBroken = v.Outcome } } - if gotGood != tc.wantOutcome { - t.Errorf("good.Outcome = %v, want %v", gotGood, tc.wantOutcome) - } - if gotBroken != OutcomeUnobservable { - t.Errorf("broken.Outcome = %v, want unobservable", gotBroken) - } + require.Equal(t, tc.wantOutcome, gotGood) + require.Equal(t, OutcomeUnobservable, gotBroken) }) } } @@ -603,15 +538,10 @@ func TestDecide_RecoveredOutcomeOverriddenByItsOwnCoverageGap(t *testing.T) { sentinel := to res, err := decide(Header{StartedAt: from.Add(-time.Hour)}, polls, &sentinel, defs, rt, gt, pol) - if err == nil { - t.Fatalf("err = nil, want non-nil: r1's own coverage gap must fail the run even though it recovered") - } - if len(res.Verdicts) != 1 || res.Verdicts[0].Outcome != OutcomeUnobservable { - t.Fatalf("Verdicts = %+v, want unobservable, never recovered", res.Verdicts) - } - if cov := res.Coverage["r1"]; cov.Proved { - t.Fatalf("Coverage = %+v, want not proved", cov) - } + require.Error(t, err, "r1's own coverage gap must fail the run even though it recovered") + require.Len(t, res.Verdicts, 1) + require.Equal(t, OutcomeUnobservable, res.Verdicts[0].Outcome, "never recovered") + require.False(t, res.Coverage["r1"].Proved) } func TestDecide_CleanWindowIsAPass(t *testing.T) { @@ -630,15 +560,9 @@ func TestDecide_CleanWindowIsAPass(t *testing.T) { sentinel := to res, err := decide(Header{StartedAt: from.Add(-time.Hour)}, polls, &sentinel, defs, rt, gt, pol) - if err != nil { - t.Fatalf("err = %v, want nil", err) - } - if len(res.Violations) != 0 { - t.Fatalf("Violations = %+v, want none: a pass is exactly len(Violations)==0 && err==nil", res.Violations) - } - if res.Verdicts[0].Outcome != OutcomeClean { - t.Fatalf("Outcome = %v, want clean", res.Verdicts[0].Outcome) - } + require.NoError(t, err) + require.Empty(t, res.Violations, "a pass is exactly len(Violations)==0 && err==nil") + require.Equal(t, OutcomeClean, res.Verdicts[0].Outcome) } // A pause and then an unpause inside the window, with an episode that would @@ -678,15 +602,10 @@ func TestDecide_PauseThenUnpauseWithHiddenEpisodeGivesUnobservableNotClean(t *te sentinel := to res, err := decide(Header{StartedAt: from.Add(-time.Hour)}, polls, &sentinel, defs, rt, gt, pol) - if err == nil { - t.Fatalf("err = nil, want the pause-then-unpause blind interval to fail closed") - } - if len(res.Verdicts) != 1 || res.Verdicts[0].Outcome != OutcomeUnobservable { - t.Fatalf("Verdicts = %+v, want unobservable, never clean", res.Verdicts) - } - if cov := res.Coverage["r1"]; cov.Proved { - t.Fatalf("Coverage = %+v, want not proved", cov) - } + require.Error(t, err, "the pause-then-unpause blind interval must fail closed") + require.Len(t, res.Verdicts, 1) + require.Equal(t, OutcomeUnobservable, res.Verdicts[0].Outcome, "never clean") + require.False(t, res.Coverage["r1"].Proved) } // --- MinObserved shortfall --- @@ -713,19 +632,14 @@ func TestDecide_SkippedOnlyShortfallProducesAViolationWithoutAnError(t *testing. sentinel := to res, err := decide(pausedHeader(from.Add(-time.Hour), "paused"), polls, &sentinel, defs, rt, gt, pol) - if err != nil { - t.Fatalf("err = %v, want nil: a shortfall caused only by a skipped rule is exit 1, not exit 2", err) - } - if len(res.Violations) != 1 { - t.Fatalf("Violations = %+v, want exactly one: a shortfall must be visible through Violations like any other fail reason", res.Violations) - } - if v := res.Violations[0]; v.Outcome != OutcomeSkipped || v.RuleUID != "paused" || v.Alert != "Paused" { - t.Fatalf("Violations[0] = %+v, want Outcome=skipped naming the paused rule", v) - } - if res.Violations[0].Note == "" { - t.Fatalf("Violations[0].Note is empty, want an explanation: the shortfall reason must not be smuggled into LastError, " + - "which is reporting-only rule state from a real poll this synthetic Violation never touched") - } + require.NoError(t, err, "a shortfall caused only by a skipped rule is exit 1, not exit 2") + require.Len(t, res.Violations, 1) + v := res.Violations[0] + require.Equal(t, OutcomeSkipped, v.Outcome) + require.Equal(t, "paused", v.RuleUID) + require.Equal(t, "Paused", v.Alert) + require.NotEmpty(t, v.Note, + "the shortfall reason must not be smuggled into LastError") } // An operator-supplied MinObserved that exceeds what could ever be resolved is @@ -748,17 +662,10 @@ func TestDecide_ExplicitMinObservedShortfallWithNoPausedRuleStillProducesAViolat sentinel := to res, err := decide(Header{StartedAt: from.Add(-time.Hour)}, polls, &sentinel, defs, rt, gt, pol) - if err != nil { - t.Fatalf("err = %v, want nil: an unmet MinObserved is exit 1, never exit 2", err) - } - if len(res.Violations) != 2 { - t.Fatalf("Violations = %+v, want two: the shortfall (3-1=2) is not explained by any paused rule, "+ - "so it must surface directly rather than pass silently", res.Violations) - } + require.NoError(t, err, "an unmet MinObserved is exit 1, never exit 2") + require.Len(t, res.Violations, 2, "the shortfall (3-1=2) must surface directly rather than pass silently") for _, v := range res.Violations { - if v.Outcome != OutcomeSkipped { - t.Fatalf("Violations = %+v, want Outcome=skipped on the synthetic shortfall entries", res.Violations) - } + require.Equal(t, OutcomeSkipped, v.Outcome) } } @@ -783,12 +690,8 @@ func TestDecide_AllowPausedSuppressesTheShortfall(t *testing.T) { sentinel := to res, err := decide(pausedHeader(from.Add(-time.Hour), "paused"), polls, &sentinel, defs, rt, gt, pol) - if err != nil { - t.Fatalf("err = %v, want nil", err) - } - if len(res.Violations) != 0 { - t.Fatalf("Violations = %+v, want none: --allow-paused must suppress the shortfall entirely", res.Violations) - } + require.NoError(t, err) + require.Empty(t, res.Violations, "--allow-paused must suppress the shortfall entirely") } // --- nodata escalation (decide's own Policy-driven check) --- @@ -809,12 +712,8 @@ func TestDecide_NodataIsUnobservableEscalatesASustainedRun(t *testing.T) { sentinel := to res, err := decide(Header{StartedAt: from.Add(-time.Hour)}, polls, &sentinel, defs, rt, gt, pol) - if err == nil { - t.Fatalf("err = nil, want non-nil: a sustained nodata run must be unobservable under --nodata-is-unobservable") - } - if res.Coverage["r1"].Reason != ReasonNodata { - t.Fatalf("Reason = %q, want %q", res.Coverage["r1"].Reason, ReasonNodata) - } + require.Error(t, err, "a sustained nodata run must be unobservable under --nodata-is-unobservable") + require.Equal(t, ReasonNodata, res.Coverage["r1"].Reason) } func TestDecide_NodataIsANoteByDefault(t *testing.T) { @@ -833,12 +732,8 @@ func TestDecide_NodataIsANoteByDefault(t *testing.T) { sentinel := to res, err := decide(Header{StartedAt: from.Add(-time.Hour)}, polls, &sentinel, defs, rt, gt, pol) - if err != nil { - t.Fatalf("err = %v, want nil: 96%% of the fleet runs no_data_state:OK and must not fail by default", err) - } - if res.Coverage["r1"].Unobservable { - t.Fatalf("Coverage[r1].Unobservable = true, want false by default") - } + require.NoError(t, err, "96%% of the fleet runs no_data_state:OK and must not fail by default") + require.False(t, res.Coverage["r1"].Unobservable) } // --- preexisting is decided by ActiveAt, not poll timing --- @@ -862,16 +757,11 @@ func TestClassifyRule_OnsetBetweenFromAndFirstPollIsNewlyBadNotRecovered(t *test quietPoll("r1", to), } outcome, badFor, viols := classifyRule(def, polls, from, to, defaultBad, PreexistingFailUnlessRecovered) - if outcome != OutcomeNewlyBad { - t.Fatalf("outcome = %v, want newly_bad: the onset is after `from`, so it is not preexisting even though "+ - "the FIRST in-window poll already observes it bad", outcome) - } - if len(viols) != 1 || viols[0].Outcome != OutcomeNewlyBad { - t.Fatalf("viols = %+v, want one newly_bad violation: a policy=fail-unless-recovered default must still fail this", viols) - } - if want := clearAt.Sub(onset); badFor != want { - t.Fatalf("badFor = %v, want %v: BadFor must count from the true onset, not from `from`", badFor, want) - } + require.Equal(t, OutcomeNewlyBad, outcome, + "the onset is after `from`, so it is not preexisting even though the FIRST in-window poll already observes it bad") + require.Len(t, viols, 1) + require.Equal(t, OutcomeNewlyBad, viols[0].Outcome) + require.Equal(t, clearAt.Sub(onset), badFor, "BadFor must count from the true onset, not from `from`") } // TestClassifyRule_OnsetJustBeforeFromIsPreexisting is the mirror check: an @@ -892,15 +782,10 @@ func TestClassifyRule_OnsetJustBeforeFromIsPreexisting(t *testing.T) { quietPoll("r1", to), } outcome, badFor, viols := classifyRule(def, polls, from, to, defaultBad, PreexistingFailUnlessRecovered) - if outcome != OutcomeRecovered { - t.Fatalf("outcome = %v, want recovered: the onset is at/before `from`, genuinely preexisting", outcome) - } - if len(viols) != 0 { - t.Fatalf("viols = %+v, want none: default policy passes a recovered preexisting instance", viols) - } - if want := clearAt.Sub(from); badFor != want { - t.Fatalf("badFor = %v, want %v: a preexisting episode's BadFor is clamped to window-open, not backdated past it", badFor, want) - } + require.Equal(t, OutcomeRecovered, outcome, "the onset is at/before `from`, genuinely preexisting") + require.Empty(t, viols, "default policy passes a recovered preexisting instance") + require.Equal(t, clearAt.Sub(from), badFor, + "a preexisting episode's BadFor is clamped to window-open, not backdated past it") } // A poll carrying a nonzero skew must have its ActiveAt (and GrafanaNow) @@ -928,12 +813,9 @@ func TestClassifyRule_SkewTranslatesActiveAtAcrossTheWindowBoundary(t *testing.T stillBad.LastEvaluation = to.Add(skew) outcome, badFor, _ := classifyRule(def, []Poll{poll, stillBad}, from, to, defaultBad, PreexistingFailUnlessRecovered) - if outcome != OutcomePersistentlyBad { - t.Fatalf("outcome = %v, want persistently_bad: a +90s skew must translate ActiveAt back to exactly `from`", outcome) - } - if badFor != to.Sub(from) { - t.Fatalf("badFor = %v, want the full window %v", badFor, to.Sub(from)) - } + require.Equal(t, OutcomePersistentlyBad, outcome, + "a +90s skew must translate ActiveAt back to exactly `from`") + require.Equal(t, to.Sub(from), badFor) } // --- InstanceLabels must survive a timeline first created by a bare marker --- @@ -954,13 +836,10 @@ func TestClassifyRule_LabelsSurviveWhenTimelineStartsFromAClearedMarker(t *testi quietPoll("r1", to), } _, _, viols := classifyRule(def, polls, from, to, defaultBad, PreexistingFailUnlessRecovered) - if len(viols) != 1 { - t.Fatalf("viols = %+v, want exactly one newly_bad violation", viols) - } - if viols[0].InstanceLabels == nil || viols[0].InstanceLabels["instance"] != "a" { - t.Fatalf("InstanceLabels = %+v, want {instance: a}: labels must backfill even though the "+ - "timeline was first created by a label-less Cleared marker", viols[0].InstanceLabels) - } + require.Len(t, viols, 1) + require.NotNil(t, viols[0].InstanceLabels) + require.Equal(t, "a", viols[0].InstanceLabels["instance"], + "labels must backfill even though the timeline was first created by a label-less Cleared marker") } // FirstSeen/ClearedAt are pinned exactly, not just that a violation exists. @@ -978,19 +857,11 @@ func TestClassifyRule_ViolationFieldsArePrecise(t *testing.T) { quietPoll("r1", to), } _, _, viols := classifyRule(def, polls, from, to, defaultBad, PreexistingFailUnlessRecovered) - if len(viols) != 1 { - t.Fatalf("viols = %+v, want exactly one violation", viols) - } + require.Len(t, viols, 1) v := viols[0] - if !v.FirstSeen.Equal(onset) { - t.Fatalf("FirstSeen = %v, want %v", v.FirstSeen, onset) - } - if !v.ClearedAt.Equal(clearAt) { - t.Fatalf("ClearedAt = %v, want %v", v.ClearedAt, clearAt) - } - if v.InstanceLabels["instance"] != "a" { - t.Fatalf("InstanceLabels = %+v, want {instance: a}", v.InstanceLabels) - } + require.True(t, v.FirstSeen.Equal(onset)) + require.True(t, v.ClearedAt.Equal(clearAt)) + require.Equal(t, "a", v.InstanceLabels["instance"]) } // The episode.end clamp: inWindowPolls admits a poll up to its own skew bound @@ -1013,13 +884,9 @@ func TestClassifyRule_ClearedEventPastWindowEndClampsToWindowEnd(t *testing.T) { }, } outcome, badFor, _ := classifyRule(def, polls, from, to, defaultBad, PreexistingFailUnlessRecovered) - if outcome != OutcomeRecovered { - t.Fatalf("outcome = %v, want recovered", outcome) - } - if badFor != to.Sub(from) { - t.Fatalf("badFor = %v, want the window %v exactly: the episode end must clamp to windowEnd, "+ - "not extend to the late Cleared event's raw time", badFor, to.Sub(from)) - } + require.Equal(t, OutcomeRecovered, outcome) + require.Equal(t, to.Sub(from), badFor, + "the episode end must clamp to windowEnd, not extend to the late Cleared event's raw time") } // TestClassifyRule_OnsetJustPastWindowEndIsNewlyBadNotClean pins the fail-closed @@ -1047,16 +914,10 @@ func TestClassifyRule_OnsetJustPastWindowEndIsNewlyBadNotClean(t *testing.T) { } outcome, badFor, viols := classifyRule(def, []Poll{poll}, from, windowEnd, defaultBad, PreexistingFailUnlessRecovered) - if outcome != OutcomeNewlyBad { - t.Fatalf("outcome = %v, want newly_bad: an onset past windowEnd seen only via the skew bound must fail closed", outcome) - } - if badFor != 0 { - t.Fatalf("badFor = %v, want 0: the zero-length episode must truncate to the window end", badFor) - } - if len(viols) != 1 { - t.Fatalf("viols = %+v, want exactly one newly_bad violation", viols) - - } + require.Equal(t, OutcomeNewlyBad, outcome, + "an onset past windowEnd seen only via the skew bound must fail closed") + require.Zero(t, badFor, "the zero-length episode must truncate to the window end") + require.Len(t, viols, 1) } // A clear after `to` gives persistently_bad. classifyRule filters @@ -1075,15 +936,10 @@ func TestClassifyRule_ClearAfterWindowEndIsPersistentlyBad(t *testing.T) { clearedPoll("r1", to.Add(time.Hour), key), // far past `to`, not a boundary case } outcome, badFor, viols := classifyRule(def, polls, from, to, defaultBad, PreexistingFailUnlessRecovered) - if outcome != OutcomePersistentlyBad { - t.Fatalf("outcome = %v, want persistently_bad: a clear outside the window must not read as a recovery", outcome) - } - if badFor != to.Sub(from) { - t.Fatalf("badFor = %v, want the full window %v", badFor, to.Sub(from)) - } - if len(viols) != 1 || viols[0].Outcome != OutcomePersistentlyBad { - t.Fatalf("viols = %+v, want one persistently_bad violation", viols) - } + require.Equal(t, OutcomePersistentlyBad, outcome, "a clear outside the window must not read as a recovery") + require.Equal(t, to.Sub(from), badFor) + require.Len(t, viols, 1) + require.Equal(t, OutcomePersistentlyBad, viols[0].Outcome) } // TestClassifyRule_CloseBeforeOpenClampsToZeroNotNegative pins the @@ -1110,19 +966,11 @@ func TestClassifyRule_CloseBeforeOpenClampsToZeroNotNegative(t *testing.T) { quietPoll("r1", to), } outcome, badFor, viols := classifyRule(def, polls, from, to, defaultBad, PreexistingFailUnlessRecovered) - if outcome != OutcomeNewlyBad { - t.Fatalf("outcome = %v, want newly_bad", outcome) - } - if badFor < 0 { - t.Fatalf("badFor = %v, want a non-negative duration even though the closing poll's translated "+ - "time landed before the opening poll's", badFor) - } - if badFor != 0 { - t.Fatalf("badFor = %v, want 0: the clamp collapses the inverted span to a zero-length episode", badFor) - } - if len(viols) != 1 { - t.Fatalf("viols = %+v, want one violation", viols) - } + require.Equal(t, OutcomeNewlyBad, outcome) + require.GreaterOrEqual(t, badFor, time.Duration(0), + "a non-negative duration even though the closing poll's translated time landed before the opening poll's") + require.Zero(t, badFor, "the clamp collapses the inverted span to a zero-length episode") + require.Len(t, viols, 1) } // --- mergeDurations --- @@ -1136,13 +984,9 @@ func TestMergeDurations_OverlappingEpisodesCountOnce(t *testing.T) { } got := mergeDurations(eps) want := 8*time.Minute + 1*time.Minute // [0,8) merged = 8m, plus the disjoint 1m - if got != want { - t.Fatalf("mergeDurations = %v, want %v: two simultaneously-bad instances must not double-count their overlap", got, want) - } + require.Equal(t, want, got, "two simultaneously-bad instances must not double-count their overlap") } func TestMergeDurations_Empty(t *testing.T) { - if got := mergeDurations(nil); got != 0 { - t.Fatalf("mergeDurations(nil) = %v, want 0", got) - } + require.Zero(t, mergeDurations(nil)) } diff --git a/grafana-alertcheck/internal/gate/coverage.go b/grafana-alertcheck/internal/gate/coverage.go index 527720bfe..603a7f7d8 100644 --- a/grafana-alertcheck/internal/gate/coverage.go +++ b/grafana-alertcheck/internal/gate/coverage.go @@ -88,12 +88,16 @@ func proveCoverage(h Header, polls []Poll, sentinel *time.Time, t ruleTimings, d // Check 2 — from bounds: from < StartedAt makes coverage unprovable, no // matter how healthy the polls that DO exist look. Both are runner-domain // clock reads (the recorder's own Clock.Now()), so no cross-domain - // translation applies here. The other half of the bound — from too far - // ahead of the runner's clock — is Check's input validation, once per run - // rather than per rule. - if from.Before(h.StartedAt) { + // translation applies here. The comparison is at whole-second granularity: + // `from` is supplied at second precision (--from RFC3339) while StartedAt + // carries the recorder's sub-second clock stamp, so an operator naming the + // exact second the recording opened must not be judged early for the + // sub-second sliver inside that same second. The other half of the bound — + // from too far ahead of the runner's clock — is Check's input validation, + // once per run rather than per rule. + if from.Truncate(time.Second).Before(h.StartedAt.Truncate(time.Second)) { fail(ReasonFromBeforeRecord, fmt.Sprintf( - "requested from %s is before recording started at %s", from.Format(time.RFC3339), h.StartedAt.Format(time.RFC3339))) + "requested from %s is before recording started at %s", from.Format(time.RFC3339Nano), h.StartedAt.Format(time.RFC3339Nano))) } // Filtered once and threaded through every remaining check. @@ -142,7 +146,7 @@ func proveCoverage(h Header, polls []Poll, sentinel *time.Time, t ruleTimings, d // corrupted or hand-edited data (ReadLog does no field validation); // GrafanaNow-LastEvaluation would go negative and silently read as // fresh — fail-open. Treat it as unobservable instead. - if p.LastEvaluation.After(p.GrafanaNow) { + if p.LastEvaluation.Truncate(time.Second).After(p.GrafanaNow) { fail(ReasonFutureEvaluation, fmt.Sprintf( "lastEvaluation %s is after grafana_now %s (corrupted poll)", p.LastEvaluation.Format(time.RFC3339), p.GrafanaNow.Format(time.RFC3339))) diff --git a/grafana-alertcheck/internal/gate/coverage_test.go b/grafana-alertcheck/internal/gate/coverage_test.go index 720386d65..3507ee4fc 100644 --- a/grafana-alertcheck/internal/gate/coverage_test.go +++ b/grafana-alertcheck/internal/gate/coverage_test.go @@ -4,6 +4,8 @@ import ( "strings" "testing" "time" + + "github.com/stretchr/testify/require" ) func TestProveCoverage_CleanWindowIsProved(t *testing.T) { @@ -19,9 +21,9 @@ func TestProveCoverage_CleanWindowIsProved(t *testing.T) { sentinel := to res := proveCoverage(Header{StartedAt: from.Add(-time.Hour)}, polls, &sentinel, rt, def, from, to, 0) - if !res.Proved || res.Unobservable || res.Reason != "" { - t.Fatalf("res = %+v, want a clean proved window", res) - } + require.True(t, res.Proved) + require.False(t, res.Unobservable) + require.Empty(t, res.Reason) } func TestProveCoverage_FiltersPollsByUID(t *testing.T) { @@ -40,9 +42,7 @@ func TestProveCoverage_FiltersPollsByUID(t *testing.T) { sentinel := to res := proveCoverage(Header{StartedAt: from.Add(-time.Hour)}, polls, &sentinel, rt, def, from, to, 0) - if !res.Proved { - t.Fatalf("res = %+v, want proved: a different rule's broken polls must not affect this rule's verdict", res) - } + require.True(t, res.Proved, "a different rule's broken polls must not affect this rule's verdict") } // --- Check 1: sentinel --- @@ -54,9 +54,8 @@ func TestProveCoverage_NoSentinelIsUnobservable(t *testing.T) { def := Definition{UID: "r1", Title: "R1"} res := proveCoverage(Header{StartedAt: from.Add(-time.Hour)}, nil, nil, rt, def, from, to, 0) - if res.Proved || res.Reason != ReasonNoSentinel { - t.Fatalf("res = %+v, want unobservable/no_sentinel: an absent sentinel must never be a pass", res) - } + require.False(t, res.Proved) + require.Equal(t, ReasonNoSentinel, res.Reason, "an absent sentinel must never be a pass") } func TestProveCoverage_SentinelBeforeGraceIsUnobservable(t *testing.T) { @@ -68,12 +67,9 @@ func TestProveCoverage_SentinelBeforeGraceIsUnobservable(t *testing.T) { sentinel := to.Add(grace).Add(-time.Second) // one second short of to+grace res := proveCoverage(Header{StartedAt: from.Add(-time.Hour)}, nil, &sentinel, rt, def, from, to, grace) - if res.Reason != ReasonSentinelEarly { - t.Fatalf("Reason = %q, want sentinel_early", res.Reason) - } - if !res.Unobservable || res.Proved { - t.Fatalf("res = %+v, want Unobservable and not Proved — a reason string with no consequence is not a coverage failure", res) - } + require.Equal(t, ReasonSentinelEarly, res.Reason) + require.True(t, res.Unobservable) + require.False(t, res.Proved, "a reason string with no consequence is not a coverage failure") // The consequence: decide() must turn this into exit 2, never a pass. defs := []Definition{def} @@ -81,12 +77,9 @@ func TestProveCoverage_SentinelBeforeGraceIsUnobservable(t *testing.T) { gt := globalTimings{transitionGrace: grace} pol := Policy{From: from, To: to} dres, err := decide(Header{StartedAt: from.Add(-time.Hour)}, nil, &sentinel, defs, drt, gt, pol) - if err == nil { - t.Fatalf("decide() err = nil, want non-nil: a sentinel short of to+grace must fail the run") - } - if len(dres.Verdicts) != 1 || dres.Verdicts[0].Outcome != OutcomeUnobservable { - t.Fatalf("Verdicts = %+v, want one unobservable verdict", dres.Verdicts) - } + require.Error(t, err, "a sentinel short of to+grace must fail the run") + require.Len(t, dres.Verdicts, 1) + require.Equal(t, OutcomeUnobservable, dres.Verdicts[0].Outcome) } func TestProveCoverage_SentinelExactlyAtGraceIsFine(t *testing.T) { @@ -104,9 +97,7 @@ func TestProveCoverage_SentinelExactlyAtGraceIsFine(t *testing.T) { sentinel := windowEnd res := proveCoverage(Header{StartedAt: from.Add(-time.Hour)}, polls, &sentinel, rt, def, from, to, grace) - if !res.Proved { - t.Fatalf("Proved = false, want true: sentinel exactly at to+grace must satisfy check 1: %+v", res) - } + require.True(t, res.Proved, "sentinel exactly at to+grace must satisfy check 1") } // --- Check 2: from bounds --- @@ -120,12 +111,9 @@ func TestProveCoverage_FromBeforeRecordIsUnobservable(t *testing.T) { sentinel := to res := proveCoverage(Header{StartedAt: started}, nil, &sentinel, rt, def, from, to, 0) - if res.Reason != ReasonFromBeforeRecord { - t.Fatalf("Reason = %q, want from_before_record", res.Reason) - } - if !res.Unobservable || res.Proved { - t.Fatalf("res = %+v, want Unobservable and not Proved — a reason string with no consequence is not a coverage failure", res) - } + require.Equal(t, ReasonFromBeforeRecord, res.Reason) + require.True(t, res.Unobservable) + require.False(t, res.Proved) // The consequence: decide() must turn this into exit 2, never a pass. defs := []Definition{def} @@ -133,12 +121,66 @@ func TestProveCoverage_FromBeforeRecordIsUnobservable(t *testing.T) { gt := globalTimings{} pol := Policy{From: from, To: to} dres, err := decide(Header{StartedAt: started}, nil, &sentinel, defs, drt, gt, pol) - if err == nil { - t.Fatalf("decide() err = nil, want non-nil: `from` before the recording started must fail the run") - } - if len(dres.Verdicts) != 1 || dres.Verdicts[0].Outcome != OutcomeUnobservable { - t.Fatalf("Verdicts = %+v, want one unobservable verdict", dres.Verdicts) + require.Error(t, err, "`from` before the recording started must fail the run") + require.Len(t, dres.Verdicts, 1) + require.Equal(t, OutcomeUnobservable, dres.Verdicts[0].Outcome) +} + +// The from-bounds check compares at whole-second granularity: a whole-second +// `from` may precede the recorder's sub-second StartedAt INSIDE the same second +// without being judged early. That one sliver is the --from truncation, not a +// blind interval, so the window is still proved. +func TestProveCoverage_FromSameSecondAsStartedAtIsProved(t *testing.T) { + from := time.Date(2026, 1, 1, 0, 0, 0, 0, time.UTC) + to := from.Add(10 * time.Minute) + started := from.Add(500 * time.Millisecond) + rt := newRuleTimings(30*time.Second, 60) + def := Definition{UID: "r1", Title: "R1"} + + var polls []Poll + for ts := from; !ts.After(to); ts = ts.Add(30 * time.Second) { + polls = append(polls, Poll{RuleUID: "r1", GrafanaNow: ts, Found: true, Health: "ok", State: "inactive", LastEvaluation: ts}) } + sentinel := to + + res := proveCoverage(Header{StartedAt: started}, polls, &sentinel, rt, def, from, to, 0) + require.True(t, res.Proved) + require.False(t, res.Unobservable) + require.Empty(t, res.Reason) +} + +// Exactly one whole second later is a different second: even at the boundary, +// the whole-second comparison reads it as before, however healthy the polls. +func TestProveCoverage_FromExactlyOneSecondBeforeStartedAtIsUnobservable(t *testing.T) { + from := time.Date(2026, 1, 1, 0, 0, 0, 0, time.UTC) + to := from.Add(10 * time.Minute) + started := from.Add(time.Second) + rt := newRuleTimings(30*time.Second, 60) + def := Definition{UID: "r1", Title: "R1"} + + sentinel := to + res := proveCoverage(Header{StartedAt: started}, nil, &sentinel, rt, def, from, to, 0) + require.Equal(t, ReasonFromBeforeRecord, res.Reason) + require.True(t, res.Unobservable) + require.False(t, res.Proved) +} + +// A sub-second sliver that straddles the second boundary is still "before": +// 900ms into one second vs 100ms into the next are distinct seconds, so the +// 200ms gap is a from_before_record, not rounding noise. +func TestProveCoverage_FromSubSecondEarlierAcrossSecondBoundaryIsUnobservable(t *testing.T) { + base := time.Date(2026, 1, 1, 0, 0, 0, 0, time.UTC) + from := base.Add(900 * time.Millisecond) + started := base.Add(time.Second + 100*time.Millisecond) + to := from.Add(10 * time.Minute) + rt := newRuleTimings(30*time.Second, 60) + def := Definition{UID: "r1", Title: "R1"} + + sentinel := to + res := proveCoverage(Header{StartedAt: started}, nil, &sentinel, rt, def, from, to, 0) + require.Equal(t, ReasonFromBeforeRecord, res.Reason) + require.True(t, res.Unobservable) + require.False(t, res.Proved) } // --- Check 3: heartbeat continuity --- @@ -158,18 +200,11 @@ func TestProveCoverage_HeartbeatGapBetweenBoundariesIsUnobservable(t *testing.T) sentinel := to res := proveCoverage(Header{StartedAt: from.Add(-time.Hour)}, polls, &sentinel, rt, def, from, to, 0) - if res.Reason != ReasonHeartbeatGap { - t.Fatalf("Reason = %q, want heartbeat_gap: healthy edges with a hole in the middle must still fail", res.Reason) - } + require.Equal(t, ReasonHeartbeatGap, res.Reason, "healthy edges with a hole in the middle must still fail") // The gap is the SPACING between the two polls (598s), not either // boundary segment (1s each) — pin the actual values, not just the verdict. - if res.LargestGap != 598*time.Second { - t.Fatalf("LargestGap = %s, want 598s (the spacing between the two polls, not a boundary segment)", res.LargestGap) - } - wantAt := from.Add(time.Second) - if !res.LargestGapAt.Equal(wantAt) { - t.Fatalf("LargestGapAt = %s, want %s (where the gap starts, at the first poll)", res.LargestGapAt, wantAt) - } + require.Equal(t, 598*time.Second, res.LargestGap) + require.True(t, res.LargestGapAt.Equal(from.Add(time.Second))) } // --- Check 4/5: health --- @@ -192,12 +227,8 @@ func TestProveCoverage_HealthErrorShortBlipPassesWithNote(t *testing.T) { sentinel := to res := proveCoverage(Header{StartedAt: from.Add(-time.Hour)}, polls, &sentinel, rt, def, from, to, 0) - if !res.Proved { - t.Fatalf("Proved = false, want true: one failed evaluation must not fail an otherwise clean window: %+v", res) - } - if !anyContains(res.Notes, "health=error") { - t.Fatalf("Notes = %v, want a health=error note even though it did not fail the window", res.Notes) - } + require.True(t, res.Proved, "one failed evaluation must not fail an otherwise clean window") + require.True(t, anyContains(res.Notes, "health=error"), "want a health=error note even though it did not fail the window") } func TestProveCoverage_HealthErrorSustainedIsUnobservable(t *testing.T) { @@ -218,9 +249,7 @@ func TestProveCoverage_HealthErrorSustainedIsUnobservable(t *testing.T) { sentinel := to res := proveCoverage(Header{StartedAt: from.Add(-time.Hour)}, polls, &sentinel, rt, def, from, to, 0) - if res.Reason != ReasonHealthError { - t.Fatalf("Reason = %q, want health_error for a run that outlasts healthGrace", res.Reason) - } + require.Equal(t, ReasonHealthError, res.Reason, "a run that outlasts healthGrace") } func TestProveCoverage_HealthNodataNeverFatalHere(t *testing.T) { @@ -236,13 +265,8 @@ func TestProveCoverage_HealthNodataNeverFatalHere(t *testing.T) { sentinel := to res := proveCoverage(Header{StartedAt: from.Add(-time.Hour)}, polls, &sentinel, rt, def, from, to, 0) - if !res.Proved { - t.Fatalf("Proved = false, want true: health=nodata for the WHOLE window must still not be fatal by itself "+ - "(escalating it is Policy.NodataIsUnobservable's job, applied by decide): %+v", res) - } - if !anyContains(res.Notes, "health=nodata") { - t.Fatalf("Notes = %v, want a health=nodata note", res.Notes) - } + require.True(t, res.Proved, "health=nodata for the WHOLE window must still not be fatal by itself") + require.True(t, anyContains(res.Notes, "health=nodata")) } // --- Check 6: liveness --- @@ -270,13 +294,9 @@ func TestProveCoverage_LivenessAbsoluteNeverFalseStale(t *testing.T) { sentinel := windowEnd res := proveCoverage(Header{StartedAt: from.Add(-time.Hour)}, polls, &sentinel, rt, def, from, windowEnd, 0) - if res.Reason == ReasonStaleEvaluation || res.BlindFor != 0 { - t.Fatalf("proveCoverage flagged staleness on a healthy rule polled at intervalSeconds/2 — liveness must be absolute, "+ - "never a delta against a previous poll: %+v", res) - } - if !res.Proved { - t.Fatalf("Proved = false, want true: %+v (notes: %v)", res, res.Notes) - } + require.NotEqual(t, ReasonStaleEvaluation, res.Reason, "liveness must be absolute, never a delta against a previous poll") + require.Zero(t, res.BlindFor) + require.True(t, res.Proved) } func TestProveCoverage_StaleEvaluationIsUnobservable(t *testing.T) { @@ -297,12 +317,8 @@ func TestProveCoverage_StaleEvaluationIsUnobservable(t *testing.T) { sentinel := to res := proveCoverage(Header{StartedAt: from.Add(-time.Hour)}, polls, &sentinel, rt, def, from, to, 0) - if res.Reason != ReasonStaleEvaluation { - t.Fatalf("Reason = %q, want stale_evaluation", res.Reason) - } - if res.BlindFor != 3*time.Minute { - t.Fatalf("BlindFor = %s, want 3m", res.BlindFor) - } + require.Equal(t, ReasonStaleEvaluation, res.Reason) + require.Equal(t, 3*time.Minute, res.BlindFor) } func TestProveCoverage_ZeroLastEvaluationNeverFalseStale(t *testing.T) { @@ -319,9 +335,7 @@ func TestProveCoverage_ZeroLastEvaluationNeverFalseStale(t *testing.T) { sentinel := to res := proveCoverage(Header{StartedAt: from.Add(-time.Hour)}, polls, &sentinel, rt, def, from, to, 0) - if res.Reason == ReasonStaleEvaluation { - t.Fatalf("a zero lastEvaluation on a paused poll must not trigger check 6: %+v", res) - } + require.NotEqual(t, ReasonStaleEvaluation, res.Reason, "a zero lastEvaluation on a paused poll must not trigger check 6") } // --- Check 7: isPaused in-window --- @@ -345,9 +359,7 @@ func TestProveCoverage_PausedInWindowIsUnobservable(t *testing.T) { sentinel := to res := proveCoverage(Header{StartedAt: from.Add(-time.Hour)}, polls, &sentinel, rt, def, from, to, 0) - if res.Reason != ReasonPausedInWindow { - t.Fatalf("Reason = %q, want paused_in_window", res.Reason) - } + require.Equal(t, ReasonPausedInWindow, res.Reason) } // TestProveCoverage_PausedAfterWindowIsFine pins check 7's respect for the @@ -369,12 +381,8 @@ func TestProveCoverage_PausedAfterWindowIsFine(t *testing.T) { sentinel := to res := proveCoverage(Header{StartedAt: from.Add(-time.Hour)}, polls, &sentinel, rt, def, from, to, 0) - if res.Reason == ReasonPausedInWindow { - t.Fatalf("a paused poll after windowEnd tripped check 7: %+v", res.Notes) - } - if !res.Proved { - t.Fatalf("Proved = false, want a clean window: %+v", res.Notes) - } + require.NotEqual(t, ReasonPausedInWindow, res.Reason, "a paused poll after windowEnd tripped check 7") + require.True(t, res.Proved) } // --- Check 8: rule absent --- @@ -399,9 +407,7 @@ func TestProveCoverage_RuleAbsentIsUnobservable(t *testing.T) { sentinel := to res := proveCoverage(Header{StartedAt: from.Add(-time.Hour)}, polls, &sentinel, rt, def, from, to, 0) - if res.Reason != ReasonRuleAbsent { - t.Fatalf("Reason = %q, want rule_absent", res.Reason) - } + require.Equal(t, ReasonRuleAbsent, res.Reason) } // denseHealthyPolls builds a clean poll sequence at a fixed cadence, with @@ -436,12 +442,8 @@ func TestProveCoverage_KeepLastObservedIsNoteOnly(t *testing.T) { sentinel := to res := proveCoverage(Header{StartedAt: from.Add(-time.Hour)}, polls, &sentinel, rt, def, from, to, 0) - if !res.Proved { - t.Fatalf("Proved = false, want true: KeepLast is a note, never fatal: %+v", res) - } - if !anyContains(res.Notes, "KeepLast") { - t.Fatalf("Notes = %v, want a KeepLast note (comma-joined membership, not a literal-key match)", res.Notes) - } + require.True(t, res.Proved, "KeepLast is a note, never fatal") + require.True(t, anyContains(res.Notes, "KeepLast"), "comma-joined membership, not a literal-key match") } // KeepLast in the CONFIGURATION gives a note — a different claim from the @@ -468,12 +470,9 @@ func TestProveCoverage_KeepLastConfiguredIsNoteOnly(t *testing.T) { sentinel := to res := proveCoverage(Header{StartedAt: from.Add(-time.Hour)}, polls, &sentinel, rt, tc.def, from, to, 0) - if !res.Proved { - t.Fatalf("Proved = false, want true: a declared KeepLast is a note, never fatal: %+v", res) - } - if !anyContains(res.Notes, "KeepLast") { - t.Fatalf("Notes = %v, want a KeepLast note from the definition alone, with zero KeepLast reasons observed", res.Notes) - } + require.True(t, res.Proved, "a declared KeepLast is a note, never fatal") + require.True(t, anyContains(res.Notes, "KeepLast"), + "from the definition alone, with zero KeepLast reasons observed") }) } } @@ -503,9 +502,7 @@ func TestProveCoverage_SkewTranslationAtWindowBoundary(t *testing.T) { sentinel := to res := proveCoverage(Header{StartedAt: from.Add(-time.Hour)}, polls, &sentinel, rt, def, from, to, 0) - if !res.Proved { - t.Fatalf("res = %+v, want proved: a constant clock skew must not itself read as a coverage gap", res) - } + require.True(t, res.Proved, "a constant clock skew must not itself read as a coverage gap") } // --- Override round-trip: one authority for the cadence --- @@ -524,9 +521,7 @@ func TestProveCoverage_OverrideRoundTrip(t *testing.T) { } defs := []Definition{{UID: "r1", Title: "R1", IntervalSeconds: 60}} rt, _, err := DeriveTimingsFromLog(h, defs) - if err != nil { - t.Fatalf("DeriveTimingsFromLog: %v", err) - } + require.NoError(t, err) var polls []Poll for ts := from; !ts.After(windowEnd); ts = ts.Add(120 * time.Second) { @@ -535,9 +530,8 @@ func TestProveCoverage_OverrideRoundTrip(t *testing.T) { sentinel := windowEnd res := proveCoverage(h, polls, &sentinel, rt["r1"], defs[0], from, windowEnd, 0) - if !res.Proved { - t.Fatalf("Proved = false, want true (maxGap must come from the recorded 120s cadence, not the 30s default): %+v", res) - } + require.True(t, res.Proved, + "maxGap must come from the recorded 120s cadence, not the 30s default") }) t.Run("faster override still catches a real recorder gap", func(t *testing.T) { @@ -548,9 +542,7 @@ func TestProveCoverage_OverrideRoundTrip(t *testing.T) { } defs := []Definition{{UID: "r1", Title: "R1", IntervalSeconds: 300}} rt, _, err := DeriveTimingsFromLog(h, defs) - if err != nil { - t.Fatalf("DeriveTimingsFromLog: %v", err) - } + require.NoError(t, err) var polls []Poll ts := from @@ -571,10 +563,8 @@ func TestProveCoverage_OverrideRoundTrip(t *testing.T) { sentinel := windowEnd res := proveCoverage(h, polls, &sentinel, rt["r1"], defs[0], from, windowEnd, 0) - if res.Reason != ReasonHeartbeatGap { - t.Fatalf("Reason = %q, want heartbeat_gap: if maxGap had been re-derived from the 300s definition instead of "+ - "the recorded 5s cadence, this 250s gap would pass silently — the fail-open direction", res.Reason) - } + require.Equal(t, ReasonHeartbeatGap, res.Reason, + "if maxGap had been re-derived from the 300s definition, this 250s gap would pass silently") }) } @@ -612,10 +602,8 @@ func TestProveCoverage_ZeroLastEvaluationWithoutPauseIsStale(t *testing.T) { sentinel := to res := proveCoverage(Header{StartedAt: from.Add(-time.Hour)}, polls, &sentinel, rt, def, from, to, 0) - if res.Reason != ReasonStaleEvaluation { - t.Fatalf("Reason = %q, want stale_evaluation: a zero lastEvaluation on a found, non-paused poll must fail "+ - "closed, not be silently skipped as if it were a legitimately paused observation", res.Reason) - } + require.Equal(t, ReasonStaleEvaluation, res.Reason, + "a zero lastEvaluation on a found, non-paused poll must fail closed") } // A lastEvaluation in the future of grafana_now (corrupted log) must fail closed. @@ -635,10 +623,8 @@ func TestProveCoverage_FutureLastEvaluationIsUnobservable(t *testing.T) { sentinel := to res := proveCoverage(Header{StartedAt: from.Add(-time.Hour)}, polls, &sentinel, rt, def, from, to, 0) - if res.Reason != ReasonFutureEvaluation { - t.Fatalf("Reason = %q, want future_evaluation: a lastEvaluation in the future of grafana_now must fail "+ - "closed rather than read its negative staleness as fresh", res.Reason) - } + require.Equal(t, ReasonFutureEvaluation, res.Reason, + "a lastEvaluation in the future of grafana_now must fail closed rather than read its negative staleness as fresh") } // --- Check 3, tightened: the boundary segments must widen by the skew bound --- @@ -663,10 +649,8 @@ func TestProveCoverage_BoundaryGapWidensBySkewBound(t *testing.T) { sentinel := to res := proveCoverage(Header{StartedAt: from.Add(-time.Hour)}, polls, &sentinel, rt, def, from, to, 0) - if res.Reason != ReasonHeartbeatGap { - t.Fatalf("Reason = %q, want heartbeat_gap: the leading boundary segment sits at EXACTLY maxGap (60s) before "+ - "widening; the poll's own %s skew bound must push it past the threshold, not just the skew translation", res.Reason, bound) - } + require.Equal(t, ReasonHeartbeatGap, res.Reason, + "the poll's own %s skew bound must push it past the threshold", bound) } // --- Multi-failure contract --- @@ -698,16 +682,10 @@ func TestProveCoverage_MultipleFailuresReasonIsFirstButAllNoted(t *testing.T) { sentinel := to res := proveCoverage(Header{StartedAt: from.Add(-time.Hour)}, polls, &sentinel, rt, def, from, to, 0) - if res.Reason != ReasonPausedInWindow { - t.Fatalf("Reason = %q, want paused_in_window: the FIRST check to fail names the reason", res.Reason) - } - if !anyContains(res.Notes, "paused") { - t.Fatalf("Notes = %v, want a note about the pause", res.Notes) - } - if !anyContains(res.Notes, "no rule") { - t.Fatalf("Notes = %v, want a note about the absence too — a later failure must still be recorded, "+ - "not swallowed once Reason is already set", res.Notes) - } + require.Equal(t, ReasonPausedInWindow, res.Reason, "the FIRST check to fail names the reason") + require.True(t, anyContains(res.Notes, "paused")) + require.True(t, anyContains(res.Notes, "no rule"), + "a later failure must still be recorded, not swallowed once Reason is already set") } // --- Skipped rules --- @@ -728,9 +706,6 @@ func TestProveCoverage_SkippedRuleWithZeroPollsPinnedAsHeartbeatGap(t *testing.T sentinel := to res := proveCoverage(Header{StartedAt: from.Add(-time.Hour)}, nil, &sentinel, rt, def, from, to, 0) - if res.Reason != ReasonHeartbeatGap { - t.Fatalf("Reason = %q, want heartbeat_gap (pinned, not the desired end state): proveCoverage has no "+ - "'skipped' concept, so decide must handle a skipped rule's classification itself, before or "+ - "instead of calling this function", res.Reason) - } + require.Equal(t, ReasonHeartbeatGap, res.Reason, + "proveCoverage has no 'skipped' concept, so decide must handle a skipped rule's classification itself") } diff --git a/grafana-alertcheck/internal/gate/duration_test.go b/grafana-alertcheck/internal/gate/duration_test.go index ba3a1629a..6262c713e 100644 --- a/grafana-alertcheck/internal/gate/duration_test.go +++ b/grafana-alertcheck/internal/gate/duration_test.go @@ -3,6 +3,8 @@ package gate import ( "testing" "time" + + "github.com/stretchr/testify/require" ) func TestParsePromDuration(t *testing.T) { @@ -35,13 +37,8 @@ func TestParsePromDuration(t *testing.T) { } for _, c := range cases { got, err := ParsePromDuration(c.in) - if err != nil { - t.Errorf("ParsePromDuration(%q): unexpected error: %v", c.in, err) - continue - } - if got != c.want { - t.Errorf("ParsePromDuration(%q) = %v, want %v", c.in, got, c.want) - } + require.NoErrorf(t, err, "ParsePromDuration(%q)", c.in) + require.Equalf(t, c.want, got, "ParsePromDuration(%q)", c.in) } } @@ -60,8 +57,7 @@ func TestParsePromDuration_Errors(t *testing.T) { "carrot", // completely invalid } for _, in := range cases { - if _, err := ParsePromDuration(in); err == nil { - t.Errorf("ParsePromDuration(%q): expected an error, got none", in) - } + _, err := ParsePromDuration(in) + require.Errorf(t, err, "ParsePromDuration(%q): expected an error, got none", in) } } diff --git a/grafana-alertcheck/internal/gate/flock_test.go b/grafana-alertcheck/internal/gate/flock_test.go index 3e9ef268c..3dfdd8b73 100644 --- a/grafana-alertcheck/internal/gate/flock_test.go +++ b/grafana-alertcheck/internal/gate/flock_test.go @@ -5,18 +5,16 @@ import ( "fmt" "syscall" "testing" + + "github.com/stretchr/testify/require" ) func TestIsLockContention(t *testing.T) { contended := []error{syscall.EWOULDBLOCK, syscall.EAGAIN} for _, e := range contended { - if !isLockContention(e) { - t.Errorf("isLockContention(%v) = false, want true", e) - } + require.Truef(t, isLockContention(e), "isLockContention(%v)", e) // lockExclusive wraps the raw error via fmt.Errorf("flock: %w", ...). - if !isLockContention(fmt.Errorf("flock: %w", e)) { - t.Errorf("isLockContention(wrapped %v) = false, want true", e) - } + require.Truef(t, isLockContention(fmt.Errorf("flock: %w", e)), "isLockContention(wrapped %v)", e) } notContended := []error{ @@ -28,11 +26,7 @@ func TestIsLockContention(t *testing.T) { errors.New("something else"), } for _, e := range notContended { - if isLockContention(e) { - t.Errorf("isLockContention(%v) = true, want false (not a contender)", e) - } - if isLockContention(fmt.Errorf("flock: %w", e)) { - t.Errorf("isLockContention(wrapped %v) = true, want false", e) - } + require.Falsef(t, isLockContention(e), "isLockContention(%v)", e) + require.Falsef(t, isLockContention(fmt.Errorf("flock: %w", e)), "isLockContention(wrapped %v)", e) } } diff --git a/grafana-alertcheck/internal/gate/jsonreq_test.go b/grafana-alertcheck/internal/gate/jsonreq_test.go index bf3881e4b..72b446503 100644 --- a/grafana-alertcheck/internal/gate/jsonreq_test.go +++ b/grafana-alertcheck/internal/gate/jsonreq_test.go @@ -3,14 +3,14 @@ package gate import ( "encoding/json" "testing" + + "github.com/stretchr/testify/require" ) func rawMap(t *testing.T, jsonObj string) map[string]json.RawMessage { t.Helper() var m map[string]json.RawMessage - if err := json.Unmarshal([]byte(jsonObj), &m); err != nil { - t.Fatalf("rawMap: %v", err) - } + require.NoError(t, json.Unmarshal([]byte(jsonObj), &m)) return m } @@ -19,34 +19,24 @@ func TestReq(t *testing.T) { t.Run("present key decodes", func(t *testing.T) { var s string - if err := req(m, "present", &s); err != nil { - t.Fatalf("unexpected error: %v", err) - } - if s != "hello" { - t.Errorf("got %q, want hello", s) - } + require.NoError(t, req(m, "present", &s)) + require.Equal(t, "hello", s) }) t.Run("absent key errors", func(t *testing.T) { var s string - if err := req(m, "missing", &s); err == nil { - t.Fatalf("expected an error, got none") - } + require.Error(t, req(m, "missing", &s)) }) t.Run("wrong type errors", func(t *testing.T) { var s string - if err := req(m, "wrongtype", &s); err == nil { - t.Fatalf("expected an error, got none") - } + require.Error(t, req(m, "wrongtype", &s)) }) t.Run("explicit JSON null errors, never a zero value", func(t *testing.T) { var s string err := req(m, "nullval", &s) - if err == nil { - t.Fatalf("expected an error, got none (s=%q) — a null required field must not silently become a zero value", s) - } + require.Error(t, err, "a null required field must not silently become a zero value") }) } @@ -55,38 +45,24 @@ func TestOpt(t *testing.T) { t.Run("present key decodes", func(t *testing.T) { var s string - if err := opt(m, "present", &s); err != nil { - t.Fatalf("unexpected error: %v", err) - } - if s != "hello" { - t.Errorf("got %q, want hello", s) - } + require.NoError(t, opt(m, "present", &s)) + require.Equal(t, "hello", s) }) t.Run("absent key leaves dst untouched", func(t *testing.T) { s := "unchanged" - if err := opt(m, "missing", &s); err != nil { - t.Fatalf("unexpected error: %v", err) - } - if s != "unchanged" { - t.Errorf("got %q, want unchanged", s) - } + require.NoError(t, opt(m, "missing", &s)) + require.Equal(t, "unchanged", s) }) t.Run("wrong type errors", func(t *testing.T) { var s string - if err := opt(m, "wrongtype", &s); err == nil { - t.Fatalf("expected an error, got none") - } + require.Error(t, opt(m, "wrongtype", &s)) }) t.Run("explicit JSON null leaves dst at its zero value", func(t *testing.T) { var s string - if err := opt(m, "nullval", &s); err != nil { - t.Fatalf("unexpected error: %v", err) - } - if s != "" { - t.Errorf("got %q, want empty string", s) - } + require.NoError(t, opt(m, "nullval", &s)) + require.Equal(t, "", s) }) } diff --git a/grafana-alertcheck/internal/gate/log_test.go b/grafana-alertcheck/internal/gate/log_test.go index 179adb5fe..8ef3863e1 100644 --- a/grafana-alertcheck/internal/gate/log_test.go +++ b/grafana-alertcheck/internal/gate/log_test.go @@ -5,17 +5,18 @@ import ( "fmt" "os" "path/filepath" - "reflect" "strings" "sync" "testing" "time" + + "github.com/stretchr/testify/require" ) // Every time literal in this file is UTC and built with time.Date, so it // carries no monotonic reading and survives a JSON round trip byte-identical — -// which is what lets the round-trip tests below use reflect.DeepEqual on whole -// Poll values instead of comparing field by field. +// which is what lets the round-trip tests below compare whole Poll values +// instead of comparing field by field. var testNow = time.Date(2026, 8, 31, 9, 0, 0, 0, time.UTC) func testInstance(state State, reason, instanceLabel string) Instance { @@ -57,30 +58,20 @@ func TestLogReduceKeepsOnlyAbnormalInstances(t *testing.T) { p := NewReducer().Reduce("rule1", observation(testNow, rule)) - if !p.Found { - t.Fatalf("Found = false, want true") - } - if len(p.Abnormal) != 1 || p.Abnormal[0].Labels["instance"] != "b" { - t.Errorf("Abnormal = %+v, want only the firing instance b", p.Abnormal) - } - if want := map[string]int{"NoData": 1, "Error": 1}; !reflect.DeepEqual(p.Reasons, want) { - t.Errorf("Reasons = %v, want %v", p.Reasons, want) - } + require.True(t, p.Found) + require.Len(t, p.Abnormal, 1) + require.Equal(t, "b", p.Abnormal[0].Labels["instance"]) + require.Equal(t, map[string]int{"NoData": 1, "Error": 1}, p.Reasons) // The histogram is a verbatim copy of the response totals — raw keys, no // normalization. - if want := map[string]int{"alerting": 1, "normal": 2}; !reflect.DeepEqual(p.Histogram, want) { - t.Errorf("Histogram = %v, want %v", p.Histogram, want) - } + require.Equal(t, map[string]int{"alerting": 1, "normal": 2}, p.Histogram) // Rule-level state and health stay raw and unnormalized. - if p.State != "firing" || p.Health != "ok" { - t.Errorf("State/Health = %q/%q, want firing/ok", p.State, p.Health) - } - if p.Skew() != 1500*time.Millisecond || p.SkewBound() != 40*time.Millisecond || p.Latency() != 1800*time.Millisecond { - t.Errorf("durations = %s/%s/%s, want 1.5s/40ms/1.8s", p.Skew(), p.SkewBound(), p.Latency()) - } - if p.Reasons["MissingSeries"] != 0 { - t.Errorf("unexpected MissingSeries count") - } + require.Equal(t, "firing", p.State) + require.Equal(t, "ok", p.Health) + require.Equal(t, 1500*time.Millisecond, p.Skew()) + require.Equal(t, 40*time.Millisecond, p.SkewBound()) + require.Equal(t, 1800*time.Millisecond, p.Latency()) + require.Zero(t, p.Reasons["MissingSeries"]) } // A filtered response can hold several rules sharing one title, so the reducer @@ -94,9 +85,8 @@ func TestLogReduceSelectsRuleByUID(t *testing.T) { p := NewReducer().Reduce("ruleB", observation(testNow, first, second)) - if p.Health != "error" || len(p.Abnormal) != 1 { - t.Errorf("reduced the wrong rule: %+v", p) - } + require.Equal(t, "error", p.Health) + require.Len(t, p.Abnormal, 1) } func TestLogReduceRuleAbsentIsAuthoritative(t *testing.T) { @@ -104,20 +94,14 @@ func TestLogReduceRuleAbsentIsAuthoritative(t *testing.T) { p := NewReducer().Reduce("rule1", observation(testNow, other)) - if p.Found { - t.Errorf("Found = true, want false for a rule absent from an authoritative 2xx") - } - if p.RuleUID != "rule1" { - t.Errorf("RuleUID = %q, want rule1 — an absent rule is still attributed", p.RuleUID) - } + require.False(t, p.Found, "a rule absent from an authoritative 2xx") + require.Equal(t, "rule1", p.RuleUID, "an absent rule is still attributed") // The heartbeat still exists: a not-found poll is evidence that Grafana // answered at this time, which the coverage proof reads. - if !p.GrafanaNow.Equal(testNow) || p.Latency() == 0 { - t.Errorf("absent-rule poll lost its timing evidence: %+v", p) - } - if p.Health != "" || p.Abnormal != nil { - t.Errorf("absent-rule poll carries rule fields: %+v", p) - } + require.True(t, p.GrafanaNow.Equal(testNow)) + require.NotZero(t, p.Latency()) + require.Empty(t, p.Health) + require.Nil(t, p.Abnormal) } // An instance that leaves the abnormal set is resolved against the SAME @@ -175,19 +159,14 @@ func TestTransitionMarkersClearedVersusVanished(t *testing.T) { Instances: []Instance{testInstance(StateFiring, "", "b")}, } first := r.Reduce("rule1", observation(testNow, firing)) - if first.Cleared != nil || first.Vanished != nil { - t.Fatalf("first poll produced markers with no previous poll: %+v", first) - } + require.Nil(t, first.Cleared, "first poll produced cleared markers with no previous poll") + require.Nil(t, first.Vanished, "first poll produced vanished markers with no previous poll") next := StateRule{UID: "rule1", Health: "ok", State: "inactive", LastEvaluation: testNow, Instances: c.second} p := r.Reduce("rule1", observation(testNow.Add(30*time.Second), next)) - if !reflect.DeepEqual(p.Cleared, c.wantCleared) { - t.Errorf("Cleared = %q, want %q", p.Cleared, c.wantCleared) - } - if !reflect.DeepEqual(p.Vanished, c.wantVanished) { - t.Errorf("Vanished = %q, want %q", p.Vanished, c.wantVanished) - } + require.Equal(t, c.wantCleared, p.Cleared) + require.Equal(t, c.wantVanished, p.Vanished) }) } } @@ -204,15 +183,12 @@ func TestTransitionMarkersSurviveAnAbsentPoll(t *testing.T) { r.Reduce("rule1", observation(testNow, firing)) absent := r.Reduce("rule1", observation(testNow.Add(30*time.Second))) - if absent.Vanished != nil || absent.Cleared != nil { - t.Fatalf("an absent rule produced markers: %+v", absent) - } + require.Nil(t, absent.Vanished, "an absent rule produced vanished markers") + require.Nil(t, absent.Cleared, "an absent rule produced cleared markers") back := StateRule{UID: "rule1", Health: "ok", State: "inactive", LastEvaluation: testNow} p := r.Reduce("rule1", observation(testNow.Add(60*time.Second), back)) - if len(p.Vanished) != 1 { - t.Errorf("Vanished = %q, want the instance that disappeared across the absent poll", p.Vanished) - } + require.Len(t, p.Vanished, 1, "want the instance that disappeared across the absent poll") } func TestTransitionMarkersAreSortedAndPerRule(t *testing.T) { @@ -234,19 +210,14 @@ func TestTransitionMarkersAreSortedAndPerRule(t *testing.T) { clearedOne := StateRule{UID: "rule1", Health: "ok", State: "inactive", LastEvaluation: testNow} p := r.Reduce("rule1", observation(testNow.Add(time.Minute), clearedOne, ruleTwo)) - if len(p.Vanished) != 3 { - t.Fatalf("Vanished = %q, want 3 keys", p.Vanished) - } + require.Len(t, p.Vanished, 3) for i := 1; i < len(p.Vanished); i++ { - if p.Vanished[i-1] > p.Vanished[i] { - t.Errorf("Vanished is not sorted: %q", p.Vanished) - } + require.Less(t, p.Vanished[i-1], p.Vanished[i], "Vanished is not sorted: %q", p.Vanished) } // rule2's own abnormal set is untouched by rule1's transitions. q := r.Reduce("rule2", observation(testNow.Add(time.Minute), clearedOne, ruleTwo)) - if q.Cleared != nil || q.Vanished != nil { - t.Errorf("rule2 picked up rule1's transitions: %+v", q) - } + require.Nil(t, q.Cleared, "rule2 picked up rule1's transitions") + require.Nil(t, q.Vanished, "rule2 picked up rule1's transitions") } // The reduction depends on the state endpoint returning normal instances. If it @@ -267,22 +238,14 @@ func TestLogVerifyNormalInstancesVisible(t *testing.T) { for _, c := range cases { t.Run(c.fixture, func(t *testing.T) { rules, err := ParseState(readFixture(t, c.fixture)) - if err != nil { - t.Fatalf("ParseState: %v", err) - } + require.NoError(t, err) err = VerifyNormalInstancesVisible(rules) if c.wantError { - if err == nil { - t.Fatalf("VerifyNormalInstancesVisible: want an error, got nil") - } - if !strings.Contains(err.Error(), "no longer returns normal instances") { - t.Errorf("error does not say the endpoint stopped returning normal instances: %v", err) - } + require.Error(t, err) + require.Contains(t, err.Error(), "no longer returns normal instances") return } - if err != nil { - t.Fatalf("VerifyNormalInstancesVisible: unexpected error: %v", err) - } + require.NoError(t, err) }) } } @@ -310,9 +273,7 @@ func TestLogVerifyNormalInstancesVisibleVocabularies(t *testing.T) { Instances: []Instance{testInstance(StateFiring, "", "b")}, }} err := VerifyNormalInstancesVisible(rules) - if (err != nil) != c.wantError { - t.Errorf("VerifyNormalInstancesVisible: error = %v, want error = %v", err, c.wantError) - } + require.Equal(t, c.wantError, err != nil) }) } } @@ -328,37 +289,25 @@ func TestLogModeCadenceComesFromTheHeader(t *testing.T) { h.Rules[0].PollEverySeconds = 5 // an operator override far tighter than the default 150s rt, _, err := DeriveTimingsFromLog(h, defs) - if err != nil { - t.Fatalf("DeriveTimingsFromLog: %v", err) - } + require.NoError(t, err) got := rt["rule1"] - if got.pollEvery != 5*time.Second { - t.Errorf("pollEvery = %s, want the header's 5s, not the default 150s", got.pollEvery) - } + require.Equal(t, 5*time.Second, got.pollEvery, "the header's 5s, not the default 150s") // maxGap and healthGrace follow the recorded cadence; without this a 250s // hole in a log recorded at 5s would pass silently. - if got.maxGap != 10*time.Second { - t.Errorf("maxGap = %s, want 10s (2 x the recorded cadence)", got.maxGap) - } - if got.healthGrace != 300*time.Second { - t.Errorf("healthGrace = %s, want 300s (max(maxGap, interval))", got.healthGrace) - } + require.Equal(t, 10*time.Second, got.maxGap) + require.Equal(t, 300*time.Second, got.healthGrace) // evalStaleAfter is a rule fact, so it stays 2 x intervalSeconds from the // definitions regardless of how often the gate polled. - if got.evalStaleAfter != 600*time.Second { - t.Errorf("evalStaleAfter = %s, want 600s from the definition's interval", got.evalStaleAfter) - } + require.Equal(t, 600*time.Second, got.evalStaleAfter) // A log that cannot say how often it was written cannot have its coverage // proved, and neither can one naming a rule that no longer resolves. missingCadence := testHeader() missingCadence.Rules[0].PollEverySeconds = 0 - if _, _, err := DeriveTimingsFromLog(missingCadence, defs); err == nil { - t.Errorf("a header with no recorded cadence was accepted") - } - if _, _, err := DeriveTimingsFromLog(testHeader(), nil); err == nil { - t.Errorf("a header naming an unresolvable rule was accepted") - } + require.Error(t, func() error { _, _, err := DeriveTimingsFromLog(missingCadence, defs); return err }(), + "a header with no recorded cadence was accepted") + require.Error(t, func() error { _, _, err := DeriveTimingsFromLog(testHeader(), nil); return err }(), + "a header naming an unresolvable rule was accepted") // A duplicated UID must not resolve last-one-wins: the slower duplicate // would widen maxGap, which is fail-open through log corruption alone. @@ -366,9 +315,8 @@ func TestLogModeCadenceComesFromTheHeader(t *testing.T) { slower := duplicated.Rules[0] slower.PollEverySeconds = 600 duplicated.Rules = append(duplicated.Rules, slower) - if _, _, err := DeriveTimingsFromLog(duplicated, defs); err == nil { - t.Errorf("a header naming one rule twice was accepted") - } + require.Error(t, func() error { _, _, err := DeriveTimingsFromLog(duplicated, defs); return err }(), + "a header naming one rule twice was accepted") } // watch polls a fleet concurrently through one Reducer, so the marker @@ -398,9 +346,8 @@ func TestLogReduceIsSafeForConcurrentUse(t *testing.T) { // transition — the concurrency must not corrupt the per-rule state either. for _, rule := range rules { p := r.Reduce(rule.UID, obs) - if p.Cleared != nil || p.Vanished != nil { - t.Errorf("rule %s: markers after concurrent reduction: %+v", rule.UID, p) - } + require.Nilf(t, p.Cleared, "rule %s: cleared markers after concurrent reduction", rule.UID) + require.Nilf(t, p.Vanished, "rule %s: vanished markers after concurrent reduction", rule.UID) } } @@ -409,30 +356,18 @@ func TestLogReduceIsSafeForConcurrentUse(t *testing.T) { func TestLogPollOmitsTheZeroEvaluationTime(t *testing.T) { absent := NewReducer().Reduce("rule1", observation(testNow)) b, err := json.Marshal(pollRecord{Type: RecordPoll, Poll: absent}) - if err != nil { - t.Fatalf("marshal: %v", err) - } - if strings.Contains(string(b), "0001-01-01") { - t.Errorf("a not-found poll wrote the zero time: %s", b) - } - if strings.Contains(string(b), "last_evaluation") { - t.Errorf("a not-found poll wrote last_evaluation at all: %s", b) - } + require.NoError(t, err) + require.NotContains(t, string(b), "0001-01-01") + require.NotContains(t, string(b), "last_evaluation") // A real evaluation time still round-trips. found := StateRule{UID: "rule1", Health: "ok", State: "inactive", LastEvaluation: testNow} p := NewReducer().Reduce("rule1", observation(testNow, found)) b, err = json.Marshal(pollRecord{Type: RecordPoll, Poll: p}) - if err != nil { - t.Fatalf("marshal: %v", err) - } + require.NoError(t, err) var back pollRecord - if err := json.Unmarshal(b, &back); err != nil { - t.Fatalf("unmarshal: %v", err) - } - if !back.LastEvaluation.Equal(testNow) { - t.Errorf("last_evaluation = %s, want %s", back.LastEvaluation, testNow) - } + require.NoError(t, json.Unmarshal(b, &back)) + require.True(t, back.LastEvaluation.Equal(testNow)) } func testHeader() Header { @@ -452,9 +387,7 @@ func newTestWriter(t *testing.T, path string) (*Writer, *fakeClock) { t.Helper() clock := newFakeClock(testNow) w, err := NewWriter(path, clock) - if err != nil { - t.Fatalf("NewWriter: %v", err) - } + require.NoError(t, err) return w, clock } @@ -463,9 +396,7 @@ func TestWriterReadLogRoundTrip(t *testing.T) { w, clock := newTestWriter(t, path) h := testHeader() - if err := w.WriteHeader(h); err != nil { - t.Fatalf("WriteHeader: %v", err) - } + require.NoError(t, w.WriteHeader(h)) r := NewReducer() firing := StateRule{ @@ -479,35 +410,21 @@ func TestWriterReadLogRoundTrip(t *testing.T) { r.Reduce("rule1", observation(testNow.Add(time.Minute), cleared)), } for _, p := range want { - if err := w.WritePoll(p); err != nil { - t.Fatalf("WritePoll: %v", err) - } + require.NoError(t, w.WritePoll(p)) } clock.Advance(2 * time.Minute) - if err := w.Stop(); err != nil { - t.Fatalf("Stop: %v", err) - } + require.NoError(t, w.Stop()) gotHeader, gotPolls, sentinel, err := ReadLog(path) - if err != nil { - t.Fatalf("ReadLog: %v", err) - } + require.NoError(t, err) h.SchemaVersion = LogSchemaVersion // WriteHeader stamps it - if !reflect.DeepEqual(gotHeader, h) { - t.Errorf("header round trip:\n got %+v\nwant %+v", gotHeader, h) - } - if !reflect.DeepEqual(gotPolls, want) { - t.Errorf("poll round trip:\n got %+v\nwant %+v", gotPolls, want) - } - if sentinel == nil { - t.Fatalf("sentinel is nil after Stop") - } + require.Equal(t, h, gotHeader, "header round trip") + require.Equal(t, want, gotPolls, "poll round trip") + require.NotNil(t, sentinel, "sentinel is nil after Stop") // Stop stamps the recorder's own stop time and makes no comparison // against `to` — watch never knows it. - if !sentinel.Equal(testNow.Add(2 * time.Minute)) { - t.Errorf("sentinel = %s, want the writer's stop time %s", sentinel, testNow.Add(2*time.Minute)) - } + require.True(t, sentinel.Equal(testNow.Add(2*time.Minute))) } // The log is append-only. A second run against the same path must never @@ -515,44 +432,25 @@ func TestWriterReadLogRoundTrip(t *testing.T) { func TestWriterAppendsAndNeverTruncates(t *testing.T) { path := filepath.Join(t.TempDir(), "log.jsonl") w, _ := newTestWriter(t, path) - if err := w.WriteHeader(testHeader()); err != nil { - t.Fatalf("WriteHeader: %v", err) - } - if err := w.WritePoll(Poll{RuleUID: "rule1", Found: true, GrafanaNow: testNow}); err != nil { - t.Fatalf("WritePoll: %v", err) - } - if err := w.Close(); err != nil { - t.Fatalf("Close: %v", err) - } + require.NoError(t, w.WriteHeader(testHeader())) + require.NoError(t, w.WritePoll(Poll{RuleUID: "rule1", Found: true, GrafanaNow: testNow})) + require.NoError(t, w.Close()) before, err := os.ReadFile(path) - if err != nil { - t.Fatalf("read: %v", err) - } + require.NoError(t, err) // The handoff: the parent wrote the header and closed; the child // reopens the same path and appends without a second header. child, _ := newTestWriter(t, path) - if err := child.WritePoll(Poll{RuleUID: "rule1", Found: true, GrafanaNow: testNow.Add(time.Minute)}); err != nil { - t.Fatalf("child WritePoll: %v", err) - } - if err := child.Stop(); err != nil { - t.Fatalf("child Stop: %v", err) - } + require.NoError(t, child.WritePoll(Poll{RuleUID: "rule1", Found: true, GrafanaNow: testNow.Add(time.Minute)})) + require.NoError(t, child.Stop()) after, err := os.ReadFile(path) - if err != nil { - t.Fatalf("read: %v", err) - } - if !strings.HasPrefix(string(after), string(before)) { - t.Fatalf("reopening the log rewrote earlier records:\n%s", after) - } + require.NoError(t, err) + require.True(t, strings.HasPrefix(string(after), string(before)), "reopening the log rewrote earlier records") _, polls, sentinel, err := ReadLog(path) - if err != nil { - t.Fatalf("ReadLog: %v", err) - } - if len(polls) != 2 || sentinel == nil { - t.Errorf("got %d polls, sentinel %v; want 2 polls and a sentinel", len(polls), sentinel) - } + require.NoError(t, err) + require.Len(t, polls, 2) + require.NotNil(t, sentinel) } // Two recorders on one log means one of them is recording a window nobody @@ -570,67 +468,41 @@ func TestWriterSecondWriterFails(t *testing.T) { select { case err := <-done: - if err == nil { - t.Fatalf("a second writer took the lock") - } - if !strings.Contains(err.Error(), "another writer") { - t.Errorf("error does not name the conflict: %v", err) - } + require.Error(t, err, "a second writer took the lock") + require.Contains(t, err.Error(), "another writer") case <-time.After(5 * time.Second): - t.Fatalf("the second NewWriter blocked instead of failing immediately") + require.Fail(t, "the second NewWriter blocked instead of failing immediately") } } func TestWriterHeaderRefusesANonEmptyLog(t *testing.T) { path := filepath.Join(t.TempDir(), "log.jsonl") w, _ := newTestWriter(t, path) - if err := w.WriteHeader(testHeader()); err != nil { - t.Fatalf("WriteHeader: %v", err) - } - if err := w.WriteHeader(testHeader()); err == nil { - t.Fatalf("a second header was accepted") - } - if err := w.Close(); err != nil { - t.Fatalf("Close: %v", err) - } + require.NoError(t, w.WriteHeader(testHeader())) + require.Error(t, w.WriteHeader(testHeader()), "a second header was accepted") + require.NoError(t, w.Close()) reopened, _ := newTestWriter(t, path) defer reopened.Close() - if err := reopened.WriteHeader(testHeader()); err == nil { - t.Fatalf("a header was accepted on a non-empty log") - } + require.Error(t, reopened.WriteHeader(testHeader()), "a header was accepted on a non-empty log") } func TestSentinelStopIsIdempotentAndLast(t *testing.T) { path := filepath.Join(t.TempDir(), "log.jsonl") w, _ := newTestWriter(t, path) - if err := w.WriteHeader(testHeader()); err != nil { - t.Fatalf("WriteHeader: %v", err) - } - if err := w.Stop(); err != nil { - t.Fatalf("Stop: %v", err) - } + require.NoError(t, w.WriteHeader(testHeader())) + require.NoError(t, w.Stop()) // watch reaches Stop from both a signal handler and a defer; a second // sentinel would be indistinguishable from a second writer. - if err := w.Stop(); err != nil { - t.Errorf("second Stop: %v", err) - } + require.NoError(t, w.Stop()) // Nothing may be appended after the sentinel — not even by the same writer. - if err := w.WritePoll(Poll{RuleUID: "rule1"}); err == nil { - t.Errorf("WritePoll after Stop was accepted") - } + require.Error(t, w.WritePoll(Poll{RuleUID: "rule1"}), "WritePoll after Stop was accepted") b, err := os.ReadFile(path) - if err != nil { - t.Fatalf("read: %v", err) - } + require.NoError(t, err) lines := strings.Split(strings.TrimSuffix(string(b), "\n"), "\n") - if len(lines) != 2 { - t.Fatalf("got %d lines, want header + one sentinel:\n%s", len(lines), b) - } - if !strings.Contains(lines[1], `"type":"stopped"`) { - t.Errorf("last line is not the sentinel: %s", lines[1]) - } + require.Len(t, lines, 2, "header + one sentinel") + require.Contains(t, lines[1], `"type":"stopped"`) } // Close is the parent's handoff path: a sentinel there would tell check the @@ -638,23 +510,13 @@ func TestSentinelStopIsIdempotentAndLast(t *testing.T) { func TestSentinelCloseWritesNone(t *testing.T) { path := filepath.Join(t.TempDir(), "log.jsonl") w, _ := newTestWriter(t, path) - if err := w.WriteHeader(testHeader()); err != nil { - t.Fatalf("WriteHeader: %v", err) - } - if err := w.Close(); err != nil { - t.Fatalf("Close: %v", err) - } + require.NoError(t, w.WriteHeader(testHeader())) + require.NoError(t, w.Close()) _, polls, sentinel, err := ReadLog(path) - if err != nil { - t.Fatalf("ReadLog: %v", err) - } - if sentinel != nil { - t.Errorf("Close wrote a sentinel: %s", sentinel) - } - if polls != nil { - t.Errorf("polls = %+v, want none", polls) - } + require.NoError(t, err) + require.Nil(t, sentinel, "Close wrote a sentinel") + require.Nil(t, polls) } // An unfinished recording reads cleanly with a nil sentinel — ReadLog reports @@ -664,23 +526,14 @@ func TestSentinelCloseWritesNone(t *testing.T) { func TestReadLogWithoutASentinel(t *testing.T) { path := filepath.Join(t.TempDir(), "log.jsonl") w, _ := newTestWriter(t, path) - if err := w.WriteHeader(testHeader()); err != nil { - t.Fatalf("WriteHeader: %v", err) - } - if err := w.WritePoll(Poll{RuleUID: "rule1", Found: true, GrafanaNow: testNow}); err != nil { - t.Fatalf("WritePoll: %v", err) - } - if err := w.Close(); err != nil { - t.Fatalf("Close: %v", err) - } + require.NoError(t, w.WriteHeader(testHeader())) + require.NoError(t, w.WritePoll(Poll{RuleUID: "rule1", Found: true, GrafanaNow: testNow})) + require.NoError(t, w.Close()) _, polls, sentinel, err := ReadLog(path) - if err != nil { - t.Fatalf("ReadLog: %v", err) - } - if sentinel != nil || len(polls) != 1 { - t.Errorf("got %d polls, sentinel %v; want 1 poll and no sentinel", len(polls), sentinel) - } + require.NoError(t, err) + require.Nil(t, sentinel) + require.Len(t, polls, 1) } // The read rules are deliberately the crudest possible: any unparseable @@ -691,23 +544,17 @@ func TestReadLogRejectsBadLogs(t *testing.T) { h := testHeader() h.SchemaVersion = version b, err := json.Marshal(headerRecord{Type: RecordHeader, Header: h}) - if err != nil { - t.Fatalf("marshal header: %v", err) - } + require.NoError(t, err, "marshal header") return string(b) } poll := func() string { b, err := json.Marshal(pollRecord{Type: RecordPoll, Poll: Poll{RuleUID: "rule1", Found: true, GrafanaNow: testNow}}) - if err != nil { - t.Fatalf("marshal poll: %v", err) - } + require.NoError(t, err, "marshal poll") return string(b) } sentinel := func() string { b, err := json.Marshal(stoppedRecord{Type: RecordStopped, At: testNow}) - if err != nil { - t.Fatalf("marshal sentinel: %v", err) - } + require.NoError(t, err, "marshal sentinel") return string(b) } @@ -758,25 +605,17 @@ func TestReadLogRejectsBadLogs(t *testing.T) { for _, c := range cases { t.Run(c.name, func(t *testing.T) { path := filepath.Join(t.TempDir(), "log.jsonl") - if err := os.WriteFile(path, []byte(c.content), 0o600); err != nil { - t.Fatalf("write: %v", err) - } + require.NoError(t, os.WriteFile(path, []byte(c.content), 0o600)) _, _, _, err := ReadLog(path) - if err == nil { - t.Fatalf("ReadLog: want an error, got nil") - } - if !strings.Contains(err.Error(), c.wantIn) { - t.Errorf("error %q does not contain %q", err, c.wantIn) - } + require.Error(t, err) + require.Contains(t, err.Error(), c.wantIn) }) } } func TestReadLogMissingFile(t *testing.T) { _, _, _, err := ReadLog(filepath.Join(t.TempDir(), "absent.jsonl")) - if err == nil { - t.Fatalf("ReadLog on a missing log: want an error, got nil") - } + require.Error(t, err) } // Per-poll log size must not grow across polls on a high-cardinality @@ -786,69 +625,46 @@ func TestReadLogMissingFile(t *testing.T) { func TestLogSizeIsFlatAcrossPollsOnAHighCardinalityRule(t *testing.T) { body := synthesizeHighCardinalityState(t, 1, 2445) rules, err := ParseState(body) - if err != nil { - t.Fatalf("ParseState: %v", err) - } - if len(rules[0].Instances) != 2446 { - t.Fatalf("got %d instances, want 2446", len(rules[0].Instances)) - } + require.NoError(t, err) + require.Len(t, rules[0].Instances, 2446) uid := rules[0].UID path := filepath.Join(t.TempDir(), "log.jsonl") w, _ := newTestWriter(t, path) - if err := w.WriteHeader(testHeader()); err != nil { - t.Fatalf("WriteHeader: %v", err) - } + require.NoError(t, w.WriteHeader(testHeader())) r := NewReducer() var sizes []int64 // Measure from the end of the header line, so sizes[0] is the first poll // record alone rather than the header plus it. info, err := os.Stat(path) - if err != nil { - t.Fatalf("stat: %v", err) - } + require.NoError(t, err) previous := info.Size() for i := range 5 { p := r.Reduce(uid, observation(testNow.Add(time.Duration(i)*30*time.Second), rules[0])) - if len(p.Abnormal) != 1 { - t.Fatalf("poll %d: Abnormal = %d instances, want the single firing one", i, len(p.Abnormal)) - } - if got := p.Abnormal[0].Labels["instance"]; got != "alerting-0" { - t.Fatalf("poll %d: the firing instance lost its identity: %q", i, got) - } - if err := w.WritePoll(p); err != nil { - t.Fatalf("WritePoll: %v", err) - } + require.Lenf(t, p.Abnormal, 1, "poll %d", i) + require.Equalf(t, "alerting-0", p.Abnormal[0].Labels["instance"], "poll %d: the firing instance lost its identity", i) + require.NoError(t, w.WritePoll(p)) info, err := os.Stat(path) - if err != nil { - t.Fatalf("stat: %v", err) - } + require.NoError(t, err) sizes = append(sizes, info.Size()-previous) previous = info.Size() } for i := 1; i < len(sizes); i++ { - if sizes[i] != sizes[0] { - t.Errorf("per-poll size grew across polls: %v", sizes) - } + require.Equal(t, sizes[0], sizes[i], "per-poll size grew across polls: %v", sizes) } // One firing instance among 2446 costs a few hundred bytes, against the // ~600 KB the unreduced response carries. - if sizes[0] > 2048 { - t.Errorf("per-poll size %d bytes is not a reduction of a %d-byte response", sizes[0], len(body)) - } + require.LessOrEqual(t, sizes[0], int64(2048)) // When the firing instance clears, the record collapses further and the // transition is still attributed. rules[0].Instances[0].State = StateNormal p := r.Reduce(uid, observation(testNow.Add(5*30*time.Second), rules[0])) - if len(p.Cleared) != 1 || len(p.Abnormal) != 0 { - t.Errorf("cleared poll = %+v, want exactly one cleared key and no abnormal instances", p) - } - if err := w.Stop(); err != nil { - t.Fatalf("Stop: %v", err) - } + require.Len(t, p.Cleared, 1) + require.Empty(t, p.Abnormal) + require.NoError(t, w.Stop()) } // The log must stay readable by anything that reads JSONL, one flat object per @@ -857,39 +673,22 @@ func TestLogSizeIsFlatAcrossPollsOnAHighCardinalityRule(t *testing.T) { func TestLogRecordsAreFlatOneLineObjects(t *testing.T) { path := filepath.Join(t.TempDir(), "log.jsonl") w, _ := newTestWriter(t, path) - if err := w.WriteHeader(testHeader()); err != nil { - t.Fatalf("WriteHeader: %v", err) - } - if err := w.WritePoll(Poll{RuleUID: "rule1", Found: true, GrafanaNow: testNow}); err != nil { - t.Fatalf("WritePoll: %v", err) - } - if err := w.Stop(); err != nil { - t.Fatalf("Stop: %v", err) - } + require.NoError(t, w.WriteHeader(testHeader())) + require.NoError(t, w.WritePoll(Poll{RuleUID: "rule1", Found: true, GrafanaNow: testNow})) + require.NoError(t, w.Stop()) b, err := os.ReadFile(path) - if err != nil { - t.Fatalf("read: %v", err) - } + require.NoError(t, err) lines := strings.Split(strings.TrimSuffix(string(b), "\n"), "\n") wantTypes := []RecordType{RecordHeader, RecordPoll, RecordStopped} - if len(lines) != len(wantTypes) { - t.Fatalf("got %d lines, want %d:\n%s", len(lines), len(wantTypes), b) - } + require.Len(t, lines, len(wantTypes)) for i, line := range lines { var m map[string]json.RawMessage - if err := json.Unmarshal([]byte(line), &m); err != nil { - t.Fatalf("line %d is not one JSON object: %v", i+1, err) - } + require.NoErrorf(t, json.Unmarshal([]byte(line), &m), "line %d is not one JSON object", i+1) var gotType RecordType - if err := json.Unmarshal(m["type"], &gotType); err != nil { - t.Fatalf("line %d has no type tag: %v", i+1, err) - } - if gotType != wantTypes[i] { - t.Errorf("line %d type = %q, want %q", i+1, gotType, wantTypes[i]) - } - if _, nested := m["header"]; nested { - t.Errorf("line %d wraps its payload instead of being flat: %s", i+1, line) - } + require.NoErrorf(t, json.Unmarshal(m["type"], &gotType), "line %d has no type tag", i+1) + require.Equalf(t, wantTypes[i], gotType, "line %d type", i+1) + _, nested := m["header"] + require.Falsef(t, nested, "line %d wraps its payload instead of being flat", i+1) } } diff --git a/grafana-alertcheck/internal/gate/parse_ruler.go b/grafana-alertcheck/internal/gate/parse_ruler.go index ae878d4c9..5517fcce2 100644 --- a/grafana-alertcheck/internal/gate/parse_ruler.go +++ b/grafana-alertcheck/internal/gate/parse_ruler.go @@ -84,7 +84,7 @@ func ParseDefinitions(body []byte) ([]Definition, error) { func parseDefinition(raw json.RawMessage, folder, group string) (Definition, error) { var m map[string]json.RawMessage if err := json.Unmarshal(raw, &m); err != nil { - return Definition{}, fmt.Errorf("%w", err) + return Definition{}, err } var forStr string diff --git a/grafana-alertcheck/internal/gate/parse_ruler_test.go b/grafana-alertcheck/internal/gate/parse_ruler_test.go index b225ffc2c..280dc9910 100644 --- a/grafana-alertcheck/internal/gate/parse_ruler_test.go +++ b/grafana-alertcheck/internal/gate/parse_ruler_test.go @@ -3,123 +3,88 @@ package gate import ( "testing" "time" + + "github.com/stretchr/testify/require" ) func TestParseDefinitions_RulerRules(t *testing.T) { defs, err := ParseDefinitions(readFixture(t, "ruler_rules.json")) - if err != nil { - t.Fatalf("ParseDefinitions: unexpected error: %v", err) - } + require.NoError(t, err) byUID := map[string]Definition{} for _, d := range defs { - if d.Kind != KindGrafanaManaged { - t.Errorf("rule %q: Kind = %v, want KindGrafanaManaged", d.UID, d.Kind) - } + require.Equalf(t, KindGrafanaManaged, d.Kind, "rule %q: Kind", d.UID) byUID[d.UID] = d } // The real 2-way duplicate title: same folder, same group, same title, // distinct UIDs — only uid: can tell them apart. a, ok := byUID["rule0000006a"] - if !ok { - t.Fatalf("missing rule0000006a") - } + require.True(t, ok, "missing rule0000006a") b, ok := byUID["rule0000006b"] - if !ok { - t.Fatalf("missing rule0000006b") - } - if a.Title != b.Title || a.Folder != b.Folder || a.Group != b.Group { - t.Errorf("duplicate-title pair should share Title/Folder/Group: a=%+v b=%+v", a, b) - } - if a.UID == b.UID { - t.Errorf("duplicate-title pair should have distinct UIDs") - } + require.True(t, ok, "missing rule0000006b") + require.Equal(t, a.Title, b.Title, "duplicate-title pair should share Title") + require.Equal(t, a.Folder, b.Folder, "duplicate-title pair should share Folder") + require.Equal(t, a.Group, b.Group, "duplicate-title pair should share Group") + require.NotEqual(t, a.UID, b.UID, "duplicate-title pair should have distinct UIDs") // The 3 real paused rules. pausedUIDs := []string{"rule0000002", "rule0000007", "rule0000008"} for _, uid := range pausedUIDs { d, ok := byUID[uid] - if !ok { - t.Fatalf("missing paused rule %q", uid) - } - if !d.IsPaused { - t.Errorf("rule %q: IsPaused = false, want true", uid) - } + require.True(t, ok, "missing paused rule %q", uid) + require.Truef(t, d.IsPaused, "rule %q: IsPaused = false, want true", uid) } // for:1d and the derived for:1w rule. dayRule, ok := byUID["rule0000009"] - if !ok || dayRule.For != 24*time.Hour { - t.Fatalf("rule0000009: For = %v, want 24h (ok=%v)", dayRule.For, ok) - } + require.Truef(t, ok, "missing rule0000009") + require.Equal(t, 24*time.Hour, dayRule.For) weekRule, ok := byUID["rule0000010"] - if !ok || weekRule.For != 7*24*time.Hour { - t.Fatalf("rule0000010: For = %v, want 168h (ok=%v)", weekRule.For, ok) - } + require.Truef(t, ok, "missing rule0000010") + require.Equal(t, 7*24*time.Hour, weekRule.For) // Identity shared with testdata/state_paused.json. shared := byUID["rule0000002"] - if shared.FolderUID != "folder0000002" { - t.Errorf("rule0000002: FolderUID = %q, want folder0000002", shared.FolderUID) - } + require.Equal(t, "folder0000002", shared.FolderUID) } func TestParseDefinitions_DatasourceManaged(t *testing.T) { defs, err := ParseDefinitions(readFixture(t, "ruler_datasource_managed.json")) - if err != nil { - t.Fatalf("ParseDefinitions: unexpected error: %v", err) - } - if len(defs) != 1 { - t.Fatalf("got %d definitions, want 1", len(defs)) - } - if defs[0].Kind != KindDatasourceManaged { - t.Errorf("Kind = %v, want KindDatasourceManaged", defs[0].Kind) - } - if defs[0].For != 5*time.Minute { - t.Errorf("For = %v, want 5m", defs[0].For) - } + require.NoError(t, err) + require.Len(t, defs, 1) + require.Equal(t, KindDatasourceManaged, defs[0].Kind) + require.Equal(t, 5*time.Minute, defs[0].For) // A datasource-managed rule has no uid in this shape; its only identity // is the Prometheus "alert" name — a synthetic UID would be invented // shape, and an empty Title would make Resolve's refusal-by-name // unreachable. - if defs[0].Title != "ExampleTargetDown" { - t.Errorf("Title = %q, want ExampleTargetDown", defs[0].Title) - } - if defs[0].UID != "" { - t.Errorf("UID = %q, want empty (this shape has no uid)", defs[0].UID) - } + require.Equal(t, "ExampleTargetDown", defs[0].Title) + require.Empty(t, defs[0].UID, "this shape has no uid") } func TestParseDefinitions_Recording(t *testing.T) { defs, err := ParseDefinitions(readFixture(t, "ruler_recording.json")) - if err != nil { - t.Fatalf("ParseDefinitions: unexpected error: %v", err) - } - if len(defs) != 1 { - t.Fatalf("got %d definitions, want 1", len(defs)) - } + require.NoError(t, err) + require.Len(t, defs, 1) d := defs[0] - if d.Kind != KindRecording { - t.Errorf("Kind = %v, want KindRecording", d.Kind) - } - if d.UID != "rule0000011" { - t.Errorf("UID = %q, want rule0000011", d.UID) - } + require.Equal(t, KindRecording, d.Kind) + require.Equal(t, "rule0000011", d.UID) // The fixture deliberately omits no_data_state/exec_err_state/is_paused/ // intervalSeconds/namespace_uid — alerting-only concepts a recording // rule may not carry. Requiring them would brick ParseDefinitions for // every named rule in the same response over one recording rule // elsewhere in the fleet; they must come back as zero values, not errors. - if d.NoDataState != "" || d.ExecErrState != "" || d.IsPaused || d.IntervalSeconds != 0 || d.FolderUID != "" { - t.Errorf("expected zero-valued alert-only fields for a recording rule, got %+v", d) - } + require.Empty(t, d.NoDataState) + require.Empty(t, d.ExecErrState) + require.False(t, d.IsPaused) + require.Zero(t, d.IntervalSeconds) + require.Empty(t, d.FolderUID) } // A datasource-managed rule with no alert/record name must fail parsing. func TestParseDefinitions_DatasourceManagedNoName(t *testing.T) { body := []byte(`{"ExampleMetrics":[{"name":"g","rules":[{"expr":"up == 0","for":"5m"}]}]}`) - if _, err := ParseDefinitions(body); err == nil { - t.Fatalf("ParseDefinitions: expected error for datasource-managed rule with no alert/record, got nil") - } + _, err := ParseDefinitions(body) + require.Error(t, err, "a datasource-managed rule with no alert/record must fail") } diff --git a/grafana-alertcheck/internal/gate/parse_state_test.go b/grafana-alertcheck/internal/gate/parse_state_test.go index 4ccca1391..f55c10185 100644 --- a/grafana-alertcheck/internal/gate/parse_state_test.go +++ b/grafana-alertcheck/internal/gate/parse_state_test.go @@ -1,23 +1,20 @@ package gate import ( - "bytes" "encoding/json" "fmt" - "maps" "os" "path/filepath" - "strings" "testing" "time" + + "github.com/stretchr/testify/require" ) func readFixture(t *testing.T, name string) []byte { t.Helper() b, err := os.ReadFile(filepath.Join("testdata", name)) - if err != nil { - t.Fatalf("reading fixture %s: %v", name, err) - } + require.NoErrorf(t, err, "reading fixture %s", name) return b } @@ -33,24 +30,15 @@ func TestParseState_HappyPaths(t *testing.T) { wantRules: 1, wantInstances: 1, checkFirst: func(t *testing.T, r StateRule) { - if r.UID != "rule0000001" { - t.Errorf("UID = %q, want rule0000001", r.UID) - } - if r.Folder != "ExampleTeam" || r.Group != "Example Service - Prod" { - t.Errorf("Folder/Group = %q/%q, want ExampleTeam/Example Service - Prod", r.Folder, r.Group) - } - if r.Health != "ok" || r.State != "inactive" { - t.Errorf("Health/State = %q/%q, want ok/inactive", r.Health, r.State) - } - if r.Interval.Seconds() != 60 { - t.Errorf("Interval = %v, want 60s", r.Interval) - } - if r.IsPaused { - t.Errorf("IsPaused = true, want false") - } - if len(r.Instances) != 1 || r.Instances[0].State != StateNormal { - t.Fatalf("Instances = %+v, want one normal instance", r.Instances) - } + require.Equal(t, "rule0000001", r.UID) + require.Equal(t, "ExampleTeam", r.Folder) + require.Equal(t, "Example Service - Prod", r.Group) + require.Equal(t, "ok", r.Health) + require.Equal(t, "inactive", r.State) + require.Equal(t, float64(60), r.Interval.Seconds()) + require.False(t, r.IsPaused) + require.Len(t, r.Instances, 1) + require.Equal(t, StateNormal, r.Instances[0].State) inst := r.Instances[0] wantLabels := map[string]string{ @@ -62,19 +50,11 @@ func TestParseState_HappyPaths(t *testing.T) { "severity": "critical", "team": "example-team", } - if !maps.Equal(inst.Labels, wantLabels) { - t.Errorf("Labels = %+v, want %+v", inst.Labels, wantLabels) - } + require.Equal(t, wantLabels, inst.Labels) wantActiveAt, err := time.Parse(time.RFC3339, "2026-08-31T08:02:50Z") - if err != nil { - t.Fatalf("test setup: %v", err) - } - if !inst.ActiveAt.Equal(wantActiveAt) { - t.Errorf("ActiveAt = %v, want %v", inst.ActiveAt, wantActiveAt) - } - if inst.Value != "" { - t.Errorf("Value = %q, want empty string", inst.Value) - } + require.NoError(t, err, "test setup") + require.True(t, inst.ActiveAt.Equal(wantActiveAt)) + require.Empty(t, inst.Value) }, }, { @@ -82,15 +62,10 @@ func TestParseState_HappyPaths(t *testing.T) { wantRules: 1, wantInstances: 0, checkFirst: func(t *testing.T, r StateRule) { - if !r.IsPaused { - t.Errorf("IsPaused = false, want true") - } - if !r.LastEvaluation.IsZero() { - t.Errorf("LastEvaluation = %v, want zero time", r.LastEvaluation) - } - if r.Health != "ok" || r.State != "inactive" { - t.Errorf("Health/State = %q/%q, want ok/inactive", r.Health, r.State) - } + require.True(t, r.IsPaused) + require.True(t, r.LastEvaluation.IsZero()) + require.Equal(t, "ok", r.Health) + require.Equal(t, "inactive", r.State) }, }, { @@ -98,15 +73,10 @@ func TestParseState_HappyPaths(t *testing.T) { wantRules: 1, wantInstances: 1, checkFirst: func(t *testing.T, r StateRule) { - if r.Health != "error" { - t.Errorf("Health = %q, want error", r.Health) - } - if r.LastError == "" { - t.Errorf("LastError is empty, want a message") - } - if len(r.Instances) != 1 || r.Instances[0].State != StateError { - t.Fatalf("Instances = %+v, want one error instance", r.Instances) - } + require.Equal(t, "error", r.Health) + require.NotEmpty(t, r.LastError) + require.Len(t, r.Instances, 1) + require.Equal(t, StateError, r.Instances[0].State) }, }, { @@ -114,12 +84,9 @@ func TestParseState_HappyPaths(t *testing.T) { wantRules: 1, wantInstances: 1, checkFirst: func(t *testing.T, r StateRule) { - if r.Health != "nodata" { - t.Errorf("Health = %q, want nodata", r.Health) - } - if len(r.Instances) != 1 || r.Instances[0].State != StateNodata { - t.Fatalf("Instances = %+v, want one nodata instance", r.Instances) - } + require.Equal(t, "nodata", r.Health) + require.Len(t, r.Instances, 1) + require.Equal(t, StateNodata, r.Instances[0].State) }, }, { @@ -132,17 +99,14 @@ func TestParseState_HappyPaths(t *testing.T) { byReason[inst.Reason] = inst } errInst, ok := byReason["Error"] - if !ok || errInst.State != StateNormal { - t.Errorf(`want an instance with State=normal Reason="Error", got %+v`, byReason["Error"]) - } + require.True(t, ok, `want an instance with Reason="Error"`) + require.Equal(t, StateNormal, errInst.State) nodataInst, ok := byReason["NoData"] - if !ok || nodataInst.State != StateNormal { - t.Errorf(`want an instance with State=normal Reason="NoData", got %+v`, byReason["NoData"]) - } + require.True(t, ok, `want an instance with Reason="NoData"`) + require.Equal(t, StateNormal, nodataInst.State) plain, ok := byReason[""] - if !ok || plain.State != StateNormal { - t.Errorf(`want a plain State=normal Reason="" instance, got %+v`, byReason[""]) - } + require.True(t, ok, `want a plain Reason="" instance`) + require.Equal(t, StateNormal, plain.State) }, }, { @@ -150,12 +114,8 @@ func TestParseState_HappyPaths(t *testing.T) { wantRules: 1, wantInstances: 0, checkFirst: func(t *testing.T, r StateRule) { - if r.Instances != nil { - t.Errorf("Instances = %+v, want nil", r.Instances) - } - if r.Totals != nil { - t.Errorf("Totals = %+v, want nil", r.Totals) - } + require.Nil(t, r.Instances) + require.Nil(t, r.Totals) }, }, { @@ -163,12 +123,10 @@ func TestParseState_HappyPaths(t *testing.T) { wantRules: 1, wantInstances: 1, checkFirst: func(t *testing.T, r StateRule) { - if len(r.Instances) != 1 || r.Instances[0].State != StateFiring { - t.Fatalf("Instances = %+v, want one firing instance", r.Instances) - } - if r.Totals["normal"] == 0 { - t.Errorf(`Totals["normal"] = 0, want >0 (the totals/instances mismatch this fixture exists to capture)`) - } + require.Len(t, r.Instances, 1) + require.Equal(t, StateFiring, r.Instances[0].State) + require.NotZero(t, r.Totals["normal"], + "the totals/instances mismatch this fixture exists to capture") }, }, } @@ -176,15 +134,9 @@ func TestParseState_HappyPaths(t *testing.T) { for _, c := range cases { t.Run(c.fixture, func(t *testing.T) { rules, err := ParseState(readFixture(t, c.fixture)) - if err != nil { - t.Fatalf("ParseState(%s): unexpected error: %v", c.fixture, err) - } - if len(rules) != c.wantRules { - t.Fatalf("ParseState(%s): got %d rules, want %d", c.fixture, len(rules), c.wantRules) - } - if got := len(rules[0].Instances); got != c.wantInstances { - t.Fatalf("ParseState(%s): got %d instances, want %d", c.fixture, got, c.wantInstances) - } + require.NoErrorf(t, err, "ParseState(%s)", c.fixture) + require.Lenf(t, rules, c.wantRules, "ParseState(%s)", c.fixture) + require.Lenf(t, rules[0].Instances, c.wantInstances, "ParseState(%s)", c.fixture) if c.checkFirst != nil { c.checkFirst(t, rules[0]) } @@ -214,13 +166,9 @@ func TestParseState_MustError(t *testing.T) { for _, c := range cases { t.Run(c.fixture, func(t *testing.T) { _, err := ParseState(readFixture(t, c.fixture)) - if err == nil { - t.Fatalf("ParseState(%s): expected an error, got none", c.fixture) - } + require.Errorf(t, err, "ParseState(%s): expected an error, got none", c.fixture) for _, want := range c.wantContains { - if !strings.Contains(err.Error(), want) { - t.Errorf("ParseState(%s): error %q does not mention %q", c.fixture, err.Error(), want) - } + require.Containsf(t, err.Error(), want, "ParseState(%s): error", c.fixture) } }) } @@ -248,49 +196,31 @@ func TestParseNormalizeInstanceState(t *testing.T) { for _, c := range cases { state, reason, err := normalizeInstanceState(c.in) if c.wantErr { - if err == nil { - t.Errorf("normalizeInstanceState(%q): expected an error, got none", c.in) - } - continue - } - if err != nil { - t.Errorf("normalizeInstanceState(%q): unexpected error: %v", c.in, err) + require.Errorf(t, err, "normalizeInstanceState(%q)", c.in) continue } - if state != c.wantState || reason != c.wantReason { - t.Errorf("normalizeInstanceState(%q) = (%q, %q), want (%q, %q)", c.in, state, reason, c.wantState, c.wantReason) - } + require.NoErrorf(t, err, "normalizeInstanceState(%q)", c.in) + require.Equalf(t, c.wantState, state, "normalizeInstanceState(%q)", c.in) + require.Equalf(t, c.wantReason, reason, "normalizeInstanceState(%q)", c.in) } } func TestInstanceKey(t *testing.T) { a := instanceKey(map[string]string{"b": "2", "a": "1"}) b := instanceKey(map[string]string{"a": "1", "b": "2"}) - if a != b { - t.Errorf("instanceKey order-independence: %q != %q", a, b) - } - if a != `{"a":"1","b":"2"}` { - t.Errorf("instanceKey = %q, want %q", a, `{"a":"1","b":"2"}`) - } + require.Equal(t, b, a, "instanceKey order-independence") + require.Equal(t, `{"a":"1","b":"2"}`, a) diff := instanceKey(map[string]string{"a": "1", "b": "3"}) - if a == diff { - t.Errorf("instanceKey should differ when a label value differs") - } + require.NotEqual(t, a, diff, "instanceKey should differ when a label value differs") - if instanceKey(nil) != "null" { - t.Errorf("instanceKey(nil) = %q, want \"null\"", instanceKey(nil)) - } + require.Equal(t, "null", instanceKey(nil), "instanceKey(nil) should be \"null\"") } -// Label values may contain "\n" or "="; the JSON encoding must keep them distinct. func TestInstanceKey_NoCollision(t *testing.T) { - if instanceKey(map[string]string{"a": "1\nb=2"}) == instanceKey(map[string]string{"a": "1", "b": "2"}) { - t.Errorf("instanceKey collided for sets {a:1\\nb=2} and {a:1,b:2}") - } - if instanceKey(map[string]string{"a": "1=b"}) == instanceKey(map[string]string{"a": "1", "b": ""}) { - t.Errorf("instanceKey collided for a value containing '='") - } + require.NotEqual(t, instanceKey(map[string]string{"a": "1\nb=2"}), instanceKey(map[string]string{"a": "1", "b": "2"}), "instanceKey should not collide for sets {a:1\\nb=2} and {a:1,b:2}") + + require.NotEqual(t, instanceKey(map[string]string{"a": "1=b"}), instanceKey(map[string]string{"a": "1", "b": ""}), "instanceKey should not collide for sets {a:1=b} and {a:1,b:}") } // minimalStateBody is the smallest legal state response: one group, one @@ -319,12 +249,8 @@ func TestParseState_KeepFiringForIsOptional(t *testing.T) { for _, tc := range tests { t.Run(tc.name, func(t *testing.T) { rules, err := ParseState(minimalStateBody(tc.extra)) - if err != nil { - t.Fatalf("ParseState: %v", err) - } - if len(rules) != 1 { - t.Fatalf("rules = %+v, want one", rules) - } + require.NoError(t, err) + require.Len(t, rules, 1) }) } } @@ -335,15 +261,10 @@ func TestParseState_KeepFiringForIsOptional(t *testing.T) { func TestParseState_InstanceWithoutLabelsParses(t *testing.T) { body := minimalStateBody(`,"alerts":[{"state":"Normal","activeAt":"2026-01-01T00:00:00Z"}]`) rules, err := ParseState(body) - if err != nil { - t.Fatalf("ParseState: %v", err) - } - if len(rules) != 1 || len(rules[0].Instances) != 1 { - t.Fatalf("rules = %+v, want one rule with one instance", rules) - } - if got := rules[0].Instances[0].Labels; len(got) != 0 { - t.Errorf("Instance.Labels = %v, want empty/nil", got) - } + require.NoError(t, err) + require.Len(t, rules, 1) + require.Len(t, rules[0].Instances, 1) + require.Empty(t, rules[0].Instances[0].Labels) } // synthesizeHighCardinalityState builds a state response with a single rule @@ -357,25 +278,15 @@ func synthesizeHighCardinalityState(t *testing.T, alerting, normal int) []byte { base := readFixture(t, "state_one_instance.json") var top map[string]json.RawMessage - if err := json.Unmarshal(base, &top); err != nil { - t.Fatalf("synthesize: %v", err) - } + require.NoError(t, json.Unmarshal(base, &top)) var data map[string]json.RawMessage - if err := json.Unmarshal(top["data"], &data); err != nil { - t.Fatalf("synthesize: %v", err) - } + require.NoError(t, json.Unmarshal(top["data"], &data)) var groups []map[string]json.RawMessage - if err := json.Unmarshal(data["groups"], &groups); err != nil { - t.Fatalf("synthesize: %v", err) - } + require.NoError(t, json.Unmarshal(data["groups"], &groups)) var rules []map[string]json.RawMessage - if err := json.Unmarshal(groups[0]["rules"], &rules); err != nil { - t.Fatalf("synthesize: %v", err) - } + require.NoError(t, json.Unmarshal(groups[0]["rules"], &rules)) var alerts []map[string]json.RawMessage - if err := json.Unmarshal(rules[0]["alerts"], &alerts); err != nil { - t.Fatalf("synthesize: %v", err) - } + require.NoError(t, json.Unmarshal(rules[0]["alerts"], &alerts)) template := alerts[0] newAlerts := make([]map[string]json.RawMessage, 0, alerting+normal) @@ -400,9 +311,7 @@ func synthesizeHighCardinalityState(t *testing.T, alerting, normal int) []byte { top["data"] = mustRaw(t, data) out, err := json.Marshal(top) - if err != nil { - t.Fatalf("synthesize: %v", err) - } + require.NoError(t, err) return out } @@ -419,29 +328,19 @@ func cloneRawMap(m map[string]json.RawMessage) map[string]json.RawMessage { func mustRaw(t *testing.T, v any) json.RawMessage { t.Helper() b, err := json.Marshal(v) - if err != nil { - t.Fatalf("marshal: %v", err) - } + require.NoError(t, err) return json.RawMessage(b) } func TestParseState_HighCardinality(t *testing.T) { body := synthesizeHighCardinalityState(t, 445, 2004) - if !bytes.Contains(body, []byte("alerting-0")) { - t.Fatalf("synthesized body missing expected content") - } + require.Contains(t, string(body), "alerting-0") rules, err := ParseState(body) - if err != nil { - t.Fatalf("ParseState: unexpected error: %v", err) - } - if len(rules) != 1 { - t.Fatalf("got %d rules, want 1", len(rules)) - } + require.NoError(t, err) + require.Len(t, rules, 1) r := rules[0] - if len(r.Instances) != 445+2004 { - t.Fatalf("got %d instances, want %d", len(r.Instances), 445+2004) - } + require.Len(t, r.Instances, 445+2004) var firing, normal int for _, inst := range r.Instances { @@ -451,28 +350,21 @@ func TestParseState_HighCardinality(t *testing.T) { case StateNormal: normal++ default: - t.Fatalf("unexpected instance state %q", inst.State) + require.Fail(t, fmt.Sprintf("unexpected instance state %q", inst.State)) } } - if firing != 445 || normal != 2004 { - t.Fatalf("got firing=%d normal=%d, want firing=445 normal=2004", firing, normal) - } + require.Equal(t, 445, firing) + require.Equal(t, 2004, normal) // Each synthesized instance carries a distinct "instance" label; confirm // Labels actually made it through parsing (not just State) by checking // instanceKey produces one unique key per instance, with no collisions. seen := make(map[string]bool, len(r.Instances)) for _, inst := range r.Instances { - if inst.Labels == nil { - t.Fatalf("instance has nil Labels") - } + require.NotNil(t, inst.Labels) k := instanceKey(inst.Labels) - if seen[k] { - t.Fatalf("duplicate instance key %q", k) - } + require.Falsef(t, seen[k], "duplicate instance key %q", k) seen[k] = true } - if len(seen) != 445+2004 { - t.Fatalf("got %d unique instance keys, want %d", len(seen), 445+2004) - } + require.Len(t, seen, 445+2004) } diff --git a/grafana-alertcheck/internal/gate/resolve_test.go b/grafana-alertcheck/internal/gate/resolve_test.go index d58ec73bd..33214516e 100644 --- a/grafana-alertcheck/internal/gate/resolve_test.go +++ b/grafana-alertcheck/internal/gate/resolve_test.go @@ -2,128 +2,89 @@ package gate import ( "fmt" - "strings" "testing" + + "github.com/stretchr/testify/require" ) func rulerDefs(t *testing.T) []Definition { t.Helper() defs, err := ParseDefinitions(readFixture(t, "ruler_rules.json")) - if err != nil { - t.Fatalf("ParseDefinitions: unexpected error: %v", err) - } + require.NoError(t, err) return defs } func TestResolve_SingleMatch(t *testing.T) { defs := rulerDefs(t) resolved, notes, err := Resolve(defs, []string{"example_workflow_paused_rule"}, "") - if err != nil { - t.Fatalf("Resolve: unexpected error: %v", err) - } - if len(notes) != 0 { - t.Errorf("notes = %v, want none", notes) - } - if len(resolved) != 1 || resolved[0].UID != "rule0000007" { - t.Fatalf("resolved = %+v, want [rule0000007]", resolved) - } + require.NoError(t, err) + require.Empty(t, notes) + require.Len(t, resolved, 1) + require.Equal(t, "rule0000007", resolved[0].UID) } func TestResolve_UIDForm(t *testing.T) { defs := rulerDefs(t) resolved, _, err := Resolve(defs, []string{"uid:rule0000006a"}, "") - if err != nil { - t.Fatalf("Resolve: unexpected error: %v", err) - } - if len(resolved) != 1 || resolved[0].UID != "rule0000006a" { - t.Fatalf("resolved = %+v, want [rule0000006a]", resolved) - } + require.NoError(t, err) + require.Len(t, resolved, 1) + require.Equal(t, "rule0000006a", resolved[0].UID) } func TestResolve_FolderTitleForm(t *testing.T) { defs := rulerDefs(t) resolved, _, err := Resolve(defs, []string{"ExampleFeeds/TEMP - Example depeg alert"}, "") - if err != nil { - t.Fatalf("Resolve: unexpected error: %v", err) - } - if len(resolved) != 1 || resolved[0].UID != "rule0000008" { - t.Fatalf("resolved = %+v, want [rule0000008]", resolved) - } + require.NoError(t, err) + require.Len(t, resolved, 1) + require.Equal(t, "rule0000008", resolved[0].UID) } func TestResolve_FolderGroupTitleForm(t *testing.T) { defs := rulerDefs(t) resolved, _, err := Resolve(defs, []string{"Example-Zone-A/Gateway/Example No Gateways Available"}, "") - if err == nil { - t.Fatalf("Resolve: want ambiguous error (real 2-way collision), got resolved=%+v", resolved) - } - if !strings.Contains(err.Error(), "matches 2 rules") { - t.Fatalf("Resolve: error = %q, want it to report 2 matches", err) - } - if !strings.Contains(err.Error(), "uid:rule0000006a") || !strings.Contains(err.Error(), "uid:rule0000006b") { - t.Fatalf("Resolve: error = %q, want both candidate uids listed", err) - } + require.Error(t, err, "want ambiguous error (real 2-way collision), got resolved=%+v", resolved) + require.Contains(t, err.Error(), "matches 2 rules") + require.Contains(t, err.Error(), "uid:rule0000006a") + require.Contains(t, err.Error(), "uid:rule0000006b") } func TestResolve_TrueCollisionResolvesByUID(t *testing.T) { defs := rulerDefs(t) resolved, _, err := Resolve(defs, []string{"uid:rule0000006a", "uid:rule0000006b"}, "") - if err != nil { - t.Fatalf("Resolve: unexpected error: %v", err) - } - if len(resolved) != 2 { - t.Fatalf("resolved = %+v, want 2 distinct rules", resolved) - } + require.NoError(t, err) + require.Len(t, resolved, 2) } func TestResolve_NoMatch(t *testing.T) { defs := rulerDefs(t) _, _, err := Resolve(defs, []string{"Does Not Exist"}, "") - if err == nil { - t.Fatal("Resolve: want error for unknown name") - } - if !strings.Contains(err.Error(), "no rule matched") || !strings.Contains(err.Error(), "list") { - t.Errorf("Resolve: error = %q, want it to name 'no rule matched' and point at 'list'", err) - } + require.Error(t, err) + require.Contains(t, err.Error(), "no rule matched") + require.Contains(t, err.Error(), "list") } func TestResolve_NoMatchSubstringSuggestion(t *testing.T) { defs := rulerDefs(t) _, _, err := Resolve(defs, []string{"paused rule"}, "") - if err == nil { - t.Fatal("Resolve: want error for unknown name") - } - if !strings.Contains(err.Error(), "did you mean") || !strings.Contains(err.Error(), "Example Paused Rule") { - t.Errorf("Resolve: error = %q, want a case-insensitive substring suggestion", err) - } + require.Error(t, err) + require.Contains(t, err.Error(), "did you mean") + require.Contains(t, err.Error(), "Example Paused Rule") } func TestResolve_RefusesDatasourceManaged(t *testing.T) { defs, err := ParseDefinitions(readFixture(t, "ruler_datasource_managed.json")) - if err != nil { - t.Fatalf("ParseDefinitions: unexpected error: %v", err) - } + require.NoError(t, err) _, _, err = Resolve(defs, []string{"ExampleTargetDown"}, "") - if err == nil { - t.Fatal("Resolve: want refusal for a datasource-managed rule") - } - if !strings.Contains(err.Error(), "datasource-managed") { - t.Errorf("Resolve: error = %q, want it to name the datasource-managed kind", err) - } + require.Error(t, err) + require.Contains(t, err.Error(), "datasource-managed") } func TestResolve_RefusesRecording(t *testing.T) { defs, err := ParseDefinitions(readFixture(t, "ruler_recording.json")) - if err != nil { - t.Fatalf("ParseDefinitions: unexpected error: %v", err) - } + require.NoError(t, err) _, _, err = Resolve(defs, []string{"uid:rule0000011"}, "") - if err == nil { - t.Fatal("Resolve: want refusal for a recording rule") - } - if !strings.Contains(err.Error(), "recording rule") { - t.Errorf("Resolve: error = %q, want it to name the recording kind", err) - } + require.Error(t, err) + require.Contains(t, err.Error(), "recording rule") } func TestResolve_RejectsEmptySegments(t *testing.T) { @@ -132,12 +93,8 @@ func TestResolve_RejectsEmptySegments(t *testing.T) { for _, name := range cases { t.Run(name, func(t *testing.T) { _, _, err := Resolve(defs, []string{name}, "") - if err == nil { - t.Fatalf("Resolve(%q): want error for an empty /-separated segment", name) - } - if !strings.Contains(err.Error(), "empty") { - t.Errorf("Resolve(%q): error = %q, want it to name the empty segment", name, err) - } + require.Error(t, err) + require.Contains(t, err.Error(), "empty") }) } } @@ -148,51 +105,29 @@ func TestResolve_UIDEmptySuffix(t *testing.T) { // — that would report the misleading "datasource-managed rule, not // supported" for what is really a typo'd/empty uid. defs, err := ParseDefinitions(readFixture(t, "ruler_datasource_managed.json")) - if err != nil { - t.Fatalf("ParseDefinitions: unexpected error: %v", err) - } + require.NoError(t, err) _, _, err = Resolve(defs, []string{"uid:"}, "") - if err == nil { - t.Fatal("Resolve: want error for an empty uid: suffix") - } - if !strings.Contains(err.Error(), "no rule has this uid") { - t.Errorf("Resolve: error = %q, want it to say no rule has this uid", err) - } - if strings.Contains(err.Error(), "datasource-managed") { - t.Errorf("Resolve: error = %q, must not misreport this as a datasource-managed refusal", err) - } + require.Error(t, err) + require.Contains(t, err.Error(), "no rule has this uid") + require.NotContains(t, err.Error(), "datasource-managed") } func TestResolve_UnsupportedKindsExcludedFromNoMatchSurfaces(t *testing.T) { dsDefs, err := ParseDefinitions(readFixture(t, "ruler_datasource_managed.json")) - if err != nil { - t.Fatalf("ParseDefinitions(datasource_managed): unexpected error: %v", err) - } + require.NoError(t, err) recDefs, err := ParseDefinitions(readFixture(t, "ruler_recording.json")) - if err != nil { - t.Fatalf("ParseDefinitions(recording): unexpected error: %v", err) - } + require.NoError(t, err) supported := rulerDefs(t) combined := append(append(append([]Definition{}, supported...), dsDefs...), recDefs...) _, _, err = Resolve(combined, []string{"Example"}, "") - if err == nil { - t.Fatal("Resolve: want a no-match error for a name matching no title exactly") - } + require.Error(t, err, "want a no-match error for a name matching no title exactly") wantCount := fmt.Sprintf("(%d rules available", len(supported)) - if !strings.Contains(err.Error(), wantCount) { - t.Errorf("Resolve: error = %q, want the available count scoped to the %d supported rules, not the %d combined", err, len(supported), len(combined)) - } - if strings.Contains(err.Error(), "ExampleTargetDown") { - t.Errorf("Resolve: error = %q, must not suggest the datasource-managed rule", err) - } - if strings.Contains(err.Error(), "example:recorded_metric:rate5m") { - t.Errorf("Resolve: error = %q, must not suggest the recording rule", err) - } - if !strings.Contains(err.Error(), "Example Paused Rule") { - t.Errorf("Resolve: error = %q, want it to still suggest a matching supported rule", err) - } + require.Contains(t, err.Error(), wantCount) + require.NotContains(t, err.Error(), "ExampleTargetDown") + require.NotContains(t, err.Error(), "example:recorded_metric:rate5m") + require.Contains(t, err.Error(), "Example Paused Rule") } func TestResolve_UnsupportedHomonymResolvesSupportedSilently(t *testing.T) { @@ -205,12 +140,9 @@ func TestResolve_UnsupportedHomonymResolvesSupportedSilently(t *testing.T) { {UID: "", Folder: "F", Group: "G", Title: "Shared Title", Kind: KindDatasourceManaged}, } resolved, _, err := Resolve(defs, []string{"F/G/Shared Title"}, "") - if err != nil { - t.Fatalf("Resolve: unexpected error: %v", err) - } - if len(resolved) != 1 || resolved[0].UID != "supported-1" { - t.Fatalf("resolved = %+v, want the supported rule alone, no ambiguity", resolved) - } + require.NoError(t, err) + require.Len(t, resolved, 1) + require.Equal(t, "supported-1", resolved[0].UID) } func TestResolve_CollapseByUIDGivesNoteNotError(t *testing.T) { @@ -221,15 +153,10 @@ func TestResolve_CollapseByUIDGivesNoteNotError(t *testing.T) { "example_workflow_paused_rule", "ExampleObservability/Example Auth Production/example_workflow_paused_rule", }, "") - if err != nil { - t.Fatalf("Resolve: unexpected error: %v", err) - } - if len(resolved) != 1 || resolved[0].UID != "rule0000007" { - t.Fatalf("resolved = %+v, want exactly one rule0000007", resolved) - } - if len(notes) != 1 { - t.Fatalf("notes = %v, want exactly one collapse note", notes) - } + require.NoError(t, err) + require.Len(t, resolved, 1) + require.Equal(t, "rule0000007", resolved[0].UID) + require.Len(t, notes, 1) } // The same rule named twice with the identical string must collapse to one @@ -240,15 +167,10 @@ func TestResolve_IdenticalDuplicateNameCollapsesWithNote(t *testing.T) { "example_workflow_paused_rule", "example_workflow_paused_rule", }, "") - if err != nil { - t.Fatalf("Resolve: unexpected error: %v", err) - } - if len(resolved) != 1 || resolved[0].UID != "rule0000007" { - t.Fatalf("resolved = %+v, want exactly one rule0000007", resolved) - } - if len(notes) != 1 { - t.Fatalf("notes = %v, want exactly one collapse note", notes) - } + require.NoError(t, err) + require.Len(t, resolved, 1) + require.Equal(t, "rule0000007", resolved[0].UID) + require.Len(t, notes, 1) } func TestResolve_MinObservedCountIsPostCollapse(t *testing.T) { @@ -259,44 +181,30 @@ func TestResolve_MinObservedCountIsPostCollapse(t *testing.T) { "Example Paused Rule", } resolved, notes, err := Resolve(defs, names, "") - if err != nil { - t.Fatalf("Resolve: unexpected error: %v", err) - } + require.NoError(t, err) // The default MinObserved must come from len(resolved) (2 distinct // rules) — never len(names) (3 input lines), which would be unsatisfiable. - if len(resolved) != 2 { - t.Fatalf("resolved = %+v, want 2 distinct rules after collapse", resolved) - } - if len(notes) != 1 { - t.Fatalf("notes = %v, want exactly one collapse note", notes) - } + require.Len(t, resolved, 2) + require.Len(t, notes, 1) } func TestResolve_EmptyAndBlankLinesDiscarded(t *testing.T) { defs := rulerDefs(t) resolved, _, err := Resolve(defs, []string{"", " ", "example_workflow_paused_rule", " \t "}, "") - if err != nil { - t.Fatalf("Resolve: unexpected error: %v", err) - } - if len(resolved) != 1 || resolved[0].UID != "rule0000007" { - t.Fatalf("resolved = %+v, want [rule0000007]", resolved) - } + require.NoError(t, err) + require.Len(t, resolved, 1) + require.Equal(t, "rule0000007", resolved[0].UID) } func TestResolve_FolderScopesBareTitle(t *testing.T) { defs := rulerDefs(t) // Bare title, scoped to the wrong folder — must not match. _, _, err := Resolve(defs, []string{"example_workflow_paused_rule"}, "Example-Zone-A") - if err == nil { - t.Fatal("Resolve: want no-match when folder scope excludes the only candidate") - } + require.Error(t, err, "want no-match when folder scope excludes the only candidate") // Scoped to the right folder — must match. resolved, _, err := Resolve(defs, []string{"example_workflow_paused_rule"}, "ExampleObservability") - if err != nil { - t.Fatalf("Resolve: unexpected error: %v", err) - } - if len(resolved) != 1 || resolved[0].UID != "rule0000007" { - t.Fatalf("resolved = %+v, want [rule0000007]", resolved) - } + require.NoError(t, err) + require.Len(t, resolved, 1) + require.Equal(t, "rule0000007", resolved[0].UID) } diff --git a/grafana-alertcheck/internal/gate/schedule.go b/grafana-alertcheck/internal/gate/schedule.go index f2b9bd759..c2fe6e1c5 100644 --- a/grafana-alertcheck/internal/gate/schedule.go +++ b/grafana-alertcheck/internal/gate/schedule.go @@ -339,6 +339,16 @@ func CheckBudget(t map[string]ruleTimings, measured map[string]time.Duration, co return fmt.Errorf("%s", b.String()) } +// graceSourceOrNone is the single "none" default for the grace-source field: +// an empty source means no rule contributed a transitionGrace. Applied here so +// StartupSummary and the human table print the same thing. +func graceSourceOrNone(source string) string { + if source == "" { + return "none" + } + return source +} + // StartupSummary formats the pre-run print an operator sees before the wait: // the total planned run time and the rule (with its `for` value) that set // transitionGrace, plus a warning when the grace eats more than @@ -347,17 +357,14 @@ func CheckBudget(t map[string]ruleTimings, measured map[string]time.Duration, co func StartupSummary(from, to time.Time, global globalTimings) (summary, warning string) { window := to.Sub(from) total := window + global.transitionGrace + global.drainTimeout - source := global.graceSource - if source == "" { - source = "none" - } + source := graceSourceOrNone(global.graceSource) summary = fmt.Sprintf( - "planned run time: %s (window %s + transitionGrace %s [source: %s] + drainTimeout %s)", - total, window, global.transitionGrace, source, global.drainTimeout) + "planned run time: %s\n window %s + transitionGrace %s + drainTimeout %s\n transitionGrace source: %s", + total, window, global.transitionGrace, global.drainTimeout, source) if window > 0 && float64(global.transitionGrace) > float64(window)*graceWarnFraction { warning = fmt.Sprintf( - "transitionGrace %s is more than %.0f%% of the window %s (source: %s) — the window may be too short for this alert's `for`", + "transitionGrace %s is more than %.0f%% of the window %s — the window may be too short for this alert's `for`\n source: %s", global.transitionGrace, graceWarnFraction*100, window, source) } return summary, warning diff --git a/grafana-alertcheck/internal/gate/schedule_test.go b/grafana-alertcheck/internal/gate/schedule_test.go index 70cc8817c..ace598650 100644 --- a/grafana-alertcheck/internal/gate/schedule_test.go +++ b/grafana-alertcheck/internal/gate/schedule_test.go @@ -1,9 +1,10 @@ package gate import ( - "strings" "testing" "time" + + "github.com/stretchr/testify/require" ) func TestDeriveTimings_Default(t *testing.T) { @@ -11,22 +12,12 @@ func TestDeriveTimings_Default(t *testing.T) { {UID: "r1", Title: "R1", IntervalSeconds: 60}, } rules, _, notes := DeriveTimings(defs, 0) - if len(notes) != 0 { - t.Fatalf("notes = %v, want none", notes) - } + require.Empty(t, notes) rt := rules["r1"] - if rt.pollEvery != 30*time.Second { - t.Errorf("pollEvery = %s, want 30s", rt.pollEvery) - } - if rt.maxGap != 60*time.Second { - t.Errorf("maxGap = %s, want 60s", rt.maxGap) - } - if rt.healthGrace != 60*time.Second { - t.Errorf("healthGrace = %s, want 60s", rt.healthGrace) - } - if rt.evalStaleAfter != 120*time.Second { - t.Errorf("evalStaleAfter = %s, want 120s", rt.evalStaleAfter) - } + require.Equal(t, 30*time.Second, rt.pollEvery) + require.Equal(t, 60*time.Second, rt.maxGap) + require.Equal(t, 60*time.Second, rt.healthGrace) + require.Equal(t, 120*time.Second, rt.evalStaleAfter) } func TestDeriveTimings_OverrideVerbatimNoClamp(t *testing.T) { @@ -35,23 +26,16 @@ func TestDeriveTimings_OverrideVerbatimNoClamp(t *testing.T) { } rules, _, notes := DeriveTimings(defs, 20*time.Second) rt := rules["r1"] - if rt.pollEvery != 20*time.Second { - t.Fatalf("pollEvery = %s, want the override verbatim (20s), never clamped down to the 5s default", rt.pollEvery) - } - if rt.maxGap != 40*time.Second { - t.Errorf("maxGap = %s, want 2x the override (40s)", rt.maxGap) - } - if len(notes) != 1 || !strings.Contains(notes[0], "R1") { - t.Fatalf("notes = %v, want one note naming R1's exceeded default", notes) - } + require.Equal(t, 20*time.Second, rt.pollEvery, "the override verbatim (20s), never clamped down to the 5s default") + require.Equal(t, 40*time.Second, rt.maxGap) + require.Len(t, notes, 1) + require.Contains(t, notes[0], "R1") } func TestDeriveTimings_OverrideBelowDefaultNoNote(t *testing.T) { defs := []Definition{{UID: "r1", Title: "R1", IntervalSeconds: 60}} // default pollEvery = 30s _, _, notes := DeriveTimings(defs, 5*time.Second) - if len(notes) != 0 { - t.Fatalf("notes = %v, want none when the override tightens rather than exceeds the default", notes) - } + require.Empty(t, notes) } func TestDeriveTimings_TransitionGraceExcludesSkippedRule(t *testing.T) { @@ -61,12 +45,8 @@ func TestDeriveTimings_TransitionGraceExcludesSkippedRule(t *testing.T) { } _, global, _ := DeriveTimings(defs, 0) want := time.Minute + 60*time.Second // r1's for+interval; r2 (skipped) must not win despite its huge `for` - if global.transitionGrace != want { - t.Fatalf("transitionGrace = %s, want %s (paused rule r2 must be excluded from the max)", global.transitionGrace, want) - } - if !strings.Contains(global.graceSource, "Tight") { - t.Errorf("graceSource = %q, want it to name the contributing rule Tight", global.graceSource) - } + require.Equal(t, want, global.transitionGrace) + require.Contains(t, global.graceSource, "Tight") } // `for: 1d` and `for: 1w` parse correctly (parse_ruler_test.go), @@ -79,25 +59,17 @@ func TestDeriveTimings_TransitionGraceExcludesSkippedRule(t *testing.T) { func TestDeriveTimings_RealForOneWeekRuleSetsTransitionGrace(t *testing.T) { defs := rulerDefs(t) _, global, notes := DeriveTimings(defs, 0) - if len(notes) != 0 { - t.Fatalf("notes = %v, want none: no --poll-interval override is given, so no override note should fire", notes) - } + require.Empty(t, notes, "no --poll-interval override is given, so no override note should fire") want := 7*24*time.Hour + 60*time.Second // rule0000010: for=1w, intervalSeconds=60 - if global.transitionGrace != want { - t.Fatalf("transitionGrace = %s, want %s (rule0000010's for:1w plus its interval)", global.transitionGrace, want) - } - if !strings.Contains(global.graceSource, "Example Failure Ratio Above 10 Percent Weekly") { - t.Errorf("graceSource = %q, want it to name rule0000010", global.graceSource) - } + require.Equal(t, want, global.transitionGrace) + require.Contains(t, global.graceSource, "Example Failure Ratio Above 10 Percent Weekly") } func TestDeriveTimings_TransitionGraceZeroWhenAllSkipped(t *testing.T) { defs := []Definition{{UID: "r1", Title: "R1", IntervalSeconds: 60, For: time.Hour, IsPaused: true}} _, global, _ := DeriveTimings(defs, 0) - if global.transitionGrace != 0 { - t.Fatalf("transitionGrace = %s, want 0 when every rule is skipped", global.transitionGrace) - } + require.Zero(t, global.transitionGrace) } // TestDeriveTimingsFromLog_TransitionGraceFollowsTheHeaderNotTheDefinition @@ -123,28 +95,18 @@ func TestDeriveTimingsFromLog_TransitionGraceFollowsTheHeaderNotTheDefinition(t t.Run("header says active: the rule stays in the max", func(t *testing.T) { h := Header{Rules: []LoggedRule{loggedRule("r1", false)}} _, global, err := DeriveTimingsFromLog(h, defs) - if err != nil { - t.Fatalf("DeriveTimingsFromLog: %v", err) - } - if global.transitionGrace != want { - t.Fatalf("transitionGrace = %s, want %s: the rule was active when the recording opened, "+ - "so a pause applied afterwards must not shrink the window", global.transitionGrace, want) - } - if !strings.Contains(global.graceSource, "R1") { - t.Errorf("graceSource = %q, want it to name R1", global.graceSource) - } + require.NoError(t, err) + require.Equal(t, want, global.transitionGrace, + "the rule was active when the recording opened, so a pause applied afterwards must not shrink the window") + require.Contains(t, global.graceSource, "R1") }) t.Run("header says paused: the rule stays out", func(t *testing.T) { h := Header{Rules: []LoggedRule{loggedRule("r1", true)}} _, global, err := DeriveTimingsFromLog(h, defs) - if err != nil { - t.Fatalf("DeriveTimingsFromLog: %v", err) - } - if global.transitionGrace != 0 { - t.Fatalf("transitionGrace = %s, want 0: a rule paused before the window opened can never fire during it", - global.transitionGrace) - } + require.NoError(t, err) + require.Zero(t, global.transitionGrace, + "a rule paused before the window opened can never fire during it") }) t.Run("drainTimeout counts every rule either way", func(t *testing.T) { @@ -153,13 +115,8 @@ func TestDeriveTimingsFromLog_TransitionGraceFollowsTheHeaderNotTheDefinition(t for _, pausedAtStart := range []bool{false, true} { h := Header{Rules: []LoggedRule{loggedRule("r1", pausedAtStart)}} _, global, err := DeriveTimingsFromLog(h, defs) - if err != nil { - t.Fatalf("DeriveTimingsFromLog: %v", err) - } - if global.drainTimeout != minDrainTimeout { - t.Fatalf("drainTimeout = %s with pausedAtStart=%v, want the %s floor", - global.drainTimeout, pausedAtStart, minDrainTimeout) - } + require.NoError(t, err) + require.Equalf(t, minDrainTimeout, global.drainTimeout, "pausedAtStart=%v", pausedAtStart) } }) } @@ -167,9 +124,7 @@ func TestDeriveTimingsFromLog_TransitionGraceFollowsTheHeaderNotTheDefinition(t func TestDeriveTimings_DrainTimeoutIncludesPaused(t *testing.T) { defs := []Definition{{UID: "r1", Title: "R1", IntervalSeconds: 10}} _, global, _ := DeriveTimings(defs, 0) - if global.drainTimeout != minDrainTimeout { - t.Fatalf("drainTimeout = %s, want the %s floor", global.drainTimeout, minDrainTimeout) - } + require.Equal(t, minDrainTimeout, global.drainTimeout) } func TestDeriveTimings_DrainTimeoutFloor(t *testing.T) { @@ -179,17 +134,13 @@ func TestDeriveTimings_DrainTimeoutFloor(t *testing.T) { } _, global, _ := DeriveTimings(defs, 0) // double the longest interval (2 * 180s) should be the drain timeout - if global.drainTimeout != 2*180*time.Second { - t.Fatalf("drainTimeout = %s, want %s", global.drainTimeout, 2*180*time.Second) - } + require.Equal(t, 2*180*time.Second, global.drainTimeout) } func TestDeriveTimings_DrainTimeoutAboveFloor(t *testing.T) { defs := []Definition{{UID: "r1", Title: "R1", IntervalSeconds: 300}} // 2x300s = 600s > 2m floor _, global, _ := DeriveTimings(defs, 0) - if global.drainTimeout != 600*time.Second { - t.Fatalf("drainTimeout = %s, want 600s", global.drainTimeout) - } + require.Equal(t, 600*time.Second, global.drainTimeout) } // The ordering invariant the burst bound depends on: when several rules become @@ -210,9 +161,8 @@ func TestScheduler_DueOrderingTiesBreakByTightestCadence(t *testing.T) { }, } due := s.Due(now) - if len(due) != 4 || due[0] != "tight" { - t.Fatalf("Due = %v, want the tightest-cadence rule (tight) first when all are simultaneously due", due) - } + require.Len(t, due, 4) + require.Equal(t, "tight", due[0], "the tightest-cadence rule must be first when all are simultaneously due") } func TestScheduler_DueExcludesNotYetDue(t *testing.T) { @@ -222,9 +172,8 @@ func TestScheduler_DueExcludesNotYetDue(t *testing.T) { every: map[string]time.Duration{"soon": 10 * time.Second, "later": 10 * time.Second}, } due := s.Due(now) - if len(due) != 1 || due[0] != "soon" { - t.Fatalf("Due = %v, want only [soon]", due) - } + require.Len(t, due, 1) + require.Equal(t, "soon", due[0]) } func TestScheduler_MarkAdvancesNextDue(t *testing.T) { @@ -233,15 +182,9 @@ func TestScheduler_MarkAdvancesNextDue(t *testing.T) { next: map[string]time.Time{"r1": now}, every: map[string]time.Duration{"r1": 30 * time.Second}, } - if err := s.Mark("r1", now); err != nil { - t.Fatalf("Mark: unexpected error: %v", err) - } - if got := s.Due(now); len(got) != 0 { - t.Fatalf("Due right after Mark = %v, want none (next due is 30s out)", got) - } - if got := s.Due(now.Add(30 * time.Second)); len(got) != 1 { - t.Fatalf("Due at next-due time = %v, want [r1]", got) - } + require.NoError(t, s.Mark("r1", now), "Mark must succeed for a known rule") + require.Empty(t, s.Due(now), "next due is 30s out") + require.Len(t, s.Due(now.Add(30*time.Second)), 1) } func TestScheduler_MarkUnknownUIDFails(t *testing.T) { @@ -250,13 +193,11 @@ func TestScheduler_MarkUnknownUIDFails(t *testing.T) { next: map[string]time.Time{"r1": now}, every: map[string]time.Duration{"r1": 30 * time.Second}, } - if err := s.Mark("not-a-rule", now); err == nil { - t.Fatalf("Mark of an unknown uid: want error, got nil (a missing cadence must not read as zero and loop)") - } + err := s.Mark("not-a-rule", now) + require.Error(t, err, "a missing cadence must not read as zero and loop") // The failed Mark must not have inserted a bogus next-due entry. - if _, ok := s.next["not-a-rule"]; ok { - t.Errorf("Mark of an unknown uid inserted a next-due entry") - } + _, ok := s.next["not-a-rule"] + require.False(t, ok, "a failed Mark must not insert a next-due entry") } // TestScheduler_PerRuleCadenceOverTime simulates a run and counts how often @@ -279,24 +220,19 @@ func TestScheduler_PerRuleCadenceOverTime(t *testing.T) { now := start.Add(elapsed) for _, uid := range s.Due(now) { counts[uid]++ - if err := s.Mark(uid, now); err != nil { - t.Fatalf("Mark(%q): unexpected error: %v", uid, err) - } + require.NoErrorf(t, s.Mark(uid, now), "Mark(%q)", uid) } } // 900s of runtime: "tight" (10s cadence) polls ~90 times, "slack" (300s // cadence) ~3 times. Assert the ratio holds rather than an exact count, // since the staggered initial offset shifts each by up to one cadence. - if counts["tight"] < 85 || counts["tight"] > 91 { - t.Errorf("tight polled %d times over 900s, want ~90 (its own 10s cadence)", counts["tight"]) - } - if counts["slack"] < 2 || counts["slack"] > 4 { - t.Errorf("slack polled %d times over 900s, want ~3 (its own 300s cadence, not tight's)", counts["slack"]) - } - if counts["slack"] >= counts["tight"] { - t.Fatalf("slack polled as often as tight (%d vs %d) — schedules must be per rule, not a shared global cycle", counts["slack"], counts["tight"]) - } + require.GreaterOrEqual(t, counts["tight"], 85) + require.LessOrEqual(t, counts["tight"], 91) + require.GreaterOrEqual(t, counts["slack"], 2) + require.LessOrEqual(t, counts["slack"], 4) + require.Less(t, counts["slack"], counts["tight"], + "schedules must be per rule, not a shared global cycle") } func TestNewScheduler_StaggersWithinPollEvery(t *testing.T) { @@ -304,16 +240,14 @@ func TestNewScheduler_StaggersWithinPollEvery(t *testing.T) { rules := map[string]time.Duration{"r1": 100 * time.Second} s := NewScheduler(rules, now) offset := s.next["r1"].Sub(now) - if offset < 0 || offset >= 100*time.Second { - t.Fatalf("initial offset = %s, want within [0, 100s)", offset) - } + require.GreaterOrEqual(t, offset, time.Duration(0)) + require.Less(t, offset, 100*time.Second) } func TestScheduler_EarliestDueEmpty(t *testing.T) { s := &Scheduler{next: map[string]time.Time{}, every: map[string]time.Duration{}} - if _, ok := s.earliestDue(); ok { - t.Fatalf("earliestDue on an empty scheduler = ok=true, want false") - } + _, ok := s.earliestDue() + require.False(t, ok) } // A zero next-due time is real, not an empty scheduler. @@ -323,12 +257,8 @@ func TestScheduler_EarliestDueZeroTime(t *testing.T) { every: map[string]time.Duration{"r1": time.Second}, } earliest, ok := s.earliestDue() - if !ok { - t.Fatalf("earliestDue = ok=false, want true (the zero time is a real next-due, not an empty scheduler)") - } - if !earliest.IsZero() { - t.Errorf("earliestDue = %v, want the zero time", earliest) - } + require.True(t, ok, "the zero time is a real next-due, not an empty scheduler") + require.Truef(t, earliest.IsZero(), "earliestDue = %v, want the zero time", earliest) } func TestScheduler_EarliestDuePicksMinimum(t *testing.T) { @@ -341,12 +271,8 @@ func TestScheduler_EarliestDuePicksMinimum(t *testing.T) { every: map[string]time.Duration{"later": time.Minute, "soon": time.Minute}, } earliest, ok := s.earliestDue() - if !ok { - t.Fatalf("earliestDue = ok=false, want true") - } - if !earliest.Equal(now.Add(time.Minute)) { - t.Errorf("earliestDue = %v, want the earliest next-due time", earliest) - } + require.True(t, ok) + require.Truef(t, earliest.Equal(now.Add(time.Minute)), "earliestDue = %v, want the earliest next-due time", earliest) } // One rule at 10s beside twenty at 300s, all measured ~1.8s, must not error at @@ -359,9 +285,7 @@ func TestCheckBudget_MixedIntervalRegression(t *testing.T) { timings[uid] = ruleTimings{pollEvery: 150 * time.Second} measured[uid] = 1800 * time.Millisecond } - if err := CheckBudget(timings, measured, 1); err != nil { - t.Fatalf("CheckBudget = %v, want nil (utilization 0.6, burst bound 1.8s <= 5s)", err) - } + require.NoError(t, CheckBudget(timings, measured, 1)) } func TestCheckBudget_UtilizationExceeded(t *testing.T) { @@ -371,9 +295,7 @@ func TestCheckBudget_UtilizationExceeded(t *testing.T) { } measured := map[string]time.Duration{"a": 9 * time.Second, "b": 9 * time.Second} err := CheckBudget(timings, measured, 1) - if err == nil { - t.Fatal("CheckBudget = nil, want an error: utilization 1.8 > concurrency 1") - } + require.Error(t, err) assertBudgetMessage(t, err.Error()) } @@ -381,9 +303,7 @@ func TestCheckBudget_SingleRuleExceedsOwnCadence(t *testing.T) { timings := map[string]ruleTimings{"slow": {pollEvery: 5 * time.Second}} measured := map[string]time.Duration{"slow": 6 * time.Second} err := CheckBudget(timings, measured, 10) - if err == nil { - t.Fatal("CheckBudget = nil, want an error: measured 6s exceeds its own 5s poll-interval") - } + require.Error(t, err, "measured 6s exceeds its own 5s poll-interval") assertBudgetMessage(t, err.Error()) } @@ -397,12 +317,8 @@ func TestCheckBudget_BurstBoundViolation(t *testing.T) { } measured := map[string]time.Duration{"tight": 100 * time.Millisecond, "slow": 3 * time.Second} err := CheckBudget(timings, measured, 10) - if err == nil { - t.Fatal("CheckBudget = nil, want a burst-bound error: slow's 3s measured exceeds tight's 2s cadence") - } - if !strings.Contains(err.Error(), "burst bound") { - t.Errorf("error = %q, want it to name the burst bound", err.Error()) - } + require.Error(t, err, "slow's 3s measured exceeds tight's 2s cadence") + require.Contains(t, err.Error(), "burst bound") assertBudgetMessage(t, err.Error()) } @@ -412,17 +328,13 @@ func TestCheckBudget_BurstBoundOKWhenNotExceeded(t *testing.T) { "slow": {pollEvery: 100 * time.Second}, } measured := map[string]time.Duration{"tight": 100 * time.Millisecond, "slow": 1800 * time.Millisecond} - if err := CheckBudget(timings, measured, 10); err != nil { - t.Fatalf("CheckBudget = %v, want nil (1.8s <= 5s tightest cadence)", err) - } + require.NoError(t, CheckBudget(timings, measured, 10)) } func TestCheckBudget_MissingMeasurementIsAnError(t *testing.T) { timings := map[string]ruleTimings{"r1": {pollEvery: 30 * time.Second}} err := CheckBudget(timings, map[string]time.Duration{}, 10) - if err == nil { - t.Fatal("CheckBudget = nil, want an error: r1 was never measured (fail closed, not a silent zero)") - } + require.Error(t, err, "r1 was never measured (fail closed, not a silent zero)") } func TestCheckBudget_MissingMixedMeasurementIsAnError(t *testing.T) { @@ -431,15 +343,12 @@ func TestCheckBudget_MissingMixedMeasurementIsAnError(t *testing.T) { "slow": {pollEvery: 100 * time.Second}, } measured := map[string]time.Duration{"tight": 100 * time.Millisecond} - if err := CheckBudget(timings, measured, 10); err == nil { - t.Fatal("CheckBudget = nil, want an error: slow was never measured (fail closed, not a silent zero)") - } + require.Error(t, CheckBudget(timings, measured, 10), + "slow was never measured (fail closed, not a silent zero)") } func TestCheckBudget_EmptyScheduleIsFine(t *testing.T) { - if err := CheckBudget(nil, nil, 1); err != nil { - t.Fatalf("CheckBudget = %v, want nil for an empty schedule", err) - } + require.NoError(t, CheckBudget(nil, nil, 1)) } func TestCheckBudget_NonPositivePollIntervalIsAnError(t *testing.T) { @@ -447,12 +356,8 @@ func TestCheckBudget_NonPositivePollIntervalIsAnError(t *testing.T) { timings := map[string]ruleTimings{"r1": {pollEvery: pe}} measured := map[string]time.Duration{"r1": time.Second} err := CheckBudget(timings, measured, 1) - if err == nil { - t.Fatalf("CheckBudget(pollEvery=%s) = nil, want error (non-positive poll-interval would divide by zero)", pe) - } - if !strings.Contains(err.Error(), "non-positive") { - t.Errorf("error %q does not name the non-positive poll-interval", err.Error()) - } + require.Errorf(t, err, "pollEvery=%s would divide by zero", pe) + require.Contains(t, err.Error(), "non-positive", "the error must name the non-positive poll-interval") } } @@ -462,9 +367,7 @@ func TestCheckBudget_NonPositivePollIntervalIsAnError(t *testing.T) { func assertBudgetMessage(t *testing.T, msg string) { t.Helper() for _, want := range []string{"measured", "concurrency", "poll-interval", "fewer"} { - if !strings.Contains(msg, want) { - t.Errorf("message %q missing %q", msg, want) - } + require.Contains(t, msg, want) } } @@ -477,15 +380,9 @@ func TestStartupSummary_WarningWhenGraceTooLarge(t *testing.T) { to := from.Add(10 * time.Minute) global := globalTimings{transitionGrace: 5 * time.Minute, graceSource: "R (for=4m30s, interval=30s)", drainTimeout: time.Minute} summary, warning := StartupSummary(from, to, global) - if !strings.Contains(summary, "planned run time") { - t.Errorf("summary = %q, want it to name the planned run time", summary) - } - if warning == "" { - t.Fatal("warning = \"\", want one: transitionGrace (5m) > 1/4 of the 10m window") - } - if !strings.Contains(warning, "R (for=4m30s, interval=30s)") { - t.Errorf("warning = %q, want it to name the grace source", warning) - } + require.Contains(t, summary, "planned run time") + require.NotEmpty(t, warning, "transitionGrace (5m) > 1/4 of the 10m window") + require.Contains(t, warning, "R (for=4m30s, interval=30s)") } // The test above pins the warning formula with a hand-built globalTimings. @@ -495,22 +392,14 @@ func TestStartupSummary_WarningWhenGraceTooLarge(t *testing.T) { func TestStartupSummary_RealForOneWeekRuleTriggersWarning(t *testing.T) { defs := rulerDefs(t) _, global, notes := DeriveTimings(defs, 0) - if len(notes) != 0 { - t.Fatalf("notes = %v, want none: no --poll-interval override is given, so no override note should fire", notes) - } + require.Empty(t, notes) from := time.Date(2026, 1, 1, 0, 0, 0, 0, time.UTC) to := from.Add(10 * time.Minute) // transitionGrace (>1w) dwarfs 1/4 of this window summary, warning := StartupSummary(from, to, global) - if !strings.Contains(summary, "planned run time") { - t.Errorf("summary = %q, want it to name the planned run time", summary) - } - if warning == "" { - t.Fatal("warning = \"\", want one: a real for:1w rule's transitionGrace vastly exceeds 1/4 of a 10m window") - } - if !strings.Contains(warning, "Example Failure Ratio Above 10 Percent Weekly") { - t.Errorf("warning = %q, want it to name rule0000010", warning) - } + require.Contains(t, summary, "planned run time") + require.NotEmpty(t, warning) + require.Contains(t, warning, "Example Failure Ratio Above 10 Percent Weekly") } func TestStartupSummary_NoWarningWhenGraceSmall(t *testing.T) { @@ -518,16 +407,12 @@ func TestStartupSummary_NoWarningWhenGraceSmall(t *testing.T) { to := from.Add(time.Hour) global := globalTimings{transitionGrace: time.Minute, graceSource: "R (for=30s, interval=30s)", drainTimeout: time.Minute} _, warning := StartupSummary(from, to, global) - if warning != "" { - t.Fatalf("warning = %q, want none: 1m grace is well under 1/4 of a 1h window", warning) - } + require.Empty(t, warning) } func TestStartupSummary_NoGraceSourceReadsNone(t *testing.T) { from := time.Date(2026, 1, 1, 0, 0, 0, 0, time.UTC) to := from.Add(time.Hour) summary, _ := StartupSummary(from, to, globalTimings{}) - if !strings.Contains(summary, "none") { - t.Fatalf("summary = %q, want it to read \"none\" when no rule set the grace", summary) - } + require.Contains(t, summary, "none") } diff --git a/grafana-alertcheck/internal/gate/source.go b/grafana-alertcheck/internal/gate/source.go index bebb1839e..721151e21 100644 --- a/grafana-alertcheck/internal/gate/source.go +++ b/grafana-alertcheck/internal/gate/source.go @@ -44,9 +44,10 @@ type Observation struct { Latency time.Duration // t_send through the full body read — see requestResult.Latency } -// TransportError marks a failure worth retrying: a non-2xx response, a network -// failure, or a body that failed to parse. Not a deleted rule (an authoritative -// 2xx) and not a clock problem (a hard error — see doRequest). +// TransportError marks a failure worth retrying: a 5xx/429 response, a network +// failure, or a body that failed to parse. Not a 4xx (wrong auth, missing +// resource), not a deleted rule (an authoritative 2xx) and not a clock problem +// (a hard error — see doRequest). type TransportError struct { Err error Status int // 0 when the failure never got a status (network/transport failure) @@ -111,17 +112,7 @@ func parseGrafanaVersion(s string) (grafanaVersion, error) { var v grafanaVersion fields := [3]*int{&v.major, &v.minor, &v.patch} for i, field := range fields { - // Trim any trailing non-digit suffix (prerelease/build metadata, e.g. - // "0+security") rather than requiring an exact numeric match. - digits := parts[i] - j := 0 - for j < len(digits) && digits[j] >= '0' && digits[j] <= '9' { - j++ - } - if j == 0 { - return grafanaVersion{}, fmt.Errorf("unparseable version %q", s) - } - n, err := strconv.Atoi(digits[:j]) + n, err := strconv.Atoi(parts[i]) if err != nil { return grafanaVersion{}, fmt.Errorf("unparseable version %q: %w", s, err) } @@ -259,10 +250,11 @@ type requestResult struct { Latency time.Duration } -// doRequest performs one HTTP GET and classifies the outcome: network failure, -// non-2xx, or body-read failure is retryable (*TransportError); a missing or -// unparseable Date header or a skew beyond SkewHardLimit is a hard error — -// retrying can never fix either, so neither enters the backoff loop. +// doRequest performs one HTTP GET and classifies the outcome: a 5xx/429, a +// network failure, or a body-read failure is retryable (*TransportError); a +// 4xx (wrong auth, missing resource — retrying cannot fix it), a missing or +// unparseable Date header, or a skew beyond SkewHardLimit is a hard error, so +// none of those enters the backoff loop. // // The Date/skew check runs on every endpoint (even /api/health): a skew only // noticed once RuleState starts polling has already masked earlier reads, so it @@ -298,7 +290,11 @@ func (s *httpSource) doRequest(ctx context.Context, path string) (requestResult, latency := tBodyRead.Sub(tSend) if resp.StatusCode < 200 || resp.StatusCode >= 300 { - return requestResult{}, &TransportError{Err: fmt.Errorf("unexpected status %d", resp.StatusCode), Status: resp.StatusCode} + err := fmt.Errorf("unexpected status %d", resp.StatusCode) + if resp.StatusCode >= 400 && resp.StatusCode < 500 && resp.StatusCode != http.StatusTooManyRequests { + return requestResult{}, err + } + return requestResult{}, &TransportError{Err: err, Status: resp.StatusCode} } dateHeader := resp.Header.Get("Date") diff --git a/grafana-alertcheck/internal/gate/source_test.go b/grafana-alertcheck/internal/gate/source_test.go index 39181cfca..e5f24f46f 100644 --- a/grafana-alertcheck/internal/gate/source_test.go +++ b/grafana-alertcheck/internal/gate/source_test.go @@ -7,11 +7,12 @@ import ( "net/http" "net/http/httptest" "net/url" - "strings" "sync" "sync/atomic" "testing" "time" + + "github.com/stretchr/testify/require" ) func healthBody(version string) string { @@ -79,20 +80,18 @@ func TestCheckGrafanaVersion(t *testing.T) { {"", true, nil}, } for _, c := range cases { - err := CheckGrafanaVersion(c.version) - if c.wantErr && err == nil { - t.Errorf("CheckGrafanaVersion(%q): want error, got nil", c.version) - continue - } - if !c.wantErr && err != nil { - t.Errorf("CheckGrafanaVersion(%q): unexpected error: %v", c.version, err) - continue - } - for _, want := range c.wantContains { - if !strings.Contains(err.Error(), want) { - t.Errorf("CheckGrafanaVersion(%q): error %q does not mention %q — it must name both what was found and what is supported", c.version, err.Error(), want) + t.Run(c.version, func(t *testing.T) { + err := CheckGrafanaVersion(c.version) + if c.wantErr { + require.Errorf(t, err, "CheckGrafanaVersion(%q)", c.version) + } else { + require.NoErrorf(t, err, "CheckGrafanaVersion(%q)", c.version) } - } + for _, want := range c.wantContains { + require.Contains(t, err.Error(), want, + "must name both what was found and what is supported") + } + }) } } @@ -102,20 +101,14 @@ func TestBackoffDelay(t *testing.T) { maxWithJitter := maxDelay + maxDelay/5 + time.Millisecond for n := 1; n <= 10; n++ { d := backoffDelay(base, maxDelay, n) - if d <= 0 { - t.Fatalf("backoffDelay(_, _, %d) = %v, want > 0", n, d) - } - if d > maxWithJitter { - t.Fatalf("backoffDelay(_, _, %d) = %v, want <= ~%v", n, d, maxWithJitter) - } + require.Positive(t, d) + require.LessOrEqual(t, d, maxWithJitter) } } func TestHTTPSource_Version_HappyPath(t *testing.T) { srv := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) { - if r.URL.Path != "/api/health" { - t.Errorf("path = %q, want /api/health", r.URL.Path) - } + require.Equal(t, "/api/health", r.URL.Path) w.Header().Set("Content-Type", "application/json") _, _ = w.Write([]byte(healthBody("13.1.0"))) })) @@ -124,20 +117,14 @@ func TestHTTPSource_Version_HappyPath(t *testing.T) { clock := newFakeClock(time.Now()) src := NewHTTPSource(srv.URL, "", clock) v, err := src.Version(context.Background()) - if err != nil { - t.Fatalf("Version(): unexpected error: %v", err) - } - if v != "13.1.0" { - t.Fatalf("Version() = %q, want 13.1.0", v) - } + require.NoError(t, err) + require.Equal(t, "13.1.0", v) } func TestHTTPSource_Version_NeverLogsToken(t *testing.T) { const secret = "super-secret-token" srv := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) { - if got := r.Header.Get("Authorization"); got != "Bearer "+secret { - t.Errorf("Authorization = %q, want Bearer %s", got, secret) - } + require.Equal(t, "Bearer "+secret, r.Header.Get("Authorization")) w.WriteHeader(http.StatusInternalServerError) })) defer srv.Close() @@ -145,12 +132,8 @@ func TestHTTPSource_Version_NeverLogsToken(t *testing.T) { clock := newFakeClock(time.Now()) src := NewHTTPSource(srv.URL, secret, clock) _, err := src.Version(context.Background()) - if err == nil { - t.Fatalf("Version(): want error, got nil") - } - if strings.Contains(err.Error(), secret) { - t.Fatalf("error %q leaks the token", err.Error()) - } + require.Error(t, err) + require.NotContains(t, err.Error(), secret) } func TestHTTPSource_RuleState_EmptyIsNotAnError(t *testing.T) { @@ -163,24 +146,16 @@ func TestHTTPSource_RuleState_EmptyIsNotAnError(t *testing.T) { clock := newFakeClock(time.Now()) src := NewHTTPSource(srv.URL, "", clock) obs, err := src.RuleState(context.Background(), "Anything") - if err != nil { - t.Fatalf("RuleState(): unexpected error: %v", err) - } - if len(obs.Rules) != 0 { - t.Fatalf("Rules = %+v, want empty (an authoritative 2xx is not a transport error)", obs.Rules) - } - if obs.GrafanaNow.IsZero() { - t.Fatalf("GrafanaNow is zero, want the response's Date header value") - } + require.NoError(t, err) + require.Empty(t, obs.Rules, "an authoritative 2xx is not a transport error") + require.False(t, obs.GrafanaNow.IsZero(), "want the response's Date header value") } func TestHTTPSource_RuleState_EscapesRuleName(t *testing.T) { var gotQuery string srv := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) { gotQuery = r.URL.RawQuery - if r.URL.Path != "/api/prometheus/grafana/api/v1/rules" { - t.Errorf("path = %q, want /api/prometheus/grafana/api/v1/rules", r.URL.Path) - } + require.Equal(t, "/api/prometheus/grafana/api/v1/rules", r.URL.Path) w.Header().Set("Content-Type", "application/json") _, _ = w.Write([]byte(emptyStateBody())) })) @@ -189,24 +164,16 @@ func TestHTTPSource_RuleState_EscapesRuleName(t *testing.T) { clock := newFakeClock(time.Now()) src := NewHTTPSource(srv.URL, "", clock) title := "[JD] No Job Proposals & More" - if _, err := src.RuleState(context.Background(), title); err != nil { - t.Fatalf("RuleState(): unexpected error: %v", err) - } - want := "rule_name=" + url.QueryEscape(title) - if gotQuery != want { - t.Fatalf("query = %q, want %q", gotQuery, want) - } + _, err := src.RuleState(context.Background(), title) + require.NoError(t, err) + require.Equal(t, "rule_name="+url.QueryEscape(title), gotQuery) } func TestHTTPSource_Definitions_HappyPath(t *testing.T) { body := readFixture(t, "ruler_rules.json") srv := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) { - if r.URL.Path != "/api/ruler/grafana/api/v1/rules" { - t.Errorf("path = %q, want /api/ruler/grafana/api/v1/rules", r.URL.Path) - } - if r.URL.RawQuery != "" { - t.Errorf("query = %q, want none — Definitions reads the ruler API unfiltered", r.URL.RawQuery) - } + require.Equal(t, "/api/ruler/grafana/api/v1/rules", r.URL.Path) + require.Empty(t, r.URL.RawQuery, "Definitions reads the ruler API unfiltered") w.Header().Set("Content-Type", "application/json") _, _ = w.Write(body) })) @@ -215,12 +182,8 @@ func TestHTTPSource_Definitions_HappyPath(t *testing.T) { clock := newFakeClock(time.Now()) src := NewHTTPSource(srv.URL, "", clock) defs, err := src.Definitions(context.Background()) - if err != nil { - t.Fatalf("Definitions(): unexpected error: %v", err) - } - if len(defs) == 0 { - t.Fatalf("Definitions(): got 0 definitions from a fixture known to have some") - } + require.NoError(t, err) + require.NotEmpty(t, defs) } func TestHTTPSource_Skew(t *testing.T) { @@ -249,14 +212,13 @@ func TestHTTPSource_Skew(t *testing.T) { }) src := NewHTTPSource(srv.URL, "", clock) _, err := src.Version(context.Background()) - if c.wantErr && err == nil { - t.Fatalf("Version(): want error, got nil") - } - if !c.wantErr && err != nil { - t.Fatalf("Version(): unexpected error: %v", err) + if c.wantErr { + require.Error(t, err) + } else { + require.NoError(t, err) } - if c.wantErr && calls.Load() != 1 { - t.Fatalf("calls = %d, want 1 — a skew hard error must never be retried", calls.Load()) + if c.wantErr { + require.Equal(t, int32(1), calls.Load(), "a skew hard error must never be retried") } }) } @@ -273,12 +235,8 @@ func TestHTTPSource_MissingDateHeader(t *testing.T) { clock := newFakeClock(time.Now()) src := NewHTTPSource(srv.URL, "", clock) _, err := src.Version(context.Background()) - if err == nil { - t.Fatalf("Version(): want error, got nil: a missing Date header is a hard error") - } - if calls.Load() != 1 { - t.Fatalf("calls = %d, want 1 — a missing Date header must never be retried", calls.Load()) - } + require.Error(t, err, "a missing Date header is a hard error") + require.Equal(t, int32(1), calls.Load(), "a missing Date header must never be retried") } func TestHTTPSource_UnparseableDateHeader(t *testing.T) { @@ -293,12 +251,8 @@ func TestHTTPSource_UnparseableDateHeader(t *testing.T) { clock := newFakeClock(time.Now()) src := NewHTTPSource(srv.URL, "", clock) _, err := src.Version(context.Background()) - if err == nil { - t.Fatalf("Version(): want error, got nil: an unparseable Date header is a hard error") - } - if calls.Load() != 1 { - t.Fatalf("calls = %d, want 1 — an unparseable Date header must never be retried", calls.Load()) - } + require.Error(t, err, "an unparseable Date header is a hard error") + require.Equal(t, int32(1), calls.Load(), "an unparseable Date header must never be retried") } // TestHTTPSource_ObservationTiming pins the arithmetic behind Observation's @@ -333,18 +287,10 @@ func TestHTTPSource_ObservationTiming(t *testing.T) { }) src := NewHTTPSource(srv.URL, "", clock) obs, err := src.RuleState(context.Background(), "Anything") - if err != nil { - t.Fatalf("RuleState(): unexpected error: %v", err) - } - if obs.Skew != c.drift { - t.Errorf("Skew = %v, want %v", obs.Skew, c.drift) - } - if obs.SkewBound != time.Second { - t.Errorf("SkewBound = %v, want 1s (RTT/2 with a 2s round trip to headers)", obs.SkewBound) - } - if obs.Latency != 4*time.Second { - t.Errorf("Latency = %v, want 4s (send through full body read) — not just the 2s header round trip", obs.Latency) - } + require.NoError(t, err) + require.Equal(t, c.drift, obs.Skew) + require.Equal(t, time.Second, obs.SkewBound, "RTT/2 with a 2s round trip to headers") + require.Equal(t, 4*time.Second, obs.Latency, "send through full body read — not just the 2s header round trip") }) } } @@ -378,15 +324,10 @@ func TestHTTPSourceStalenessNeverFalsePositiveUnderSkew(t *testing.T) { src := NewHTTPSource(srv.URL, "", clock) obs, err := src.RuleState(context.Background(), def.Title) - if err != nil { - t.Fatalf("RuleState(): %v", err) - } - if !obs.GrafanaNow.Equal(serverDate) { - t.Fatalf("GrafanaNow = %s, want the Date header %s, never the runner's clock %s", obs.GrafanaNow, serverDate, runnerNow) - } - if len(obs.Rules) != 1 { - t.Fatalf("Rules = %+v, want exactly one", obs.Rules) - } + require.NoError(t, err) + require.True(t, obs.GrafanaNow.Equal(serverDate), + "want the Date header, never the runner's clock") + require.Len(t, obs.Rules, 1) rt := newRuleTimings(30*time.Second, 60) // evalStaleAfter = 120s from := serverDate.Add(-10 * time.Minute) @@ -396,10 +337,8 @@ func TestHTTPSourceStalenessNeverFalsePositiveUnderSkew(t *testing.T) { sentinel := to res := proveCoverage(Header{StartedAt: from.Add(-time.Hour)}, polls, &sentinel, rt, def, from, to, 0) - if res.Unobservable { - t.Fatalf("Coverage = %+v, want no violation: 100s behind Grafana's TRUE now is under the 120s limit — "+ - "only a runner-clock leak (skewed +30s here) would push this over", res) - } + require.False(t, res.Unobservable, + "100s behind Grafana's TRUE now is under the 120s limit — only a runner-clock leak (skewed +30s here) would push this over") } func TestHTTPSource_Retry_TransientRecovers(t *testing.T) { @@ -422,18 +361,12 @@ func TestHTTPSource_Retry_TransientRecovers(t *testing.T) { clock := newFakeClock(time.Now()) src := NewHTTPSource(srv.URL, "", clock) v, err := src.Version(context.Background()) - if err != nil { - t.Fatalf("Version(): unexpected error after a transient failure: %v", err) - } - if v != "13.1.0" { - t.Fatalf("Version() = %q, want 13.1.0", v) - } + require.NoError(t, err) + require.Equal(t, "13.1.0", v) mu.Lock() n := calls mu.Unlock() - if n != 3 { - t.Fatalf("calls = %d, want 3 (2 failures + 1 success)", n) - } + require.Equal(t, 3, n) } func TestHTTPSource_Retry_ExceedsLimit(t *testing.T) { @@ -447,12 +380,9 @@ func TestHTTPSource_Retry_ExceedsLimit(t *testing.T) { clock := newFakeClock(time.Now()) src := NewHTTPSource(srv.URL, "", clock) _, err := src.Version(context.Background()) - if err == nil { - t.Fatalf("Version(): want error, got nil") - } - if n := calls.Load(); n != 6 { - t.Fatalf("calls = %d, want 6 (maxSequentialFailures=5 tolerates 5, gives up on the 6th)", n) - } + require.Error(t, err) + require.Equal(t, int32(6), calls.Load(), + "maxSequentialFailures=5 tolerates 5, gives up on the 6th") assertRetryExhausted(t, err, 6) } @@ -479,18 +409,12 @@ func TestHTTPSource_RuleState_GarbageBodyRetries(t *testing.T) { clock := newFakeClock(time.Now()) src := NewHTTPSource(srv.URL, "", clock) obs, err := src.RuleState(context.Background(), "Anything") - if err != nil { - t.Fatalf("RuleState(): unexpected error after a transient garbage body: %v", err) - } - if len(obs.Rules) != 0 { - t.Fatalf("Rules = %+v, want empty", obs.Rules) - } + require.NoError(t, err) + require.Empty(t, obs.Rules) mu.Lock() n := calls mu.Unlock() - if n != 3 { - t.Fatalf("calls = %d, want 3 (2 unparseable bodies + 1 valid one)", n) - } + require.Equal(t, 3, n) } func TestHTTPSource_Definitions_GarbageBodyGivesUp(t *testing.T) { @@ -505,12 +429,9 @@ func TestHTTPSource_Definitions_GarbageBodyGivesUp(t *testing.T) { clock := newFakeClock(time.Now()) src := NewHTTPSource(srv.URL, "", clock) _, err := src.Definitions(context.Background()) - if err == nil { - t.Fatalf("Definitions(): want error, got nil") - } - if n := calls.Load(); n != 6 { - t.Fatalf("calls = %d, want 6 — a persistently unparseable 2xx body retries like any other transport failure", n) - } + require.Error(t, err) + require.Equal(t, int32(6), calls.Load(), + "a persistently unparseable 2xx body retries like any other transport failure") assertRetryExhausted(t, err, 6) } @@ -534,17 +455,11 @@ func TestHTTPSource_ResponseBodyTooLarge(t *testing.T) { clock := newFakeClock(time.Now()) src := NewHTTPSource(srv.URL, "", clock) _, err := src.Version(context.Background()) - if err == nil { - t.Fatalf("Version(): want error, got nil (an oversized body must fail loudly)") - } - if !strings.Contains(err.Error(), "exceeded") { - t.Fatalf("error %q does not name the size limit", err.Error()) - } + require.Error(t, err, "an oversized body must fail loudly") + require.Contains(t, err.Error(), "exceeded", "the error must name the size limit") // An oversized body is a stable condition, not a transient one: it must // fail hard on the first attempt, never burning retries re-reading it. - if n := calls.Load(); n != 1 { - t.Fatalf("calls = %d, want 1 — an oversized body must never be retried", n) - } + require.Equal(t, int32(1), calls.Load(), "an oversized body must never be retried") } func TestHTTPSource_NetworkFailureRetries(t *testing.T) { @@ -554,9 +469,7 @@ func TestHTTPSource_NetworkFailureRetries(t *testing.T) { clock := newFakeClock(time.Now()) src := NewHTTPSource("http://127.0.0.1:1", "", clock) _, err := src.Version(context.Background()) - if err == nil { - t.Fatalf("Version(): want error, got nil") - } + require.Error(t, err) assertRetryExhausted(t, err, 6) } @@ -568,39 +481,26 @@ func TestHTTPSource_NetworkFailureRetries(t *testing.T) { func assertRetryExhausted(t *testing.T, err error, wantFailures int) { t.Helper() var reErr *RetryExhaustedError - if !errors.As(err, &reErr) { - t.Fatalf("error %v (%T): want a *RetryExhaustedError", err, err) - } - if reErr.Failures != wantFailures { - t.Errorf("RetryExhaustedError.Failures = %d, want %d", reErr.Failures, wantFailures) - } - if !strings.Contains(err.Error(), fmt.Sprintf("gave up after %d", wantFailures)) { - t.Errorf("error %q does not name the failure count", err.Error()) - } - if _, ok := errors.AsType[*TransportError](err); ok { - t.Fatalf("error %v (%T) is classified as *TransportError — an exhausted retry must be a terminal, non-retryable error", err, err) - } + require.ErrorAs(t, err, &reErr) + require.Equal(t, wantFailures, reErr.Failures) + require.Contains(t, err.Error(), fmt.Sprintf("gave up after %d", wantFailures)) + _, ok := errors.AsType[*TransportError](err) + require.False(t, ok, "an exhausted retry must be a terminal, non-retryable error") } func TestFakeClock(t *testing.T) { start := time.Date(2026, 1, 1, 0, 0, 0, 0, time.UTC) c := newFakeClock(start) - if !c.Now().Equal(start) { - t.Fatalf("Now() = %v, want %v", c.Now(), start) - } + require.True(t, c.Now().Equal(start)) c.Advance(5 * time.Minute) want := start.Add(5 * time.Minute) - if !c.Now().Equal(want) { - t.Fatalf("Now() after Advance = %v, want %v", c.Now(), want) - } + require.True(t, c.Now().Equal(want)) select { case fired := <-c.After(time.Hour): - if !fired.Equal(want.Add(time.Hour)) { - t.Fatalf("After fired with %v, want %v", fired, want.Add(time.Hour)) - } + require.True(t, fired.Equal(want.Add(time.Hour))) default: - t.Fatalf("After(1h) did not fire immediately") + require.Fail(t, "After(1h) did not fire immediately") } } @@ -610,37 +510,29 @@ func TestFakeSource(t *testing.T) { f.defs = []Definition{{UID: "u1", Title: "Rule One"}} ctx := context.Background() - if v, err := f.Version(ctx); err != nil || v != "13.1.0" { - t.Fatalf("Version() = (%q, %v), want (13.1.0, nil)", v, err) - } - if defs, err := f.Definitions(ctx); err != nil || len(defs) != 1 { - t.Fatalf("Definitions() = (%v, %v), want one definition", defs, err) - } + v, err := f.Version(ctx) + require.NoError(t, err) + require.Equal(t, "13.1.0", v) + defs, err := f.Definitions(ctx) + require.NoError(t, err) + require.Len(t, defs, 1) f.script("Rule One", Observation{Rules: []StateRule{{UID: "u1"}}}, nil) f.script("Rule One", Observation{}, fmt.Errorf("boom")) f.script("Rule One", Observation{Rules: nil}, nil) obs, err := f.RuleState(ctx, "Rule One") - if err != nil || len(obs.Rules) != 1 { - t.Fatalf("RuleState() call 1 = (%v, %v), want one rule, no error", obs, err) - } - if _, err := f.RuleState(ctx, "Rule One"); err == nil { - t.Fatalf("RuleState() call 2: want the scripted error, got nil") - } + require.NoError(t, err) + require.Len(t, obs.Rules, 1) + _, err = f.RuleState(ctx, "Rule One") + require.Error(t, err, "RuleState() call 2: want the scripted error, got nil") obs, err = f.RuleState(ctx, "Rule One") - if err != nil { - t.Fatalf("RuleState() call 3: unexpected error: %v", err) - } - if obs.Rules != nil { - t.Fatalf("RuleState() call 3: Rules = %v, want nil (last script entry, then repeats)", obs.Rules) - } + require.NoError(t, err) + require.Nil(t, obs.Rules, "last script entry, then repeats") obs, err = f.RuleState(ctx, "Rule One") - if err != nil || obs.Rules != nil { - t.Fatalf("RuleState() call 4: want the last scripted entry to repeat, got (%v, %v)", obs, err) - } + require.NoError(t, err) + require.Nil(t, obs.Rules) - if _, err := f.RuleState(ctx, "Unscripted Rule"); err == nil { - t.Fatalf("RuleState() for an unscripted title: want an error, got nil") - } + _, err = f.RuleState(ctx, "Unscripted Rule") + require.Error(t, err) } diff --git a/grafana-alertcheck/internal/gate/watch_daemon_test.go b/grafana-alertcheck/internal/gate/watch_daemon_test.go index c3a91cd12..73a6f3d7a 100644 --- a/grafana-alertcheck/internal/gate/watch_daemon_test.go +++ b/grafana-alertcheck/internal/gate/watch_daemon_test.go @@ -15,6 +15,8 @@ import ( "syscall" "testing" "time" + + "github.com/stretchr/testify/require" ) // TestMain doubles this test binary as the detached recorder. Watch spawns @@ -163,37 +165,25 @@ func grafanaTestServer(t *testing.T) *httptest.Server { func patchedStateBody(t *testing.T) []byte { t.Helper() var body map[string]any - if err := json.Unmarshal(readFixture(t, "state_one_instance.json"), &body); err != nil { - t.Fatalf("unmarshal state fixture: %v", err) - } + require.NoError(t, json.Unmarshal(readFixture(t, "state_one_instance.json"), &body)) data, ok := body["data"].(map[string]any) - if !ok { - t.Fatal("state fixture: no data object") - } + require.True(t, ok, "state fixture: no data object") groups, ok := data["groups"].([]any) - if !ok || len(groups) == 0 { - t.Fatal("state fixture: no groups") - } + require.True(t, ok, "state fixture: no groups") + require.NotEmpty(t, groups, "state fixture: no groups") group, ok := groups[0].(map[string]any) - if !ok { - t.Fatal("state fixture: group 0 is not an object") - } + require.True(t, ok, "state fixture: group 0 is not an object") rules, ok := group["rules"].([]any) - if !ok || len(rules) == 0 { - t.Fatal("state fixture: group 0 has no rules") - } + require.True(t, ok, "state fixture: group 0 has no rules") + require.NotEmpty(t, rules, "state fixture: group 0 has no rules") rule, ok := rules[0].(map[string]any) - if !ok { - t.Fatal("state fixture: rule 0 is not an object") - } + require.True(t, ok, "state fixture: rule 0 is not an object") rule["uid"] = watchActiveUID rule["name"] = watchActiveTitle rule["lastEvaluation"] = time.Now().UTC().Format(time.RFC3339Nano) b, err := json.Marshal(body) - if err != nil { - t.Fatalf("marshal patched state fixture: %v", err) - } + require.NoError(t, err) return b } @@ -209,7 +199,7 @@ func waitFor(t *testing.T, what string, timeout time.Duration, cond func() bool) } time.Sleep(20 * time.Millisecond) } - t.Fatalf("timed out after %s waiting for %s", timeout, what) + require.Fail(t, fmt.Sprintf("timed out after %s waiting for %s", timeout, what)) } // The one watch integration test: everything from the version gate to the @@ -237,9 +227,7 @@ func TestWatchSpawnsADetachedRecorder(t *testing.T) { Notes: ¬es, } - if err := Watch(context.Background(), cfg); err != nil { - t.Fatalf("Watch: %v\nnotes:\n%s", err, notes.String()) - } + require.NoError(t, Watch(context.Background(), cfg)) t.Cleanup(func() { if t.Failed() { t.Logf("notes:\n%s", notes.String()) @@ -248,20 +236,14 @@ func TestWatchSpawnsADetachedRecorder(t *testing.T) { }) pid, err := ReadPidFile(out + ".pid") - if err != nil { - t.Fatalf("ReadPidFile: %v", err) - } - if err := syscall.Kill(pid, 0); err != nil { - t.Fatalf("recorder pid %d is not running right after Watch returned: %v", pid, err) - } + require.NoError(t, err) + require.NoError(t, syscall.Kill(pid, 0), "recorder pid %d is not running right after Watch returned", pid) // Setsid, not a bare `&`: a session leader's process group id is its own // pid. Without this the child would still share the parent's process group // and die with the step that started it. - if pgid, err := syscall.Getpgid(pid); err != nil { - t.Errorf("Getpgid(%d): %v", pid, err) - } else if pgid != pid { - t.Errorf("recorder pgid = %d, want %d: it did not get its own session", pgid, pid) - } + pgid, err := syscall.Getpgid(pid) + require.NoError(t, err) + require.Equal(t, pid, pgid, "it did not get its own session") // The parent already wrote the first heartbeat before it returned; // these later ones prove the detached child is the one appending now. @@ -271,35 +253,24 @@ func TestWatchSpawnsADetachedRecorder(t *testing.T) { }) // Stop it exactly the way check does. - if err := syscall.Kill(pid, syscall.SIGTERM); err != nil { - t.Fatalf("SIGTERM %d: %v", pid, err) - } + require.NoError(t, syscall.Kill(pid, syscall.SIGTERM)) waitFor(t, "the stopped sentinel", 10*time.Second, func() bool { _, _, sentinel, err := ReadLog(out) return err == nil && sentinel != nil }) header, polls, sentinel, err := ReadLog(out) - if err != nil { - t.Fatalf("ReadLog: %v", err) - } - if header.URL != srv.URL || header.GrafanaVersion != "13.1.0" { - t.Errorf("header identity = %q/%q, want %q/13.1.0", header.URL, header.GrafanaVersion, srv.URL) - } - if len(header.Rules) != 1 || header.Rules[0].PollEverySeconds != 0.2 { - t.Errorf("header rules = %+v, want one rule recorded at 0.2s", header.Rules) - } + require.NoError(t, err) + require.Equal(t, srv.URL, header.URL) + require.Equal(t, "13.1.0", header.GrafanaVersion) + require.Len(t, header.Rules, 1) + require.Equal(t, float64(0.2), header.Rules[0].PollEverySeconds) for i, p := range polls { - if p.RuleUID != watchActiveUID || !p.Found { - t.Fatalf("poll %d = %+v, want a found observation of %s", i, p, watchActiveUID) - } - if p.GrafanaNow.IsZero() { - t.Fatalf("poll %d has no grafana_now; every poll needs the Date header of its own response", i) - } - } - if sentinel.Before(header.StartedAt) { - t.Errorf("sentinel at %s precedes the record start %s", sentinel, header.StartedAt) + require.Equalf(t, watchActiveUID, p.RuleUID, "poll %d", i) + require.Truef(t, p.Found, "poll %d", i) + require.Falsef(t, p.GrafanaNow.IsZero(), "poll %d has no grafana_now; every poll needs the Date header of its own response", i) } + require.False(t, sentinel.Before(header.StartedAt), "sentinel precedes the record start") waitFor(t, "the recorder to exit", 10*time.Second, func() bool { return syscall.Kill(pid, 0) != nil @@ -317,27 +288,17 @@ func TestDaemonChildRejectsAnAlreadyFinishedLog(t *testing.T) { clock := newFakeClock(testNow) w, err := NewWriter(path, clock) - if err != nil { - t.Fatalf("NewWriter: %v", err) - } - if err := w.WriteHeader(testHeader()); err != nil { - t.Fatalf("WriteHeader: %v", err) - } - if err := w.Stop(); err != nil { - t.Fatalf("Stop: %v", err) - } + require.NoError(t, err) + require.NoError(t, w.WriteHeader(testHeader())) + require.NoError(t, w.Stop()) err = RunDaemonChild(context.Background(), DaemonChildConfig{ URL: testHeader().URL, Out: path, Clock: clock, }) - if err == nil { - t.Fatal("RunDaemonChild: no error against a log that already carries a stopped sentinel") - } - if !strings.Contains(err.Error(), "sentinel") { - t.Errorf("error = %v, want it to name the stopped sentinel", err) - } + require.Error(t, err, "no error against a log that already carries a stopped sentinel") + require.Contains(t, err.Error(), "sentinel") } // TestWatchFailsWhenTheChildCannotStartRecording is the other half of the @@ -365,13 +326,9 @@ func TestWatchFailsWhenTheChildCannotStartRecording(t *testing.T) { Concurrency: 2, Notes: ¬es, }) - if err == nil { - t.Fatal("Watch: no error, but the child could never have started recording") - } - if !strings.Contains(err.Error(), "records url") { - t.Errorf("error does not quote the child's own reason:\n%v", err) - } - if _, statErr := os.Stat(out + ".pid"); !os.IsNotExist(statErr) { - t.Errorf("a pidfile survived a failed detach (%v); pids are reused, so the next step would signal a stranger", statErr) - } + require.Error(t, err, "the child could never have started recording") + require.Contains(t, err.Error(), "records url") + _, statErr := os.Stat(out + ".pid") + require.True(t, os.IsNotExist(statErr), + "a pidfile survived a failed detach; pids are reused, so the next step would signal a stranger") } diff --git a/grafana-alertcheck/internal/gate/watch_test.go b/grafana-alertcheck/internal/gate/watch_test.go index 33c455af7..41ee3b09b 100644 --- a/grafana-alertcheck/internal/gate/watch_test.go +++ b/grafana-alertcheck/internal/gate/watch_test.go @@ -4,14 +4,14 @@ import ( "context" "errors" "fmt" - "maps" "os" "path/filepath" - "slices" "strings" "sync" "testing" "time" + + "github.com/stretchr/testify/require" ) // The two fixture rules every prepareWatch test below uses: one live, one @@ -73,12 +73,8 @@ func testStateRule(uid, title string, interval time.Duration, grafanaNow time.Ti func newLoopWriter(t *testing.T, path string, clock Clock) *Writer { t.Helper() w, err := NewWriter(path, clock) - if err != nil { - t.Fatalf("NewWriter: %v", err) - } - if err := w.WriteHeader(testHeader()); err != nil { - t.Fatalf("WriteHeader: %v", err) - } + require.NoError(t, err) + require.NoError(t, w.WriteHeader(testHeader())) return w } @@ -123,29 +119,21 @@ func TestWatchLoopPollsEachRuleAtItsOwnCadence(t *testing.T) { Concurrency: 2, Clock: clock, }) - if err != nil { - t.Fatalf("watchLoop: %v", err) - } + require.NoError(t, err) _, polls, sentinel, readErr := ReadLog(path) - if readErr != nil { - t.Fatalf("ReadLog: %v", readErr) - } - if sentinel == nil { - t.Fatal("no stopped sentinel after a clean stop") - } - if sentinel.Before(testNow.Add(300 * time.Second)) { - t.Errorf("sentinel at %s, want >= the stop time %s", sentinel, testNow.Add(300*time.Second)) - } + require.NoError(t, readErr) + require.NotNil(t, sentinel, "no stopped sentinel after a clean stop") + require.False(t, sentinel.Before(testNow.Add(300*time.Second))) // 300s of window at 5s and 150s, minus the initial stagger offset of up to // one cadence: 59-60 and 1-2. The assertion is the ratio, not the exact // count — a single global cycle would give both rules the same number. - if got := countPolls(polls, tightUID); got < 59 || got > 61 { - t.Errorf("tight rule polled %d times, want ~60 (300s at 5s)", got) - } - if got := countPolls(polls, slackUID); got < 1 || got > 3 { - t.Errorf("slack rule polled %d times, want ~2 (300s at 150s)", got) - } + got := countPolls(polls, tightUID) + require.GreaterOrEqual(t, got, 59) + require.LessOrEqual(t, got, 61) + got = countPolls(polls, slackUID) + require.GreaterOrEqual(t, got, 1) + require.LessOrEqual(t, got, 3) } // Fail-closed from the recorder's side: a recorder that dies must look exactly @@ -174,20 +162,12 @@ func TestWatchLoopHardErrorLeavesNoSentinel(t *testing.T) { Concurrency: 1, Clock: clock, }) - if !errors.Is(err, boom) { - t.Fatalf("watchLoop error = %v, want %v", err, boom) - } + require.ErrorIs(t, err, boom) _, polls, sentinel, readErr := ReadLog(path) - if readErr != nil { - t.Fatalf("ReadLog: %v", readErr) - } - if sentinel != nil { - t.Errorf("sentinel at %s after a failed recording; check would read that as a finished window", sentinel) - } - if len(polls) != 1 { - t.Errorf("kept %d polls, want the 1 that succeeded before the failure", len(polls)) - } + require.NoError(t, readErr) + require.Nil(t, sentinel, "check would read that as a finished window") + require.Len(t, polls, 1, "want the 1 that succeeded before the failure") } // SIGTERM arriving while a poll is in flight is a clean stop, so the aborted @@ -211,7 +191,7 @@ func TestWatchLoopSignalDuringPollIsACleanStop(t *testing.T) { return observation(now, testStateRule("r1", title, time.Minute, now)), nil }) - if err := watchLoop(ctx, watchLoopConfig{ + require.NoError(t, watchLoop(ctx, watchLoopConfig{ Src: src, Writer: w, Reducer: NewReducer(), @@ -219,15 +199,11 @@ func TestWatchLoopSignalDuringPollIsACleanStop(t *testing.T) { Cadence: map[string]time.Duration{"r1": 30 * time.Second}, Concurrency: 1, Clock: clock, - }); err != nil { - t.Fatalf("watchLoop: %v", err) - } + })) - if _, _, sentinel, err := ReadLog(path); err != nil { - t.Fatalf("ReadLog: %v", err) - } else if sentinel == nil { - t.Error("no sentinel after a signalled stop; check would call a fully observed window unobservable") - } + _, _, sentinel, err := ReadLog(path) + require.NoError(t, err) + require.NotNil(t, sentinel, "no sentinel after a signalled stop; check would call a fully observed window unobservable") } // TestWatchLoopWithNothingToPollStillFinishesTheLog covers the every-rule-is- @@ -242,7 +218,7 @@ func TestWatchLoopWithNothingToPollStillFinishesTheLog(t *testing.T) { return Observation{}, fmt.Errorf("nothing should be polled, got %q", title) }) - if err := watchLoop(context.Background(), watchLoopConfig{ + require.NoError(t, watchLoop(context.Background(), watchLoopConfig{ Src: src, Writer: w, Reducer: NewReducer(), @@ -251,20 +227,12 @@ func TestWatchLoopWithNothingToPollStillFinishesTheLog(t *testing.T) { Until: testNow.Add(time.Minute), Concurrency: 1, Clock: clock, - }); err != nil { - t.Fatalf("watchLoop: %v", err) - } + })) _, polls, sentinel, err := ReadLog(path) - if err != nil { - t.Fatalf("ReadLog: %v", err) - } - if len(polls) != 0 { - t.Errorf("wrote %d polls with nothing to poll", len(polls)) - } - if sentinel == nil { - t.Error("no sentinel: check cannot tell this recording from one that died") - } + require.NoError(t, err) + require.Empty(t, polls) + require.NotNil(t, sentinel, "no sentinel: check cannot tell this recording from one that died") } // TestWatchLoopPollBatchKeepsTheHeartbeatsItGot: one rule's failure must not @@ -293,20 +261,13 @@ func TestWatchLoopPollBatchKeepsTheHeartbeatsItGot(t *testing.T) { Concurrency: 2, Clock: clock, } - if err := cfg.pollBatch(context.Background(), []string{"ok", "bad"}); !errors.Is(err, boom) { - t.Fatalf("pollBatch error = %v, want %v", err, boom) - } - if err := w.Close(); err != nil { - t.Fatalf("Close: %v", err) - } + require.ErrorIs(t, cfg.pollBatch(context.Background(), []string{"ok", "bad"}), boom) + require.NoError(t, w.Close()) _, polls, _, err := ReadLog(path) - if err != nil { - t.Fatalf("ReadLog: %v", err) - } - if len(polls) != 1 || polls[0].RuleUID != "ok" { - t.Errorf("polls = %+v, want the one heartbeat that was actually observed", polls) - } + require.NoError(t, err) + require.Len(t, polls, 1) + require.Equal(t, "ok", polls[0].RuleUID) } // The vanish-versus-clear distinction at the one seam the parent/child handoff @@ -326,27 +287,20 @@ func TestReducerSeedFromKeepsMarkersAcrossTheHandoff(t *testing.T) { r := NewReducer() r.seedFrom([]Poll{parentPoll}) p := r.Reduce("r1", childObs) - if !slices.Contains(p.Vanished, key) { - t.Errorf("vanished = %v, want it to contain %q", p.Vanished, key) - } - if len(p.Cleared) != 0 { - t.Errorf("cleared = %v, want none: a vanish is not a recovery", p.Cleared) - } + require.Contains(t, p.Vanished, key) + require.Empty(t, p.Cleared, "a vanish is not a recovery") }) t.Run("unseeded loses the transition", func(t *testing.T) { p := NewReducer().Reduce("r1", childObs) - if len(p.Vanished) != 0 { - t.Fatalf("vanished = %v; this subtest exists to show the seed is what produces the marker", p.Vanished) - } + require.Empty(t, p.Vanished, "this subtest exists to show the seed is what produces the marker") }) t.Run("a not-found poll does not clear the seed", func(t *testing.T) { r := NewReducer() r.seedFrom([]Poll{parentPoll, {RuleUID: "r1", Found: false}}) - if p := r.Reduce("r1", childObs); !slices.Contains(p.Vanished, key) { - t.Errorf("vanished = %v, want it to contain %q: an absent rule leaves the abnormal set untouched", p.Vanished, key) - } + p := r.Reduce("r1", childObs) + require.Contains(t, p.Vanished, key, "an absent rule leaves the abnormal set untouched") }) } @@ -392,50 +346,34 @@ func TestPrepareWatchDoesNotWaitForPausedRules(t *testing.T) { src := watchTestSource(t, liveObservation(testNow)) prep, err := prepareWatch(context.Background(), cfg, src) - if err != nil { - t.Fatalf("prepareWatch: %v", err) - } - if err := prep.writer.Close(); err != nil { - t.Fatalf("Close: %v", err) - } + require.NoError(t, err) + require.NoError(t, prep.writer.Close()) header, polls, sentinel, err := ReadLog(cfg.Out) - if err != nil { - t.Fatalf("ReadLog: %v", err) - } - if sentinel != nil { - t.Error("the parent wrote a sentinel; that would tell check the recording ended before the child started") - } + require.NoError(t, err) + require.Nil(t, sentinel, "the parent wrote a sentinel; that would tell check the recording ended before the child started") - if len(header.Rules) != 2 { - t.Fatalf("header names %d rules, want both the live and the paused one", len(header.Rules)) - } + require.Len(t, header.Rules, 2) for _, lr := range header.Rules { - if lr.PollEverySeconds <= 0 { - t.Errorf("header rule %s records poll_every_seconds=%v; check needs a positive cadence to derive maxGap from", lr.UID, lr.PollEverySeconds) - } - if lr.UID == watchPausedUID && !lr.IsPaused { - t.Errorf("header rule %s: is_paused = false, want the resolve-time snapshot to say true", lr.UID) + require.Positive(t, lr.PollEverySeconds, + "check needs a positive cadence to derive maxGap from") + if lr.UID == watchPausedUID { + require.True(t, lr.IsPaused, "want the resolve-time snapshot to say true") } } // One poll, for the live rule only — and it is already in the log before // prepareWatch returned, which is the whole point of the record step. - if len(polls) != 1 || polls[0].RuleUID != watchActiveUID { - t.Fatalf("polls = %+v, want exactly one first observation of %s", polls, watchActiveUID) - } - if !polls[0].Found || !polls[0].GrafanaNow.Equal(testNow) { - t.Errorf("first poll = %+v, want a found observation at %s", polls[0], testNow) - } + require.Len(t, polls, 1) + require.Equal(t, watchActiveUID, polls[0].RuleUID) + require.True(t, polls[0].Found) + require.True(t, polls[0].GrafanaNow.Equal(testNow)) // The poll record holds the state histogram, asserted through a real // prepareWatch()/Reducer call rather than log_test.go's hand-built // Writer/ReadLog round trip. - if want := map[string]int{"normal": 1}; !maps.Equal(polls[0].Histogram, want) { - t.Errorf("Histogram = %v, want %v: watch must record the state histogram on every poll it writes", polls[0].Histogram, want) - } - if !strings.Contains(notes.String(), watchPausedTitle) || !strings.Contains(notes.String(), "paused") { - t.Errorf("notes do not mention the paused rule:\n%s", notes.String()) - } + require.Equal(t, map[string]int{"normal": 1}, polls[0].Histogram) + require.Contains(t, notes.String(), watchPausedTitle) + require.Contains(t, notes.String(), "paused") } // One authority for the cadence, from the writing side: whatever @@ -448,20 +386,13 @@ func TestPrepareWatchHeaderRecordsTheOverriddenCadence(t *testing.T) { src := watchTestSource(t, liveObservation(testNow)) prep, err := prepareWatch(context.Background(), cfg, src) - if err != nil { - t.Fatalf("prepareWatch: %v", err) - } + require.NoError(t, err) defer prep.writer.Close() - if got := prep.header.Rules[0].PollEverySeconds; got != 120 { - t.Errorf("header poll_every_seconds = %v, want 120 (the override, used verbatim and never clamped)", got) - } - if got := prep.timings[watchActiveUID].maxGap; got != 240*time.Second { - t.Errorf("maxGap = %s, want 240s (2 x the recorded cadence)", got) - } - if !strings.Contains(notes.String(), "--poll-interval") { - t.Errorf("notes do not report that the override exceeds half the evaluation interval:\n%s", notes.String()) - } + require.Equal(t, float64(120), prep.header.Rules[0].PollEverySeconds, + "the override, used verbatim and never clamped") + require.Equal(t, 240*time.Second, prep.timings[watchActiveUID].maxGap) + require.Contains(t, notes.String(), "--poll-interval") } // The budget check runs on the latencies the parent just measured, before the @@ -475,9 +406,7 @@ func TestPrepareWatchFailsWhenTheScheduleDoesNotFit(t *testing.T) { src := watchTestSource(t, obs) _, err := prepareWatch(context.Background(), cfg, src) - if err == nil { - t.Fatal("prepareWatch: no error on a schedule that cannot hold its own cadence") - } + require.Error(t, err, "a schedule cannot hold its own cadence") assertBudgetMessage(t, err.Error()) } @@ -493,20 +422,14 @@ func TestPrepareWatchVerifiesNormalInstancesAreVisible(t *testing.T) { src := watchTestSource(t, observation(testNow, rule)) _, err := prepareWatch(context.Background(), cfg, src) - if err == nil { - t.Fatal("prepareWatch: no error when totals claim normal instances the response omitted") - } - if !strings.Contains(err.Error(), "no longer returns normal instances") { - t.Errorf("error does not say the endpoint stopped returning normal instances: %v", err) - } + require.Error(t, err, "totals claim normal instances the response omitted") + require.Contains(t, err.Error(), "no longer returns normal instances") // The failure happens before any poll is appended, so the log holds a // header and nothing else. - if _, polls, _, readErr := ReadLog(cfg.Out); readErr != nil { - t.Fatalf("ReadLog: %v", readErr) - } else if len(polls) != 0 { - t.Errorf("wrote %d polls from an observation it refused to trust", len(polls)) - } + _, polls, _, readErr := ReadLog(cfg.Out) + require.NoError(t, readErr) + require.Empty(t, polls) } func TestPrepareWatchRejectsAnUnsupportedGrafana(t *testing.T) { @@ -515,11 +438,10 @@ func TestPrepareWatchRejectsAnUnsupportedGrafana(t *testing.T) { src := watchTestSource(t, liveObservation(testNow)) src.version = "12.4.0" - if _, err := prepareWatch(context.Background(), cfg, src); err == nil { - t.Fatal("prepareWatch: no error on an unsupported grafana version") - } else if !strings.Contains(err.Error(), "12.4.0") || !strings.Contains(err.Error(), "13.0.0") { - t.Errorf("error names neither what was found nor what is supported: %v", err) - } + _, err := prepareWatch(context.Background(), cfg, src) + require.Error(t, err) + require.Contains(t, err.Error(), "12.4.0") + require.Contains(t, err.Error(), "13.0.0") } // A rule that resolved in the ruler API but is absent from the state endpoint @@ -531,23 +453,14 @@ func TestPrepareWatchNotesAnAbsentRule(t *testing.T) { src := watchTestSource(t, observation(testNow)) // an authoritative, empty 2xx prep, err := prepareWatch(context.Background(), cfg, src) - if err != nil { - t.Fatalf("prepareWatch: %v", err) - } - if err := prep.writer.Close(); err != nil { - t.Fatalf("Close: %v", err) - } + require.NoError(t, err) + require.NoError(t, prep.writer.Close()) _, polls, _, err := ReadLog(cfg.Out) - if err != nil { - t.Fatalf("ReadLog: %v", err) - } - if len(polls) != 1 || polls[0].Found { - t.Fatalf("polls = %+v, want one poll recorded as not found", polls) - } - if !strings.Contains(notes.String(), "absent from the state endpoint") { - t.Errorf("notes do not warn about the absent rule:\n%s", notes.String()) - } + require.NoError(t, err) + require.Len(t, polls, 1) + require.False(t, polls[0].Found, "want one poll recorded as not found") + require.Contains(t, notes.String(), "absent from the state endpoint") } func TestWatchConfigValidation(t *testing.T) { @@ -575,26 +488,16 @@ func TestWatchConfigValidation(t *testing.T) { cfg := base() tc.mutate(&cfg) err := cfg.withDefaults().validate() - if err == nil { - t.Fatalf("validate: no error, want one naming %q", tc.want) - } - if !strings.Contains(err.Error(), tc.want) { - t.Errorf("validate error = %v, want it to name %q", err, tc.want) - } + require.Errorf(t, err, "validate: no error, want one naming %q", tc.want) + require.Containsf(t, err.Error(), tc.want, "validate error") }) } t.Run("defaults derive the pidfile and daemon log from the log path", func(t *testing.T) { cfg := base().withDefaults() - if cfg.PidFile != cfg.Out+".pid" { - t.Errorf("PidFile = %q, want %q — check finds the recorder by this convention", cfg.PidFile, cfg.Out+".pid") - } - if cfg.DaemonLog == "" { - t.Error("DaemonLog is empty: a detached child would have nowhere to explain a failure") - } - if err := cfg.validate(); err != nil { - t.Errorf("validate: %v", err) - } + require.Equal(t, cfg.Out+".pid", cfg.PidFile) + require.NotEmpty(t, cfg.DaemonLog, "a detached child would have nowhere to explain a failure") + require.NoError(t, cfg.validate()) }) } @@ -609,15 +512,10 @@ func TestChildScheduleUsesTheRecordedCadence(t *testing.T) { }} titles, cadence, err := childSchedule(h) - if err != nil { - t.Fatalf("childSchedule: %v", err) - } - if _, ok := titles["paused"]; ok { - t.Error("the child scheduled a rule that was paused when the window opened") - } - if got := cadence["fast"]; got != 5*time.Second { - t.Errorf("pollEvery = %s, want 5s from the header, not %s from the interval", got, defaultPollEvery(300)) - } + require.NoError(t, err) + _, ok := titles["paused"] + require.False(t, ok, "the child scheduled a rule that was paused when the window opened") + require.Equal(t, 5*time.Second, cadence["fast"]) } func TestChildScheduleRejectsAnUnusableHeader(t *testing.T) { @@ -641,11 +539,9 @@ func TestChildScheduleRejectsAnUnusableHeader(t *testing.T) { }, } { t.Run(tc.name, func(t *testing.T) { - if _, _, err := childSchedule(tc.h); err == nil { - t.Fatalf("childSchedule: no error, want one naming %q", tc.want) - } else if !strings.Contains(err.Error(), tc.want) { - t.Errorf("error = %v, want it to name %q", err, tc.want) - } + _, _, err := childSchedule(tc.h) + require.Errorf(t, err, "childSchedule: no error, want one naming %q", tc.want) + require.Contains(t, err.Error(), tc.want) }) } } @@ -670,85 +566,55 @@ func TestChildArgsCarryNoSecretsAndNoRuleSet(t *testing.T) { joined := strings.Join(args, " ") for _, want := range []string{DaemonChildFlag, "--out /tmp/log.jsonl", "--concurrency 3", "--until ", ReadyFDFlag + " 3"} { - if !strings.Contains(joined, want) { - t.Errorf("child args %q do not contain %q", joined, want) - } + require.Contains(t, joined, want) } for _, forbidden := range []string{"secret-token", "Example", "--folder", "--poll-interval", "--pidfile"} { - if strings.Contains(joined, forbidden) { - t.Errorf("child args %q contain %q, which must not reach argv", joined, forbidden) - } + require.NotContains(t, joined, forbidden) } } func TestPidFileRoundTrip(t *testing.T) { path := filepath.Join(t.TempDir(), "log.jsonl.pid") - if err := writePidFile(path, 4242); err != nil { - t.Fatalf("writePidFile: %v", err) - } + require.NoError(t, writePidFile(path, 4242)) pid, err := ReadPidFile(path) - if err != nil { - t.Fatalf("ReadPidFile: %v", err) - } - if pid != 4242 { - t.Errorf("pid = %d, want 4242", pid) - } + require.NoError(t, err) + require.Equal(t, 4242, pid) t.Run("garbage is an error, never a pid", func(t *testing.T) { bad := filepath.Join(t.TempDir(), "bad.pid") - if err := os.WriteFile(bad, []byte("not-a-pid\n"), 0o644); err != nil { - t.Fatalf("write: %v", err) - } - if _, err := ReadPidFile(bad); err == nil { - t.Error("ReadPidFile: no error on an unparseable pidfile") - } + require.NoError(t, os.WriteFile(bad, []byte("not-a-pid\n"), 0o644)) + _, err := ReadPidFile(bad) + require.Error(t, err, "no error on an unparseable pidfile") }) } func TestDaemonLogTail(t *testing.T) { t.Run("missing file is unreadable", func(t *testing.T) { out := daemonLogTail(filepath.Join(t.TempDir(), "nope.daemon.log"), 0) - if !strings.Contains(out, "unreadable") { - t.Errorf("daemonLogTail = %q, want it to name the file as unreadable", out) - } + require.Contains(t, out, "unreadable") }) t.Run("small file returns its content", func(t *testing.T) { path := filepath.Join(t.TempDir(), "small.daemon.log") - if err := os.WriteFile(path, []byte("line one\nline two\n"), 0o644); err != nil { - t.Fatalf("write: %v", err) - } - if out := daemonLogTail(path, 0); out != "line one\nline two" { - t.Errorf("daemonLogTail = %q, want the full trimmed content", out) - } + require.NoError(t, os.WriteFile(path, []byte("line one\nline two\n"), 0o644)) + require.Equal(t, "line one\nline two", daemonLogTail(path, 0)) }) t.Run("large file keeps only the tail", func(t *testing.T) { path := filepath.Join(t.TempDir(), "large.daemon.log") prefix := strings.Repeat("P", 1000) suffix := strings.Repeat("S", daemonLogTailBytes) - if err := os.WriteFile(path, []byte(prefix+suffix), 0o644); err != nil { - t.Fatalf("write: %v", err) - } - out := daemonLogTail(path, 0) - if out != suffix { - t.Errorf("daemonLogTail = %q, want exactly the trailing %d bytes (the %d leading bytes dropped)", out, daemonLogTailBytes, len(prefix)) - } + require.NoError(t, os.WriteFile(path, []byte(prefix+suffix), 0o644)) + require.Equal(t, suffix, daemonLogTail(path, 0)) }) t.Run("offset skips a previous run's content", func(t *testing.T) { path := filepath.Join(t.TempDir(), "shared.daemon.log") prior := strings.Repeat("p", 2000) - if err := os.WriteFile(path, []byte(prior), 0o644); err != nil { - t.Fatalf("write: %v", err) - } + require.NoError(t, os.WriteFile(path, []byte(prior), 0o644)) from := int64(len(prior)) thisRun := "this run's output\n" - if err := os.WriteFile(path, []byte(prior+thisRun), 0o644); err != nil { - t.Fatalf("write: %v", err) - } - if out := daemonLogTail(path, from); out != "this run's output" { - t.Errorf("daemonLogTail = %q, want only this run's bytes after offset %d", out, from) - } + require.NoError(t, os.WriteFile(path, []byte(prior+thisRun), 0o644)) + require.Equal(t, "this run's output", daemonLogTail(path, from)) }) }