From 1bb0b9d8e0eeab03756f10210bea7bf4237514b9 Mon Sep 17 00:00:00 2001 From: Bartek Tofel Date: Mon, 31 Aug 2026 16:16:32 +0200 Subject: [PATCH 1/4] chore: implement phase 7 MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Invariant defended: H3. The one question: can a rule be called alive because it looked alive one poll ago? proveCoverage (grafana-alertcheck/internal/gate/coverage.go) is the pure coverage function: nine checks over one rule's polls — sentinel, from-bounds, heartbeat continuity, health error/nodata, liveness, in-window pause, rule absence, KeepLast. Liveness is absolute, never a delta. Cross-domain comparisons translate by each poll's own skew and widen boundary segments by its skew bound, fail-closed. --- grafana-alertcheck/internal/gate/coverage.go | 352 ++++++++++ .../internal/gate/coverage_test.go | 628 ++++++++++++++++++ 2 files changed, 980 insertions(+) create mode 100644 grafana-alertcheck/internal/gate/coverage.go create mode 100644 grafana-alertcheck/internal/gate/coverage_test.go diff --git a/grafana-alertcheck/internal/gate/coverage.go b/grafana-alertcheck/internal/gate/coverage.go new file mode 100644 index 000000000..36d2c3dac --- /dev/null +++ b/grafana-alertcheck/internal/gate/coverage.go @@ -0,0 +1,352 @@ +package gate + +import ( + "fmt" + "slices" + "time" +) + +// keepLastReason is the instance Reason that check 9 watches for (§10.2). +const keepLastReason = "KeepLast" + +// Obligations this phase leaves for later ones — carried forward the same +// way P6's own deviations list did, so a later review has something concrete +// to check against: +// +// - fromFutureTolerance (§5: 60s) has no constant and no hard-error check +// anywhere yet. Check 2 below implements only "from < StartedAt"; the +// second clause — from more than fromFutureTolerance ahead is a hard +// error — is once-per-run input validation, not a per-rule coverage +// check, and belongs to Check's construction in a later phase (P9). +// - decide (P8) must read a rule's skipped status from the definitions +// (LoggedRule.IsPaused / Definition.IsPaused), never from the polls, and +// must do so BEFORE calling proveCoverage for that rule: a rule paused +// before the window opened is never scheduled or polled (§4.3), so it +// reaches this function with zero polls and today reads as one large +// heartbeat_gap, not skipped (pinned by +// TestProveCoverage_SkippedRuleWithZeroPollsPinnedAsHeartbeatGap). + +// UnobservableReason names why proveCoverage could not prove a rule's window. +// It is machine-readable — this reaches the action's JSON outputs, so it is a +// published vocabulary like Outcome (§19.0); prose belongs in Notes. +type UnobservableReason string + +const ( + ReasonNoSentinel UnobservableReason = "no_sentinel" + ReasonSentinelEarly UnobservableReason = "sentinel_early" + ReasonFromBeforeRecord UnobservableReason = "from_before_record" + ReasonHeartbeatGap UnobservableReason = "heartbeat_gap" + ReasonHealthError UnobservableReason = "health_error" + ReasonStaleEvaluation UnobservableReason = "stale_evaluation" + ReasonPausedInWindow UnobservableReason = "paused_in_window" + ReasonRuleAbsent UnobservableReason = "rule_absent" + // ReasonDrainTimeout is set by check.go's drain wait (a later phase), + // never by proveCoverage: the wait is I/O and must not be added to this + // pure function — that would put HTTP inside the pure layer and destroy + // the seam §2's architecture depends on. + ReasonDrainTimeout UnobservableReason = "drain_timeout" +) + +// CoverageResult is proveCoverage's whole answer for one rule. No interval +// list: proved-or-not plus the largest gap and where is everything a human +// reads on exit 2, and everything §20.2's table needs. +type CoverageResult struct { + Proved bool + LargestGap time.Duration + LargestGapAt time.Time + Unobservable bool + Reason UnobservableReason + Notes []string + // BlindFor is the worst staleness (GrafanaNow - LastEvaluation) that + // tripped check 6; zero when check 6 never fired. + BlindFor time.Duration +} + +// proveCoverage applies the nine coverage checks (§6, §10, §14) to one rule's +// polls and is PURE: no HTTP, no files, no clock reads — everything it needs +// arrives as an argument, which is what lets §22's tests build []Poll literals +// instead of a fixture server (§2). +// +// polls need not be pre-filtered to this rule: proveCoverage selects by +// def.UID itself, exactly as Reduce selects by UID rather than by title +// (§14.5) — a caller handing it a whole log's polls must not have to +// pre-filter to get a correct answer. +// +// Every check always runs, even once an earlier one has already set +// Unobservable: LargestGap and the notes are diagnostics an operator reads on +// exit 2 regardless of which check actually failed (§20.2). Reason names the +// FIRST check, in the order below, that failed; a later failure still adds +// its own Note. +func proveCoverage(h Header, polls []Poll, sentinel *time.Time, t ruleTimings, def Definition, + from, to time.Time, grace time.Duration) CoverageResult { + + windowEnd := to.Add(grace) + + var rulePolls []Poll + for _, p := range polls { + if p.RuleUID == def.UID { + rulePolls = append(rulePolls, p) + } + } + // Stable, not sort.Slice: two polls sharing a GrafanaNow (a coarse Date + // header, or a corrupted/replayed log) must not reorder nondeterministically + // in a function that promises to be pure. + slices.SortStableFunc(rulePolls, func(a, b Poll) int { return a.GrafanaNow.Compare(b.GrafanaNow) }) + + var res CoverageResult + fail := func(reason UnobservableReason, note string) { + res.Unobservable = true + if res.Reason == "" { + res.Reason = reason + } + res.Notes = append(res.Notes, fmt.Sprintf("rule %q: %s", def.Title, note)) + } + + // Check 1 — sentinel (§4.5). Present and At >= to+grace -> coverage + // provable; absent, or short of it, is never a pass. A recorder that died + // early must look exactly like a coverage gap, because it is one. + switch { + case sentinel == nil: + fail(ReasonNoSentinel, "no stopped sentinel: the recorder never reported finishing") + case sentinel.Before(windowEnd): + fail(ReasonSentinelEarly, fmt.Sprintf("stopped sentinel at %s is before the required %s (to+grace)", + sentinel.Format(time.RFC3339), windowEnd.Format(time.RFC3339))) + } + + // Check 2 — from bounds (§7), first sentence only: from < StartedAt makes + // coverage unprovable, no matter how healthy the polls that DO exist look. + // Both are runner-domain clock reads (the recorder's own Clock.Now()), so + // no cross-domain translation applies here. The second sentence — from + // more than fromFutureTolerance ahead is a hard error — is Check's input + // validation, once per run rather than per rule, and belongs to a later + // phase: this function has no error return, only a per-rule verdict. + if from.Before(h.StartedAt) { + fail(ReasonFromBeforeRecord, fmt.Sprintf( + "requested from %s is before recording started at %s", from.Format(time.RFC3339), h.StartedAt.Format(time.RFC3339))) + } + + // Filtered once, here, and threaded through every remaining check — + // ruleHeartbeatGap included — rather than re-filtered per check: two + // independent filters over the same polls would only invite one of them + // drifting from the other's membership test. + inWindow := inWindowPolls(rulePolls, from, windowEnd) + + // Check 3 — heartbeat continuity (§6). Data at both ends with a hole in + // between is not enough (§22.4): this scans every gap inside the window, + // not just its edges. + res.LargestGap, res.LargestGapAt = ruleHeartbeatGap(inWindow, from, windowEnd) + if res.LargestGap > t.maxGap { + fail(ReasonHeartbeatGap, fmt.Sprintf( + "gap of %s starting at %s exceeds maxGap %s", res.LargestGap, res.LargestGapAt.Format(time.RFC3339), t.maxGap)) + } + + // Check 4 — health=="error" (§10.1). A short blip is a note only (§22.1: + // one failed evaluation must not exit 2 over an otherwise clean window); + // only a run longer than healthGrace consumes coverage. + if runLen, sawAny := longestHealthRun(inWindow, "error"); sawAny { + res.Notes = append(res.Notes, fmt.Sprintf("rule %q: health=error observed (longest run %s)", def.Title, runLen)) + if runLen > t.healthGrace { + fail(ReasonHealthError, fmt.Sprintf("health=error for %s exceeds healthGrace %s", runLen, t.healthGrace)) + } + } + + // Check 5 — health=="nodata" (§10.1/§10.2). Never fatal here: 96% of the + // fleet runs no_data_state:OK, so treating this as fatal by default would + // block nearly every healthy deploy in an idle environment. Escalating it + // under Policy.NodataIsUnobservable is decide's job (a later phase), + // applied directly against the raw polls — this pure function has no + // Policy to consult and must not invent one. + if _, sawAny := longestHealthRun(inWindow, "nodata"); sawAny { + res.Notes = append(res.Notes, fmt.Sprintf("rule %q: health=nodata observed (not fatal; see --nodata-is-unobservable)", def.Title)) + } + + // Check 6 — liveness (H3). Absolute only, per poll: GrafanaNow and + // LastEvaluation are both Grafana-domain reads off the SAME response, so + // this is a same-domain comparison and uses raw values — never a delta + // against a previous poll, which reports stale on ~half the polls of a + // perfectly healthy rule (polling runs at intervalSeconds/2). + // + // Skipped only for a poll whose own flags SAY there is nothing to check: + // IsPaused (a zero LastEvaluation is legal only while paused, §2.3; check + // 7 is its detector) or !Found (no rule, no evaluation; check 8 is its + // detector). Deliberately NOT skipped merely because LastEvaluation is + // zero: ReadLog does no field validation, so a corrupted or hand-edited + // log line can claim found:true, is_paused:false and still carry a zero + // LastEvaluation, and that combination must read as maximally stale + // rather than being silently waved through. + var staleCount int + var worstStale time.Duration + var worstStaleAt time.Time + for _, p := range inWindow { + if p.IsPaused || !p.Found { + continue + } + if stale := p.GrafanaNow.Sub(p.LastEvaluation); stale > t.evalStaleAfter { + staleCount++ + if stale > worstStale { + worstStale, worstStaleAt = stale, p.GrafanaNow + } + } + } + if staleCount > 0 { + res.BlindFor = worstStale + fail(ReasonStaleEvaluation, fmt.Sprintf( + "lastEvaluation stale on %d poll(s); worst %s (> evalStaleAfter %s) as of %s", + staleCount, worstStale, t.evalStaleAfter, worstStaleAt.Format(time.RFC3339))) + } + + // Check 7 — isPaused in-window (§12.2, §14.8). The PRIMARY pause + // detector: liveness (check 6) is only the backup for what IsPaused + // cannot show (a deleted rule, a stopped scheduler, a blocked + // evaluation). This is what catches pause-then-unpause, which the drain + // wait alone passes (§14.7). + var pausedCount int + var pausedAt time.Time + for _, p := range inWindow { + if p.IsPaused { + pausedCount++ + if pausedAt.IsZero() { + pausedAt = p.GrafanaNow + } + } + } + if pausedCount > 0 { + fail(ReasonPausedInWindow, fmt.Sprintf("observed paused on %d poll(s), first at %s", pausedCount, pausedAt.Format(time.RFC3339))) + } + + // Check 8 — rule absent (§14.5). Found==false is authoritative (P2 + // already retried every transport failure before a Poll record ever + // exists): the rule resolved at resolve time but the state endpoint + // stopped serving it. Never drop a watched rule from the verdict set + // silently. + var absentCount int + var absentAt time.Time + for _, p := range inWindow { + if !p.Found { + absentCount++ + if absentAt.IsZero() { + absentAt = p.GrafanaNow + } + } + } + if absentCount > 0 { + fail(ReasonRuleAbsent, fmt.Sprintf("state endpoint returned no rule on %d poll(s), first at %s", absentCount, absentAt.Format(time.RFC3339))) + } + + // Check 9 — KeepLast (§10.2). A note, never fatal. It surfaces only as an + // instance Reason after P1.2a's parsing, and Reasons keys can be + // comma-joined composites, so membership (reasonsContain) is required — + // indexing "KeepLast" directly would miss "KeepLast, MissingSeries". + for _, p := range inWindow { + if reasonsContain(p.Reasons, keepLastReason) { + res.Notes = append(res.Notes, fmt.Sprintf( + "rule %q: KeepLast observed at %s: a held-over state may hide a real blind spot", def.Title, p.GrafanaNow.Format(time.RFC3339))) + break + } + } + + res.Proved = !res.Unobservable + return res +} + +// inWindowPolls filters polls to those inside [from, windowEnd] using the +// CROSS-DOMAIN membership test (§16): each poll's Grafana-domain GrafanaNow +// is translated to the runner domain by its OWN skew, and its own skew bound +// is the membership tolerance, so a poll that is genuinely inside the window +// is never excluded by ordinary clock imprecision. +// +// Everything downstream of this filter (health runs, liveness, pause, +// absence) reads the poll's raw fields: GrafanaNow paired with +// LastEvaluation on the SAME response, or one poll's GrafanaNow against the +// next's, are same-domain comparisons and need no translation (§16, "Clock +// domains" — only window membership and check 3's two boundary segments do). +func inWindowPolls(polls []Poll, from, windowEnd time.Time) []Poll { + var out []Poll + for _, p := range polls { + bound := p.SkewBound() + runner := p.GrafanaNow.Add(-p.Skew()) + if runner.Before(from.Add(-bound)) || runner.After(windowEnd.Add(bound)) { + continue + } + out = append(out, p) + } + return out +} + +// ruleHeartbeatGap finds the largest unobserved span inside [from, windowEnd] +// (§6), including the two boundary segments — which is why "data at both +// ends with a hole in the middle" still fails (§22.4): the segment between +// the polls just inside each edge is exactly what this measures. in must +// already be filtered to this window (inWindowPolls) and sorted by +// GrafanaNow — proveCoverage computes that filter once and threads it through +// every check, this one included, rather than each check re-filtering. +// +// The two boundary segments compare a Grafana-domain poll time against the +// runner-domain from/windowEnd, so each is translated by its own poll's skew +// AND widened by that same poll's skew bound (§16: "with that poll's bound as +// the tolerance") — on the side that makes the segment larger, never smaller, +// so an uncertain boundary reads as at least as big a gap as it might really +// be. Understating it by up to the bound would be fail-open. The spacing +// BETWEEN consecutive polls compares two Grafana-domain reads to each other — +// same domain — and uses the raw GrafanaNow difference, no bound needed. +func ruleHeartbeatGap(in []Poll, from, windowEnd time.Time) (largestGap time.Duration, largestGapAt time.Time) { + if len(in) == 0 { + return windowEnd.Sub(from), from + } + + runnerOf := func(p Poll) time.Time { return p.GrafanaNow.Add(-p.Skew()) } + + first := in[0] + if gap := runnerOf(first).Sub(from) + first.SkewBound(); gap > largestGap { + largestGap, largestGapAt = gap, from + } + for i := 1; i < len(in); i++ { + if gap := in[i].GrafanaNow.Sub(in[i-1].GrafanaNow); gap > largestGap { + largestGap, largestGapAt = gap, runnerOf(in[i-1]) + } + } + last := in[len(in)-1] + if gap := windowEnd.Sub(runnerOf(last)) + last.SkewBound(); gap > largestGap { + largestGap, largestGapAt = gap, runnerOf(last) + } + return largestGap, largestGapAt +} + +// longestHealthRun returns the longest contiguous wall-clock span (§10.1) +// during which polls — already sorted by GrafanaNow, same-domain spacing +// (§16) — read the given rule-level Health, and whether any poll matched it +// at all. +// +// It detects the span as it accumulates rather than waiting for the run to +// end, so an open-ended run that is still failing at the last poll in the +// window is measured correctly without needing data past the window: waiting +// for the run to "end" would have to assume the best case about what happens +// next, which is exactly what this gate must not do (§1). +func longestHealthRun(polls []Poll, health string) (longest time.Duration, sawAny bool) { + var runStart time.Time + for _, p := range polls { + if p.Health != health { + runStart = time.Time{} + continue + } + sawAny = true + if runStart.IsZero() { + runStart = p.GrafanaNow + } + if span := p.GrafanaNow.Sub(runStart); span > longest { + longest = span + } + } + return longest, sawAny +} + +// reasonsContain reports whether any key of reasons names want, honoring +// Grafana's comma-joined composite reason strings via reasonNames (log.go). +func reasonsContain(reasons map[string]int, want string) bool { + for reason := range reasons { + if reasonNames(reason, want) { + return true + } + } + return false +} diff --git a/grafana-alertcheck/internal/gate/coverage_test.go b/grafana-alertcheck/internal/gate/coverage_test.go new file mode 100644 index 000000000..429139602 --- /dev/null +++ b/grafana-alertcheck/internal/gate/coverage_test.go @@ -0,0 +1,628 @@ +package gate + +import ( + "strings" + "testing" + "time" +) + +func TestProveCoverage_CleanWindowIsProved(t *testing.T) { + from := time.Date(2026, 1, 1, 0, 0, 0, 0, time.UTC) + to := from.Add(10 * time.Minute) + rt := newRuleTimings(30*time.Second, 60) + def := Definition{UID: "r1", Title: "R1"} + + var polls []Poll + for ts := from; !ts.After(to); ts = ts.Add(30 * time.Second) { + polls = append(polls, Poll{RuleUID: "r1", GrafanaNow: ts, Found: true, Health: "ok", State: "inactive", LastEvaluation: ts}) + } + sentinel := to + + res := proveCoverage(Header{StartedAt: from.Add(-time.Hour)}, polls, &sentinel, rt, def, from, to, 0) + if !res.Proved || res.Unobservable || res.Reason != "" { + t.Fatalf("res = %+v, want a clean proved window", res) + } +} + +func TestProveCoverage_FiltersPollsByUID(t *testing.T) { + from := time.Date(2026, 1, 1, 0, 0, 0, 0, time.UTC) + to := from.Add(10 * time.Minute) + rt := newRuleTimings(30*time.Second, 60) + def := Definition{UID: "r1", Title: "R1"} + + var polls []Poll + for ts := from; !ts.After(to); ts = ts.Add(30 * time.Second) { + polls = append(polls, Poll{RuleUID: "r1", GrafanaNow: ts, Found: true, Health: "ok", LastEvaluation: ts}) + // A different rule's polls, deliberately broken, must never + // contaminate r1's verdict: proveCoverage selects by UID itself. + polls = append(polls, Poll{RuleUID: "other", GrafanaNow: ts, Found: false}) + } + sentinel := to + + res := proveCoverage(Header{StartedAt: from.Add(-time.Hour)}, polls, &sentinel, rt, def, from, to, 0) + if !res.Proved { + t.Fatalf("res = %+v, want proved: a different rule's broken polls must not affect this rule's verdict", res) + } +} + +// --- Check 1: sentinel (§4.5) --- + +func TestProveCoverage_NoSentinelIsUnobservable(t *testing.T) { + from := time.Date(2026, 1, 1, 0, 0, 0, 0, time.UTC) + to := from.Add(time.Minute) + rt := newRuleTimings(30*time.Second, 60) + def := Definition{UID: "r1", Title: "R1"} + + res := proveCoverage(Header{StartedAt: from.Add(-time.Hour)}, nil, nil, rt, def, from, to, 0) + if res.Proved || res.Reason != ReasonNoSentinel { + t.Fatalf("res = %+v, want unobservable/no_sentinel: an absent sentinel must never be a pass", res) + } +} + +func TestProveCoverage_SentinelBeforeGraceIsUnobservable(t *testing.T) { + from := time.Date(2026, 1, 1, 0, 0, 0, 0, time.UTC) + to := from.Add(10 * time.Minute) + grace := 2 * time.Minute + rt := newRuleTimings(30*time.Second, 60) + def := Definition{UID: "r1", Title: "R1"} + + sentinel := to.Add(grace).Add(-time.Second) // one second short of to+grace + res := proveCoverage(Header{StartedAt: from.Add(-time.Hour)}, nil, &sentinel, rt, def, from, to, grace) + if res.Reason != ReasonSentinelEarly { + t.Fatalf("Reason = %q, want sentinel_early", res.Reason) + } +} + +func TestProveCoverage_SentinelExactlyAtGraceIsFine(t *testing.T) { + from := time.Date(2026, 1, 1, 0, 0, 0, 0, time.UTC) + to := from.Add(10 * time.Minute) + grace := 2 * time.Minute + windowEnd := to.Add(grace) + rt := newRuleTimings(30*time.Second, 60) + def := Definition{UID: "r1", Title: "R1"} + + var polls []Poll + for ts := from; !ts.After(windowEnd); ts = ts.Add(30 * time.Second) { + polls = append(polls, Poll{RuleUID: "r1", GrafanaNow: ts, Found: true, Health: "ok", LastEvaluation: ts}) + } + sentinel := windowEnd + + res := proveCoverage(Header{StartedAt: from.Add(-time.Hour)}, polls, &sentinel, rt, def, from, to, grace) + if !res.Proved { + t.Fatalf("Proved = false, want true: sentinel exactly at to+grace must satisfy check 1: %+v", res) + } +} + +// --- Check 2: from bounds (§7) --- + +func TestProveCoverage_FromBeforeRecordIsUnobservable(t *testing.T) { + started := time.Date(2026, 1, 1, 1, 0, 0, 0, time.UTC) + from := started.Add(-time.Minute) // the requested window opens before recording started + to := from.Add(10 * time.Minute) + rt := newRuleTimings(30*time.Second, 60) + def := Definition{UID: "r1", Title: "R1"} + + sentinel := to + res := proveCoverage(Header{StartedAt: started}, nil, &sentinel, rt, def, from, to, 0) + if res.Reason != ReasonFromBeforeRecord { + t.Fatalf("Reason = %q, want from_before_record", res.Reason) + } +} + +// --- Check 3: heartbeat continuity (§6) --- + +// TestProveCoverage_HeartbeatGapBetweenBoundariesIsUnobservable is §22.4's +// core regression: data at both ends with a hole between is not enough. +func TestProveCoverage_HeartbeatGapBetweenBoundariesIsUnobservable(t *testing.T) { + from := time.Date(2026, 1, 1, 0, 0, 0, 0, time.UTC) + to := from.Add(10 * time.Minute) + rt := newRuleTimings(30*time.Second, 60) // maxGap = 60s + def := Definition{UID: "r1", Title: "R1"} + + polls := []Poll{ + {RuleUID: "r1", GrafanaNow: from.Add(time.Second), Found: true, Health: "ok", LastEvaluation: from}, + {RuleUID: "r1", GrafanaNow: to.Add(-time.Second), Found: true, Health: "ok", LastEvaluation: to}, + } + sentinel := to + + res := proveCoverage(Header{StartedAt: from.Add(-time.Hour)}, polls, &sentinel, rt, def, from, to, 0) + if res.Reason != ReasonHeartbeatGap { + t.Fatalf("Reason = %q, want heartbeat_gap: healthy edges with a hole in the middle must still fail (§22.4)", res.Reason) + } + // The gap is the SPACING between the two polls (598s), not either + // boundary segment (1s each) — pin the actual values, not just the verdict. + if res.LargestGap != 598*time.Second { + t.Fatalf("LargestGap = %s, want 598s (the spacing between the two polls, not a boundary segment)", res.LargestGap) + } + wantAt := from.Add(time.Second) + if !res.LargestGapAt.Equal(wantAt) { + t.Fatalf("LargestGapAt = %s, want %s (where the gap starts, at the first poll)", res.LargestGapAt, wantAt) + } +} + +// --- Check 4/5: health (§10.1/§10.2) --- + +func TestProveCoverage_HealthErrorShortBlipPassesWithNote(t *testing.T) { + from := time.Date(2026, 1, 1, 0, 0, 0, 0, time.UTC) + to := from.Add(10 * time.Minute) + rt := newRuleTimings(30*time.Second, 60) // healthGrace = 60s + def := Definition{UID: "r1", Title: "R1"} + + var polls []Poll + blip := from.Add(2 * time.Minute) + for ts := from; !ts.After(to); ts = ts.Add(30 * time.Second) { + health := "ok" + if ts.Equal(blip) { + health = "error" // one isolated failed evaluation + } + polls = append(polls, Poll{RuleUID: "r1", GrafanaNow: ts, Found: true, Health: health, LastEvaluation: ts}) + } + sentinel := to + + res := proveCoverage(Header{StartedAt: from.Add(-time.Hour)}, polls, &sentinel, rt, def, from, to, 0) + if !res.Proved { + t.Fatalf("Proved = false, want true: one failed evaluation must not fail an otherwise clean window (§22.1): %+v", res) + } + if !anyContains(res.Notes, "health=error") { + t.Fatalf("Notes = %v, want a health=error note even though it did not fail the window", res.Notes) + } +} + +func TestProveCoverage_HealthErrorSustainedIsUnobservable(t *testing.T) { + from := time.Date(2026, 1, 1, 0, 0, 0, 0, time.UTC) + to := from.Add(10 * time.Minute) + rt := newRuleTimings(30*time.Second, 60) // healthGrace = 60s + def := Definition{UID: "r1", Title: "R1"} + + runStart, runEnd := from.Add(2*time.Minute), from.Add(5*time.Minute) // a 3-minute run, well past healthGrace + var polls []Poll + for ts := from; !ts.After(to); ts = ts.Add(30 * time.Second) { + health := "ok" + if !ts.Before(runStart) && !ts.After(runEnd) { + health = "error" + } + polls = append(polls, Poll{RuleUID: "r1", GrafanaNow: ts, Found: true, Health: health, LastEvaluation: ts}) + } + sentinel := to + + res := proveCoverage(Header{StartedAt: from.Add(-time.Hour)}, polls, &sentinel, rt, def, from, to, 0) + if res.Reason != ReasonHealthError { + t.Fatalf("Reason = %q, want health_error for a run that outlasts healthGrace", res.Reason) + } +} + +func TestProveCoverage_HealthNodataNeverFatalHere(t *testing.T) { + from := time.Date(2026, 1, 1, 0, 0, 0, 0, time.UTC) + to := from.Add(10 * time.Minute) + rt := newRuleTimings(30*time.Second, 60) + def := Definition{UID: "r1", Title: "R1"} + + var polls []Poll + for ts := from; !ts.After(to); ts = ts.Add(30 * time.Second) { + polls = append(polls, Poll{RuleUID: "r1", GrafanaNow: ts, Found: true, Health: "nodata", LastEvaluation: ts}) + } + sentinel := to + + res := proveCoverage(Header{StartedAt: from.Add(-time.Hour)}, polls, &sentinel, rt, def, from, to, 0) + if !res.Proved { + t.Fatalf("Proved = false, want true: health=nodata for the WHOLE window must still not be fatal by itself "+ + "(escalating it is Policy.NodataIsUnobservable's job, applied by decide in a later phase): %+v", res) + } + if !anyContains(res.Notes, "health=nodata") { + t.Fatalf("Notes = %v, want a health=nodata note", res.Notes) + } +} + +// --- Check 6: liveness / H3 --- + +// TestProveCoverage_LivenessAbsoluteNeverFalseStale is §22.7's disproportionate +// test: a healthy rule polled at intervalSeconds/2, across the full window, +// must show zero staleness violations. lastEvaluation only advances once per +// full evaluation interval here — the realistic shape a delta check +// misreads as stale on roughly half of all polls (H3). +func TestProveCoverage_LivenessAbsoluteNeverFalseStale(t *testing.T) { + from := time.Date(2026, 1, 1, 0, 0, 0, 0, time.UTC) + pollEvery := 30 * time.Second + intervalSeconds := 60 + windowEnd := from.Add(10 * time.Minute) + rt := newRuleTimings(pollEvery, intervalSeconds) + def := Definition{UID: "r1", Title: "R1"} + + var polls []Poll + lastEval := from + for ts := from; !ts.After(windowEnd); ts = ts.Add(pollEvery) { + if ts.Sub(lastEval) >= time.Duration(intervalSeconds)*time.Second { + lastEval = ts + } + polls = append(polls, Poll{RuleUID: "r1", GrafanaNow: ts, Found: true, Health: "ok", LastEvaluation: lastEval}) + } + sentinel := windowEnd + + res := proveCoverage(Header{StartedAt: from.Add(-time.Hour)}, polls, &sentinel, rt, def, from, windowEnd, 0) + if res.Reason == ReasonStaleEvaluation || res.BlindFor != 0 { + t.Fatalf("proveCoverage flagged staleness on a healthy rule polled at intervalSeconds/2 — H3 must be absolute, "+ + "never a delta against a previous poll: %+v", res) + } + if !res.Proved { + t.Fatalf("Proved = false, want true: %+v (notes: %v)", res, res.Notes) + } +} + +func TestProveCoverage_StaleEvaluationIsUnobservable(t *testing.T) { + from := time.Date(2026, 1, 1, 0, 0, 0, 0, time.UTC) + to := from.Add(10 * time.Minute) + rt := newRuleTimings(30*time.Second, 60) // evalStaleAfter = 120s + def := Definition{UID: "r1", Title: "R1"} + + // Dense, otherwise-healthy polling so heartbeat continuity (check 3) + // stays intact — only check 6 should be able to fire. + polls := denseHealthyPolls("r1", from, to, 30*time.Second) + staleAt := from.Add(5 * time.Minute) + for i := range polls { + if polls[i].GrafanaNow.Equal(staleAt) { + polls[i].LastEvaluation = staleAt.Add(-3 * time.Minute) + } + } + sentinel := to + + res := proveCoverage(Header{StartedAt: from.Add(-time.Hour)}, polls, &sentinel, rt, def, from, to, 0) + if res.Reason != ReasonStaleEvaluation { + t.Fatalf("Reason = %q, want stale_evaluation", res.Reason) + } + if res.BlindFor != 3*time.Minute { + t.Fatalf("BlindFor = %s, want 3m", res.BlindFor) + } +} + +func TestProveCoverage_ZeroLastEvaluationNeverFalseStale(t *testing.T) { + from := time.Date(2026, 1, 1, 0, 0, 0, 0, time.UTC) + to := from.Add(10 * time.Minute) + rt := newRuleTimings(30*time.Second, 60) + def := Definition{UID: "r1", Title: "R1"} + + // A paused rule legitimately reports the zero time (§2.3); check 6 must + // not read that as an enormous staleness violation. Check 7 is its + // detector. + polls := []Poll{ + {RuleUID: "r1", GrafanaNow: from.Add(time.Minute), Found: true, IsPaused: true}, + } + sentinel := to + + res := proveCoverage(Header{StartedAt: from.Add(-time.Hour)}, polls, &sentinel, rt, def, from, to, 0) + if res.Reason == ReasonStaleEvaluation { + t.Fatalf("a zero lastEvaluation on a paused poll must not trigger check 6: %+v", res) + } +} + +// --- Check 7: isPaused in-window (§12.2, §14.8) --- + +func TestProveCoverage_PausedInWindowIsUnobservable(t *testing.T) { + from := time.Date(2026, 1, 1, 0, 0, 0, 0, time.UTC) + to := from.Add(10 * time.Minute) + rt := newRuleTimings(30*time.Second, 60) + def := Definition{UID: "r1", Title: "R1"} + + // Dense, otherwise-healthy polling so heartbeat continuity (check 3) + // stays intact — only check 7 should be able to fire. + polls := denseHealthyPolls("r1", from, to, 30*time.Second) + pausedAt := from.Add(5 * time.Minute) + for i := range polls { + if polls[i].GrafanaNow.Equal(pausedAt) { + polls[i].IsPaused = true + polls[i].LastEvaluation = time.Time{} // legal only while paused, §2.3 + } + } + sentinel := to + + res := proveCoverage(Header{StartedAt: from.Add(-time.Hour)}, polls, &sentinel, rt, def, from, to, 0) + if res.Reason != ReasonPausedInWindow { + t.Fatalf("Reason = %q, want paused_in_window", res.Reason) + } +} + +// --- Check 8: rule absent (§14.5) --- + +func TestProveCoverage_RuleAbsentIsUnobservable(t *testing.T) { + from := time.Date(2026, 1, 1, 0, 0, 0, 0, time.UTC) + to := from.Add(10 * time.Minute) + rt := newRuleTimings(30*time.Second, 60) + def := Definition{UID: "r1", Title: "R1"} + + // Dense, otherwise-healthy polling so heartbeat continuity (check 3) + // stays intact — only check 8 should be able to fire. + polls := denseHealthyPolls("r1", from, to, 30*time.Second) + absentAt := from.Add(5 * time.Minute) + for i := range polls { + if polls[i].GrafanaNow.Equal(absentAt) { + polls[i].Found = false + polls[i].Health = "" + polls[i].LastEvaluation = time.Time{} + } + } + sentinel := to + + res := proveCoverage(Header{StartedAt: from.Add(-time.Hour)}, polls, &sentinel, rt, def, from, to, 0) + if res.Reason != ReasonRuleAbsent { + t.Fatalf("Reason = %q, want rule_absent", res.Reason) + } +} + +// denseHealthyPolls builds a clean poll sequence at a fixed cadence, with +// zero staleness and nothing abnormal — the baseline the single-check tests +// mutate exactly one poll of, so heartbeat continuity (check 3) never +// confounds the check under test. +func denseHealthyPolls(uid string, from, to time.Time, every time.Duration) []Poll { + var out []Poll + for ts := from; !ts.After(to); ts = ts.Add(every) { + out = append(out, Poll{RuleUID: uid, GrafanaNow: ts, Found: true, Health: "ok", LastEvaluation: ts}) + } + return out +} + +// --- Check 9: KeepLast (§10.2) --- + +func TestProveCoverage_KeepLastIsNoteOnly(t *testing.T) { + from := time.Date(2026, 1, 1, 0, 0, 0, 0, time.UTC) + to := from.Add(10 * time.Minute) + rt := newRuleTimings(30*time.Second, 60) + def := Definition{UID: "r1", Title: "R1"} + + var polls []Poll + for ts := from; !ts.After(to); ts = ts.Add(30 * time.Second) { + polls = append(polls, Poll{ + RuleUID: "r1", GrafanaNow: ts, Found: true, Health: "ok", LastEvaluation: ts, + // A comma-joined composite — reasonsContain must match by + // membership, never by an exact key, per P5's markers. + Reasons: map[string]int{"KeepLast, MissingSeries": 1}, + }) + } + sentinel := to + + res := proveCoverage(Header{StartedAt: from.Add(-time.Hour)}, polls, &sentinel, rt, def, from, to, 0) + if !res.Proved { + t.Fatalf("Proved = false, want true: KeepLast is a note, never fatal: %+v", res) + } + if !anyContains(res.Notes, "KeepLast") { + t.Fatalf("Notes = %v, want a KeepLast note (comma-joined membership, not a literal-key match)", res.Notes) + } +} + +// --- Clock domains (§16) --- + +// TestProveCoverage_SkewTranslationAtWindowBoundary pins §16's "Clock +// domains" rule: a constant clock skew on every poll must not itself read as +// a coverage gap or a from-before-record violation, because every +// cross-domain comparison translates by that poll's own skew first. +func TestProveCoverage_SkewTranslationAtWindowBoundary(t *testing.T) { + from := time.Date(2026, 1, 1, 0, 0, 0, 0, time.UTC) + to := from.Add(10 * time.Minute) + rt := newRuleTimings(30*time.Second, 60) + def := Definition{UID: "r1", Title: "R1"} + + const skew = 45 * time.Second // Grafana's clock reads 45s ahead of the runner's + const bound = 5 * time.Second + + var polls []Poll + for ts := from; !ts.After(to); ts = ts.Add(30 * time.Second) { + grafanaTime := ts.Add(skew) + polls = append(polls, Poll{ + RuleUID: "r1", GrafanaNow: grafanaTime, SkewMS: skew.Milliseconds(), SkewBoundMS: bound.Milliseconds(), + Found: true, Health: "ok", LastEvaluation: grafanaTime, + }) + } + sentinel := to + + res := proveCoverage(Header{StartedAt: from.Add(-time.Hour)}, polls, &sentinel, rt, def, from, to, 0) + if !res.Proved { + t.Fatalf("res = %+v, want proved: a constant clock skew must not itself read as a coverage gap (§16)", res) + } +} + +// --- Override round-trip (P5's "two authorities") --- + +// TestProveCoverage_OverrideRoundTrip is P7's other disproportionate done-gate +// test: it exercises DeriveTimingsFromLog and proveCoverage together, exactly +// as check will, to prove maxGap tracks the RECORDED cadence, never a +// re-derivation from the rule's own evaluation interval. +func TestProveCoverage_OverrideRoundTrip(t *testing.T) { + from := time.Date(2026, 1, 1, 0, 0, 0, 0, time.UTC) + + t.Run("slower override on a tighter rule classifies clean", func(t *testing.T) { + windowEnd := from.Add(10 * time.Minute) + h := Header{ + StartedAt: from.Add(-time.Hour), + Rules: []LoggedRule{{UID: "r1", Title: "R1", IntervalSeconds: 60, PollEverySeconds: 120}}, + } + defs := []Definition{{UID: "r1", Title: "R1", IntervalSeconds: 60}} + rt, _, err := DeriveTimingsFromLog(h, defs) + if err != nil { + t.Fatalf("DeriveTimingsFromLog: %v", err) + } + + var polls []Poll + for ts := from; !ts.After(windowEnd); ts = ts.Add(120 * time.Second) { + polls = append(polls, Poll{RuleUID: "r1", GrafanaNow: ts, Found: true, Health: "ok", LastEvaluation: ts}) + } + sentinel := windowEnd + + res := proveCoverage(h, polls, &sentinel, rt["r1"], defs[0], from, windowEnd, 0) + if !res.Proved { + t.Fatalf("Proved = false, want true (maxGap must come from the recorded 120s cadence, not the 30s default): %+v", res) + } + }) + + t.Run("faster override still catches a real recorder gap", func(t *testing.T) { + windowEnd := from.Add(20 * time.Minute) + h := Header{ + StartedAt: from.Add(-time.Hour), + Rules: []LoggedRule{{UID: "r1", Title: "R1", IntervalSeconds: 300, PollEverySeconds: 5}}, + } + defs := []Definition{{UID: "r1", Title: "R1", IntervalSeconds: 300}} + rt, _, err := DeriveTimingsFromLog(h, defs) + if err != nil { + t.Fatalf("DeriveTimingsFromLog: %v", err) + } + + var polls []Poll + ts := from + for range 20 { + polls = append(polls, Poll{RuleUID: "r1", GrafanaNow: ts, Found: true, Health: "ok", LastEvaluation: ts}) + ts = ts.Add(5 * time.Second) + } + // The one real gap: 250s, nowhere near this recording's actual 5s + // cadence. Resume 5s polling afterward all the way to windowEnd, so + // this hole is the ONLY gap in the window — otherwise an uncovered + // tail would exceed even the WRONG (definition-derived) 300s maxGap + // on its own, and the test could not tell the two derivations apart. + ts = ts.Add(250 * time.Second) + for !ts.After(windowEnd) { + polls = append(polls, Poll{RuleUID: "r1", GrafanaNow: ts, Found: true, Health: "ok", LastEvaluation: ts}) + ts = ts.Add(5 * time.Second) + } + sentinel := windowEnd + + res := proveCoverage(h, polls, &sentinel, rt["r1"], defs[0], from, windowEnd, 0) + if res.Reason != ReasonHeartbeatGap { + t.Fatalf("Reason = %q, want heartbeat_gap: if maxGap had been re-derived from the 300s definition instead of "+ + "the recorded 5s cadence, this 250s gap would pass silently — the fail-open direction P5 warns about", res.Reason) + } + }) +} + +func anyContains(notes []string, substr string) bool { + for _, n := range notes { + if strings.Contains(n, substr) { + return true + } + } + return false +} + +// --- Check 6, tightened: a corrupted log must not silently disable liveness --- + +// TestProveCoverage_ZeroLastEvaluationWithoutPauseIsStale guards check 6's +// skip condition. ReadLog does no field validation, so a log line can claim +// found:true, is_paused:false and still carry a zero LastEvaluation (a +// corrupted write, a hand-edited fixture, a future log format bug). That +// combination must read as maximally stale, not be waved through the way a +// legitimately paused poll's zero time is (§2.3) — the skip must key off +// IsPaused/Found, never off LastEvaluation being zero. +func TestProveCoverage_ZeroLastEvaluationWithoutPauseIsStale(t *testing.T) { + from := time.Date(2026, 1, 1, 0, 0, 0, 0, time.UTC) + to := from.Add(10 * time.Minute) + rt := newRuleTimings(30*time.Second, 60) + def := Definition{UID: "r1", Title: "R1"} + + polls := denseHealthyPolls("r1", from, to, 30*time.Second) + corruptAt := from.Add(5 * time.Minute) + for i := range polls { + if polls[i].GrafanaNow.Equal(corruptAt) { + polls[i].LastEvaluation = time.Time{} // found:true, is_paused:false, yet zero + } + } + sentinel := to + + res := proveCoverage(Header{StartedAt: from.Add(-time.Hour)}, polls, &sentinel, rt, def, from, to, 0) + if res.Reason != ReasonStaleEvaluation { + t.Fatalf("Reason = %q, want stale_evaluation: a zero lastEvaluation on a found, non-paused poll must fail "+ + "closed, not be silently skipped as if it were a legitimately paused observation", res.Reason) + } +} + +// --- Check 3, tightened: the boundary segments must widen by the skew bound --- + +// TestProveCoverage_BoundaryGapWidensBySkewBound pins §16's "with that +// poll's bound as the tolerance" for the two boundary segments specifically: +// a boundary gap that lands EXACTLY at maxGap must still fail once the +// poll's own skew bound is added, because the translation is only a best +// estimate and understating the gap by up to the bound would be fail-open. +func TestProveCoverage_BoundaryGapWidensBySkewBound(t *testing.T) { + from := time.Date(2026, 1, 1, 0, 0, 0, 0, time.UTC) + to := from.Add(10 * time.Minute) + rt := newRuleTimings(30*time.Second, 60) // maxGap = 60s + def := Definition{UID: "r1", Title: "R1"} + + const bound = 5 * time.Second + first := Poll{ + RuleUID: "r1", GrafanaNow: from.Add(rt.maxGap), Found: true, Health: "ok", + LastEvaluation: from.Add(rt.maxGap), SkewBoundMS: bound.Milliseconds(), + } + rest := denseHealthyPolls("r1", from.Add(rt.maxGap+30*time.Second), to, 30*time.Second) + polls := append([]Poll{first}, rest...) + sentinel := to + + res := proveCoverage(Header{StartedAt: from.Add(-time.Hour)}, polls, &sentinel, rt, def, from, to, 0) + if res.Reason != ReasonHeartbeatGap { + t.Fatalf("Reason = %q, want heartbeat_gap: the leading boundary segment sits at EXACTLY maxGap (60s) before "+ + "widening; the poll's own %s skew bound must push it past the threshold (§16), not just the skew translation", res.Reason, bound) + } +} + +// --- Multi-failure contract --- + +// TestProveCoverage_MultipleFailuresReasonIsFirstButAllNoted exercises two +// checks failing in the same rule: check 7 (paused in-window) precedes check +// 8 (rule absent) in the §5 order, so Reason must name the pause even though +// the rule also goes absent later — and the later failure must still add its +// own Note rather than being swallowed once Reason is set. +func TestProveCoverage_MultipleFailuresReasonIsFirstButAllNoted(t *testing.T) { + from := time.Date(2026, 1, 1, 0, 0, 0, 0, time.UTC) + to := from.Add(10 * time.Minute) + rt := newRuleTimings(30*time.Second, 60) + def := Definition{UID: "r1", Title: "R1"} + + polls := denseHealthyPolls("r1", from, to, 30*time.Second) + pausedAt := from.Add(3 * time.Minute) + absentAt := from.Add(6 * time.Minute) + for i := range polls { + switch { + case polls[i].GrafanaNow.Equal(pausedAt): + polls[i].IsPaused = true + polls[i].LastEvaluation = time.Time{} + case polls[i].GrafanaNow.Equal(absentAt): + polls[i].Found = false + polls[i].Health = "" + polls[i].LastEvaluation = time.Time{} + } + } + sentinel := to + + res := proveCoverage(Header{StartedAt: from.Add(-time.Hour)}, polls, &sentinel, rt, def, from, to, 0) + if res.Reason != ReasonPausedInWindow { + t.Fatalf("Reason = %q, want paused_in_window (the FIRST check to fail, in §5's order)", res.Reason) + } + if !anyContains(res.Notes, "paused") { + t.Fatalf("Notes = %v, want a note about the pause", res.Notes) + } + if !anyContains(res.Notes, "no rule") { + t.Fatalf("Notes = %v, want a note about the absence too — a later failure must still be recorded, "+ + "not swallowed once Reason is already set", res.Notes) + } +} + +// --- Skipped rules (P6/P8 obligation) --- + +// TestProveCoverage_SkippedRuleWithZeroPollsPinnedAsHeartbeatGap pins a known +// gap in this function's contract, not a bug in it: a rule paused BEFORE the +// window opened is never scheduled or polled (watch.go, §4.3), so it reaches +// proveCoverage with zero polls at all. proveCoverage has no notion of +// "skipped" — that classification belongs to the definitions +// (LoggedRule.IsPaused / Definition.IsPaused), never to the polls — so today +// it reports the whole window as one big heartbeat_gap instead. decide (P8) +// MUST read skipped status from the definitions and either skip calling this +// function for that rule entirely, or override this result — this test pins +// today's behavior so that review has something concrete to check against. +func TestProveCoverage_SkippedRuleWithZeroPollsPinnedAsHeartbeatGap(t *testing.T) { + from := time.Date(2026, 1, 1, 0, 0, 0, 0, time.UTC) + to := from.Add(10 * time.Minute) + rt := newRuleTimings(30*time.Second, 60) + def := Definition{UID: "r1", Title: "R1", IsPaused: true} + + sentinel := to + res := proveCoverage(Header{StartedAt: from.Add(-time.Hour)}, nil, &sentinel, rt, def, from, to, 0) + if res.Reason != ReasonHeartbeatGap { + t.Fatalf("Reason = %q, want heartbeat_gap (pinned, not the desired end state): proveCoverage has no "+ + "'skipped' concept, so decide (P8) must handle a skipped rule's classification itself, before or "+ + "instead of calling this function", res.Reason) + } +} From 2db19c7e219b64da13edf2db2590d58a2be2b886 Mon Sep 17 00:00:00 2001 From: Bartek Tofel Date: Wed, 2 Sep 2026 11:56:39 +0200 Subject: [PATCH 2/4] chore: enhance unit tests --- .../internal/gate/coverage_test.go | 27 +++++++++++++++++++ 1 file changed, 27 insertions(+) diff --git a/grafana-alertcheck/internal/gate/coverage_test.go b/grafana-alertcheck/internal/gate/coverage_test.go index 429139602..7ef4b5427 100644 --- a/grafana-alertcheck/internal/gate/coverage_test.go +++ b/grafana-alertcheck/internal/gate/coverage_test.go @@ -320,6 +320,33 @@ func TestProveCoverage_PausedInWindowIsUnobservable(t *testing.T) { } } +// TestProveCoverage_PausedAfterWindowIsFine pins check 7's respect for the +// window boundary: a poll that reports paused but lands beyond windowEnd (a +// rule paused only after THIS release window closed) is filtered out by +// inWindowPolls and must not fail the window. Without that filter, a pause in +// the next release's window would wrongly fail this one. +func TestProveCoverage_PausedAfterWindowIsFine(t *testing.T) { + from := time.Date(2026, 1, 1, 0, 0, 0, 0, time.UTC) + to := from.Add(10 * time.Minute) + rt := newRuleTimings(30*time.Second, 60) + def := Definition{UID: "r1", Title: "R1"} + + polls := denseHealthyPolls("r1", from, to, 30*time.Second) + polls = append(polls, Poll{ + RuleUID: "r1", GrafanaNow: to.Add(2 * time.Minute), + Found: true, Health: "ok", IsPaused: true, + }) + sentinel := to + + res := proveCoverage(Header{StartedAt: from.Add(-time.Hour)}, polls, &sentinel, rt, def, from, to, 0) + if res.Reason == ReasonPausedInWindow { + t.Fatalf("a paused poll after windowEnd tripped check 7: %+v", res.Notes) + } + if !res.Proved { + t.Fatalf("Proved = false, want a clean window: %+v", res.Notes) + } +} + // --- Check 8: rule absent (§14.5) --- func TestProveCoverage_RuleAbsentIsUnobservable(t *testing.T) { From 82dc549295c95eb2532fcb45d98e211a163e9cfe Mon Sep 17 00:00:00 2001 From: Bartek Tofel Date: Mon, 7 Sep 2026 16:36:05 +0200 Subject: [PATCH 3/4] chore: address code review comments --- grafana-alertcheck/internal/gate/coverage.go | 13 ++++++++++- .../internal/gate/coverage_test.go | 23 +++++++++++++++++++ .../internal/gate/parse_ruler_test.go | 5 +--- .../internal/gate/parse_state_test.go | 5 +--- .../internal/gate/schedule_test.go | 4 +--- 5 files changed, 38 insertions(+), 12 deletions(-) diff --git a/grafana-alertcheck/internal/gate/coverage.go b/grafana-alertcheck/internal/gate/coverage.go index 36d2c3dac..3d3704b90 100644 --- a/grafana-alertcheck/internal/gate/coverage.go +++ b/grafana-alertcheck/internal/gate/coverage.go @@ -38,6 +38,7 @@ const ( ReasonHeartbeatGap UnobservableReason = "heartbeat_gap" ReasonHealthError UnobservableReason = "health_error" ReasonStaleEvaluation UnobservableReason = "stale_evaluation" + ReasonFutureEvaluation UnobservableReason = "future_evaluation" ReasonPausedInWindow UnobservableReason = "paused_in_window" ReasonRuleAbsent UnobservableReason = "rule_absent" // ReasonDrainTimeout is set by check.go's drain wait (a later phase), @@ -181,6 +182,16 @@ func proveCoverage(h Header, polls []Poll, sentinel *time.Time, t ruleTimings, d if p.IsPaused || !p.Found { continue } + // lastEvaluation in the future of its own poll's grafana_now is + // corrupted or hand-edited data (ReadLog does no field validation); + // GrafanaNow-LastEvaluation would go negative and silently read as + // fresh — fail-open. Treat it as unobservable instead. + if p.LastEvaluation.After(p.GrafanaNow) { + fail(ReasonFutureEvaluation, fmt.Sprintf( + "lastEvaluation %s is after grafana_now %s (corrupted poll)", + p.LastEvaluation.Format(time.RFC3339), p.GrafanaNow.Format(time.RFC3339))) + continue + } if stale := p.GrafanaNow.Sub(p.LastEvaluation); stale > t.evalStaleAfter { staleCount++ if stale > worstStale { @@ -191,7 +202,7 @@ func proveCoverage(h Header, polls []Poll, sentinel *time.Time, t ruleTimings, d if staleCount > 0 { res.BlindFor = worstStale fail(ReasonStaleEvaluation, fmt.Sprintf( - "lastEvaluation stale on %d poll(s); worst %s (> evalStaleAfter %s) as of %s", + "lastEvaluation stale on %d poll(s); worst %s (> evalStaleAfter %s) as of grafana_now %s", staleCount, worstStale, t.evalStaleAfter, worstStaleAt.Format(time.RFC3339))) } diff --git a/grafana-alertcheck/internal/gate/coverage_test.go b/grafana-alertcheck/internal/gate/coverage_test.go index 7ef4b5427..6419a8b2e 100644 --- a/grafana-alertcheck/internal/gate/coverage_test.go +++ b/grafana-alertcheck/internal/gate/coverage_test.go @@ -556,6 +556,29 @@ func TestProveCoverage_ZeroLastEvaluationWithoutPauseIsStale(t *testing.T) { } } +// A lastEvaluation in the future of grafana_now (corrupted log) must fail closed. +func TestProveCoverage_FutureLastEvaluationIsUnobservable(t *testing.T) { + from := time.Date(2026, 1, 1, 0, 0, 0, 0, time.UTC) + to := from.Add(10 * time.Minute) + rt := newRuleTimings(30*time.Second, 60) + def := Definition{UID: "r1", Title: "R1"} + + polls := denseHealthyPolls("r1", from, to, 30*time.Second) + corruptAt := from.Add(5 * time.Minute) + for i := range polls { + if polls[i].GrafanaNow.Equal(corruptAt) { + polls[i].LastEvaluation = corruptAt.Add(2 * time.Minute) // in the future of its own grafana_now + } + } + sentinel := to + + res := proveCoverage(Header{StartedAt: from.Add(-time.Hour)}, polls, &sentinel, rt, def, from, to, 0) + if res.Reason != ReasonFutureEvaluation { + t.Fatalf("Reason = %q, want future_evaluation: a lastEvaluation in the future of grafana_now must fail "+ + "closed rather than read its negative staleness as fresh", res.Reason) + } +} + // --- Check 3, tightened: the boundary segments must widen by the skew bound --- // TestProveCoverage_BoundaryGapWidensBySkewBound pins §16's "with that diff --git a/grafana-alertcheck/internal/gate/parse_ruler_test.go b/grafana-alertcheck/internal/gate/parse_ruler_test.go index dda66b77e..3cf2e3261 100644 --- a/grafana-alertcheck/internal/gate/parse_ruler_test.go +++ b/grafana-alertcheck/internal/gate/parse_ruler_test.go @@ -115,10 +115,7 @@ func TestParseDefinitions_Recording(t *testing.T) { } } -// A datasource-managed rule with neither an "alert" nor a "record" name has no -// identity (its only name is the Prometheus rule name), and an empty Title -// would make P3's refusal-by-name unreachable. It must fail parsing, not hand -// back a silently unusable Definition. +// A datasource-managed rule with no alert/record name must fail parsing. func TestParseDefinitions_DatasourceManagedNoName(t *testing.T) { body := []byte(`{"ExampleMetrics":[{"name":"g","rules":[{"expr":"up == 0","for":"5m"}]}]}`) if _, err := ParseDefinitions(body); err == nil { diff --git a/grafana-alertcheck/internal/gate/parse_state_test.go b/grafana-alertcheck/internal/gate/parse_state_test.go index e370c0f5e..8fca6e7ab 100644 --- a/grafana-alertcheck/internal/gate/parse_state_test.go +++ b/grafana-alertcheck/internal/gate/parse_state_test.go @@ -283,10 +283,7 @@ func TestInstanceKey(t *testing.T) { } } -// TestInstanceKey_NoCollision guards against ambiguous identities: label -// values may legally contain "\n" or "=", and a naive "k=v\n" join would -// collide e.g. {a:"1\nb=2"} with {a:"1",b:"2"}. The JSON encoding must keep -// such sets distinct. +// Label values may contain "\n" or "="; the JSON encoding must keep them distinct. func TestInstanceKey_NoCollision(t *testing.T) { if instanceKey(map[string]string{"a": "1\nb=2"}) == instanceKey(map[string]string{"a": "1", "b": "2"}) { t.Errorf("instanceKey collided for sets {a:1\\nb=2} and {a:1,b:2}") diff --git a/grafana-alertcheck/internal/gate/schedule_test.go b/grafana-alertcheck/internal/gate/schedule_test.go index 9985a0bd3..e897414bc 100644 --- a/grafana-alertcheck/internal/gate/schedule_test.go +++ b/grafana-alertcheck/internal/gate/schedule_test.go @@ -230,9 +230,7 @@ func TestScheduler_EarliestDueEmpty(t *testing.T) { } } -// TestScheduler_EarliestDueZeroTime pins the empty-detection fix: a -// non-empty scheduler whose earliest next-due time is the zero time must still -// report ok=true. The old IsZero() sentinel misread exactly this as "no rules". +// A zero next-due time is real, not an empty scheduler. func TestScheduler_EarliestDueZeroTime(t *testing.T) { s := &Scheduler{ next: map[string]time.Time{"r1": {}}, From c77502a196f351d8a65e9836346ef37b426d663d Mon Sep 17 00:00:00 2001 From: Bartek Tofel Date: Wed, 9 Sep 2026 11:59:16 +0200 Subject: [PATCH 4/4] chore: implement phase 8 (#2786) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit * chore: implement phase 8 Invariant defended: H6/H7. The one question: can a violation ever outrank an unobservable rule, or can a pass happen without Violations empty and err nil? Adds classify.go: the pure per-instance classifier (outcome table, preexisting policy, BadFor) and decide(), the seam combining proveCoverage with those timelines under one Policy. Consolidates rule-poll filtering and skew translation onto pollsForRule/runnerTime, shared with coverage.go. * chore: rename some vars + add unit tests * chore: address code review comments * chore: implement phase 9 (#2787) * chore: implement phase 9 Invariant defended: H5/H7. The one question: can check report a pass over a window it did not prove? Check() is the I/O shell around the pure decide(). Single-step synthesizes the header and its own sentinel, so no mode flag reaches the pure layer. Log mode stops the recorder before the one full read. The header, not a definition re-resolved after the window closed, is the authority for what was paused when the window opened — it decides `skipped`, the drain set, and the transitionGrace max. The flock, not the pidfile, is the authority for whether a writer still exists. * chore: remove unix build tag * chore: implement phase 10 (#2788) * Wire watch/check subcommands to the gate library, with a table+JSON renderer and H6/H7 exit-code mapping. Extend Result with per-rule/global thresholds and a real skew bound; export SkewHardLimit; reject --states normal. * chore: fix goreleaser.yaml and add version command * chore: implement phase 11 (#2789) * chore: implement phase 11 Add coverage.go's declared-KeepLast check (no_data_state/exec_err_state, not just an observed reason) and close the remaining §22 gaps: newly_bad's no-early-exit clock assertion, a recorder-mode gap right after the deploy, a rule's own coverage gap overriding its own recovery, a genuinely skew-discriminating staleness test, exit-2 consequences on two Reason-only coverage tests, and end-to-end checks for log-name collapse, a truncated log, and the real watch-written state histogram. * chore: address code review comments * chore: more concise comments (#2790) * chore: more concise comments * fix: merge conflict * chore: shorten comments * fix: resolve conflict * chore: use testify's require in tests (#2792) * chore: use testify's require in tests * chore: move remaining assumptions to testify * chore: address code review comments * chore: fix logging and std out printing (#2798) * chore: fix logging and std out printing * chore: truncate to seconds when comparing from time * chore: add centralized docs (#2802) * chore: add centralized docs * chore: further update docs * chore: address code review comments (#2807) * chore: address code review comments * chore: get rid of goreleaser --- .../workflows/grafana-alertcheck-release.yml | 34 - grafana-alertcheck/.goreleaser.yaml | 33 - grafana-alertcheck/README.md | 46 +- grafana-alertcheck/cmd/check.go | 154 ++ grafana-alertcheck/cmd/check_test.go | 130 ++ grafana-alertcheck/cmd/common.go | 107 ++ .../cmd/{grafana-alertcheck => }/env.go | 3 +- .../cmd/grafana-alertcheck/main_test.go | 60 - .../cmd/{grafana-alertcheck => }/list.go | 11 +- .../cmd/{grafana-alertcheck => }/list_test.go | 31 +- .../cmd/{grafana-alertcheck => }/main.go | 19 +- grafana-alertcheck/cmd/main_test.go | 43 + grafana-alertcheck/cmd/style.go | 125 ++ grafana-alertcheck/cmd/table.go | 173 +++ grafana-alertcheck/cmd/table_test.go | 88 ++ grafana-alertcheck/cmd/watch.go | 153 ++ grafana-alertcheck/cmd/watch_test.go | 82 ++ grafana-alertcheck/docs/_category_.yaml | 8 + grafana-alertcheck/docs/advanced.md | 41 + grafana-alertcheck/docs/architecture.md | 68 + .../docs/how-alerts-are-evaluated.md | 85 ++ grafana-alertcheck/docs/index.md | 87 ++ .../docs/reference/_category_.yaml | 8 + grafana-alertcheck/docs/reference/cli.md | 91 ++ .../docs/reference/log-format.md | 98 ++ grafana-alertcheck/go.mod | 4 + grafana-alertcheck/go.sum | 4 + grafana-alertcheck/internal/gate/check.go | 878 ++++++++++++ .../internal/gate/check_process.go | 29 + .../internal/gate/check_test.go | 1268 +++++++++++++++++ grafana-alertcheck/internal/gate/classify.go | 610 ++++++++ .../internal/gate/classify_test.go | 992 +++++++++++++ grafana-alertcheck/internal/gate/coverage.go | 235 ++- .../internal/gate/coverage_test.go | 371 ++--- grafana-alertcheck/internal/gate/duration.go | 2 +- .../internal/gate/duration_test.go | 16 +- grafana-alertcheck/internal/gate/flock.go | 22 +- .../internal/gate/flock_test.go | 18 +- grafana-alertcheck/internal/gate/jsonreq.go | 2 +- .../internal/gate/jsonreq_test.go | 54 +- grafana-alertcheck/internal/gate/log.go | 277 ++-- grafana-alertcheck/internal/gate/log_test.go | 510 ++----- .../internal/gate/parse_ruler.go | 26 +- .../internal/gate/parse_ruler_test.go | 108 +- .../internal/gate/parse_state.go | 39 +- .../internal/gate/parse_state_test.go | 299 ++-- grafana-alertcheck/internal/gate/resolve.go | 47 +- .../internal/gate/resolve_test.go | 232 +-- grafana-alertcheck/internal/gate/schedule.go | 242 ++-- .../internal/gate/schedule_test.go | 314 ++-- grafana-alertcheck/internal/gate/source.go | 132 +- .../internal/gate/source_fake_test.go | 27 +- .../internal/gate/source_test.go | 324 ++--- .../internal/gate/testdata/README.md | 45 +- grafana-alertcheck/internal/gate/watch.go | 260 ++-- .../internal/gate/watch_daemon_test.go | 176 ++- .../internal/gate/watch_process.go | 10 +- .../internal/gate/watch_test.go | 396 ++--- 58 files changed, 7179 insertions(+), 2568 deletions(-) delete mode 100644 .github/workflows/grafana-alertcheck-release.yml delete mode 100644 grafana-alertcheck/.goreleaser.yaml create mode 100644 grafana-alertcheck/cmd/check.go create mode 100644 grafana-alertcheck/cmd/check_test.go create mode 100644 grafana-alertcheck/cmd/common.go rename grafana-alertcheck/cmd/{grafana-alertcheck => }/env.go (96%) delete mode 100644 grafana-alertcheck/cmd/grafana-alertcheck/main_test.go rename grafana-alertcheck/cmd/{grafana-alertcheck => }/list.go (86%) rename grafana-alertcheck/cmd/{grafana-alertcheck => }/list_test.go (73%) rename grafana-alertcheck/cmd/{grafana-alertcheck => }/main.go (65%) create mode 100644 grafana-alertcheck/cmd/main_test.go create mode 100644 grafana-alertcheck/cmd/style.go create mode 100644 grafana-alertcheck/cmd/table.go create mode 100644 grafana-alertcheck/cmd/table_test.go create mode 100644 grafana-alertcheck/cmd/watch.go create mode 100644 grafana-alertcheck/cmd/watch_test.go create mode 100644 grafana-alertcheck/docs/_category_.yaml create mode 100644 grafana-alertcheck/docs/advanced.md create mode 100644 grafana-alertcheck/docs/architecture.md create mode 100644 grafana-alertcheck/docs/how-alerts-are-evaluated.md create mode 100644 grafana-alertcheck/docs/index.md create mode 100644 grafana-alertcheck/docs/reference/_category_.yaml create mode 100644 grafana-alertcheck/docs/reference/cli.md create mode 100644 grafana-alertcheck/docs/reference/log-format.md create mode 100644 grafana-alertcheck/go.sum create mode 100644 grafana-alertcheck/internal/gate/check.go create mode 100644 grafana-alertcheck/internal/gate/check_process.go create mode 100644 grafana-alertcheck/internal/gate/check_test.go create mode 100644 grafana-alertcheck/internal/gate/classify.go create mode 100644 grafana-alertcheck/internal/gate/classify_test.go diff --git a/.github/workflows/grafana-alertcheck-release.yml b/.github/workflows/grafana-alertcheck-release.yml deleted file mode 100644 index 4f946b18f..000000000 --- a/.github/workflows/grafana-alertcheck-release.yml +++ /dev/null @@ -1,34 +0,0 @@ -name: Grafana Alertcheck Release - -on: - push: - tags: - - grafana-alertcheck/v* - -jobs: - release: - name: Build and Release - runs-on: ubuntu-latest - environment: integration - permissions: - id-token: write - contents: write - steps: - - name: Checkout repo - uses: actions/checkout@v7 - with: - fetch-depth: 0 - - name: Set up Go - uses: actions/setup-go@v7 - with: - go-version-file: ./grafana-alertcheck/go.mod - cache-dependency-path: ./grafana-alertcheck/go.mod - - name: Goreleaser Release - uses: goreleaser/goreleaser-action@f06c13b6b1a9625abc9e6e439d9c05a8f2190e94 # v7.2.3 - with: - distribution: goreleaser-pro - version: "~> v2" - args: release --clean -f ./grafana-alertcheck/.goreleaser.yaml - env: - GITHUB_TOKEN: ${{ secrets.GITHUB_TOKEN }} - GORELEASER_KEY: ${{ secrets.GORELEASER_KEY }} diff --git a/grafana-alertcheck/.goreleaser.yaml b/grafana-alertcheck/.goreleaser.yaml deleted file mode 100644 index 09e6032cb..000000000 --- a/grafana-alertcheck/.goreleaser.yaml +++ /dev/null @@ -1,33 +0,0 @@ -# yaml-language-server: $schema=https://goreleaser.com/static/schema-pro.json -version: 2 -project_name: grafana-alertcheck - -dist: grafana-alertcheck/dist - -monorepo: - tag_prefix: grafana-alertcheck/ - dir: grafana-alertcheck - -builds: - - id: grafana-alertcheck - main: ./cmd/grafana-alertcheck/main.go - ldflags: - - -s - - -w - - -X github.com/smartcontractkit/chainlink-testing-framework/grafana-alertcheck.version={{.Version}} - - -X github.com/smartcontractkit/chainlink-testing-framework/grafana-alertcheck.commit={{.ShortCommit}} - - -X github.com/smartcontractkit/chainlink-testing-framework/grafana-alertcheck.date={{.CommitDate}} - - -X github.com/smartcontractkit/chainlink-testing-framework/grafana-alertcheck.builtBy=goreleaser - goos: - - linux - - darwin - goarch: - - amd64 - - arm64 - binary: grafana-alertcheck - env: - - CGO_ENABLED=0 - -before: - hooks: - - sh -c "cd grafana-alertcheck && go mod tidy" diff --git a/grafana-alertcheck/README.md b/grafana-alertcheck/README.md index 43cc03fc4..96aa0e999 100644 --- a/grafana-alertcheck/README.md +++ b/grafana-alertcheck/README.md @@ -1,6 +1,46 @@ # grafana-alertcheck -A CD quality gate for Grafana alerts: bookend a release with `watch` (record) and `check` (classify) to -answer whether any watched alert was in a bad state during the release window. +A CD quality gate for Grafana alerts. It bookends a release with two commands — `watch` (record) and +`check` (classify) — and answers whether any watched alert was in a bad state during the release window. -Under construction. +``` +watch → your work → check +``` + +`watch` starts a background recorder that polls each named alert into a JSONL log. After the work emits a +`from`/`to` pair, `check` proves continuous coverage of that window, classifies each alert's state +timeline, and exits `0`, `1`, or `2`. + +It **fails closed**: if it cannot get an answer, it stops the release — never a pass on an unproven window. + +## Quickstart + +```bash +export GRAFANA_URL=https://grafana.example.com +export GRAFANA_TOKEN=… + +grafana-alertcheck watch --out /tmp/run.jsonl --alerts alerts.txt +./deploy.sh # emits deployed_at= +./verify.sh # emits finished_at= +grafana-alertcheck check --in /tmp/run.jsonl --from "$deployed_at" --to "$finished_at" +``` + +Requires Grafana >= 13.0.0 and < 14.0.0. Connection details come from the environment only — the token is +never a flag. + +## Documentation + +| Doc | Covers | +| --- | ------ | +| [`docs/index.md`](./docs/index.md) | Overview, quickstarts, exit codes, common surprises | +| [`docs/how-alerts-are-evaluated.md`](./docs/how-alerts-are-evaluated.md) | Verdict model, coverage proof, health/liveness | +| [`docs/advanced.md`](./docs/advanced.md) | Check budget, scheduling, why history isn't queried | +| [`docs/architecture.md`](./docs/architecture.md) | Design invariants, the pure-function seam, recorder lifecycle | +| [`docs/reference/cli.md`](./docs/reference/cli.md) | Full CLI reference — subcommands, flags, naming | +| [`docs/reference/log-format.md`](./docs/reference/log-format.md) | The JSONL log schema, for debugging artifacts | + +## Build + +```bash +go build ./... && go test ./... +``` diff --git a/grafana-alertcheck/cmd/check.go b/grafana-alertcheck/cmd/check.go new file mode 100644 index 000000000..e95fce376 --- /dev/null +++ b/grafana-alertcheck/cmd/check.go @@ -0,0 +1,154 @@ +package main + +import ( + "context" + "encoding/json" + "errors" + "flag" + "fmt" + "io" + "os/signal" + "syscall" + "time" + + "github.com/smartcontractkit/chainlink-testing-framework/grafana-alertcheck/internal/gate" +) + +const checkUsage = "usage: grafana-alertcheck check [--in ] [--pidfile F] --from RFC3339 --to RFC3339 " + + "[--alerts ...] [--folder F] [--states ...] [--preexisting ...] [--min-observed N] [--allow-paused] " + + "[--nodata-is-unobservable] [--concurrency N] [--output json]" + +// runCheck is the classify step's CLI surface: parse flags into a gate.Config, +// run gate.Check, and translate its (Result, error) into output and an exit +// code. All of the correctness lives in the gate package — this file's only job +// is presentation and the exit-code mapping, which exitCode below keeps as one +// pure function so it can be tested without a network. +func runCheck(args []string, stdin io.Reader, stdout, stderr io.Writer) int { + fs := flag.NewFlagSet("check", flag.ContinueOnError) + fs.SetOutput(stderr) + fs.Usage = func() { fmt.Fprintln(stderr, checkUsage) } + + common := registerCommon(fs) + in := fs.String("in", "", "path of a log recorded by watch; empty selects single-step mode") + pidfile := fs.String("pidfile", "", "pidfile of the recorder to stop before reading --in (default .pid)") + from := fs.String("from", "", "the moment the deploy finished, RFC3339 (required with --in)") + to := fs.String("to", "", "the end of the window to classify, RFC3339 (required)") + states := fs.String("states", "", "comma-separated bad states to classify against (default: firing)") + preexisting := fs.String("preexisting", "", "how to judge an instance already bad at `from` (default: fail-unless-recovered)") + minObserved := fs.Int("min-observed", 0, "minimum rules that must be observed (default: every resolved rule)") + allowPaused := fs.Bool("allow-paused", false, "do not count a rule paused before the window against --min-observed") + nodataIsUnobservable := fs.Bool("nodata-is-unobservable", false, "treat a sustained health=nodata as unobservable rather than a note") + output := fs.String("output", "", `"json" writes the machine-readable Result to stdout in addition to the table; default is the table alone`) + + if err := fs.Parse(args); err != nil { + if errors.Is(err, flag.ErrHelp) { + return 0 + } + return 2 + } + if fs.NArg() != 0 { + fmt.Fprintf(stderr, "check: unexpected arguments %v\n", fs.Args()) + return 2 + } + if *output != "" && *output != "json" { + fmt.Fprintf(stderr, "--output: unknown value %q (only \"json\" is supported)\n", *output) + return 2 + } + + url, token, err := grafanaEnv() + if err != nil { + fmt.Fprintln(stderr, err) + return 2 + } + alerts, err := readAlerts(stdin, *common.alerts) + if err != nil { + fmt.Fprintln(stderr, err) + return 2 + } + stateList, err := parseStates(*states) + if err != nil { + fmt.Fprintln(stderr, err) + return 2 + } + preexistingPolicy, err := parsePreexisting(*preexisting) + if err != nil { + fmt.Fprintln(stderr, err) + return 2 + } + + cfg := gate.Config{ + URL: url, + Token: token, + Alerts: alerts, + Folder: *common.folder, + States: stateList, + Preexisting: preexistingPolicy, + MinObserved: *minObserved, + AllowPaused: *allowPaused, + NodataIsUnobservable: *nodataIsUnobservable, + Log: *in, + PidFile: *pidfile, + Concurrency: *common.concurrency, + Clock: gate.SystemClock{}, + Notes: newNoteStyler(stderr), + } + if *to == "" { + fmt.Fprintln(stderr, "check: --to is required") + return 2 + } + t, err := time.Parse(time.RFC3339, *to) + if err != nil { + fmt.Fprintf(stderr, "--to: %v\n", err) + return 2 + } + cfg.To = t + if *from != "" { + f, err := time.Parse(time.RFC3339, *from) + if err != nil { + fmt.Fprintf(stderr, "--from: %v\n", err) + return 2 + } + cfg.From = f + } + + // SIGINT/SIGTERM cancel the run cleanly rather than leaving an operator's + // Ctrl-C to kill the process mid-collection: Check's collection loop and + // drain wait both already select on ctx.Done() (check.go), so this makes + // an interrupted run fail the way every other could-not-check path does + // — exit 2, never a silently truncated pass. + ctx, stop := signal.NotifyContext(context.Background(), syscall.SIGINT, syscall.SIGTERM) + defer stop() + + result, checkErr := gate.Check(ctx, cfg) + + if checkErr != nil { + fmt.Fprintln(stderr, checkErr) + } else if err := renderTable(stderr, result); err != nil { + fmt.Fprintln(stderr, err) + } + if *output == "json" { + enc := json.NewEncoder(stdout) + enc.SetIndent("", " ") + if err := enc.Encode(result); err != nil { + fmt.Fprintf(stderr, "encode --output json: %v\n", err) + return 2 + } + } + return exitCode(result, checkErr) +} + +// exitCode is the whole exit-code mapping, kept as one pure function of +// exactly what Check returns so it is testable without a network: err != nil +// is exit 2 UNCONDITIONALLY — never 0 and never 1, even alongside real +// violations, because an inability to check beats a violation and an error is +// never a pass. Violations without an error is exit 1. Neither is exit 0. +func exitCode(res gate.Result, err error) int { + switch { + case err != nil: + return 2 + case len(res.Violations) > 0: + return 1 + default: + return 0 + } +} diff --git a/grafana-alertcheck/cmd/check_test.go b/grafana-alertcheck/cmd/check_test.go new file mode 100644 index 000000000..c7c797a0b --- /dev/null +++ b/grafana-alertcheck/cmd/check_test.go @@ -0,0 +1,130 @@ +package main + +import ( + "bytes" + "errors" + "os" + "testing" + + "github.com/smartcontractkit/chainlink-testing-framework/grafana-alertcheck/internal/gate" + "github.com/stretchr/testify/require" +) + +// The exit-code mapping, pinned directly against exitCode with no network +// involved: err != nil is exit 2 even alongside violations (an inability to +// check beats a violation), violations alone are exit 1, and neither is 0. +func TestExitCode(t *testing.T) { + tests := []struct { + name string + res gate.Result + err error + want int + }{ + {"pass", gate.Result{}, nil, 0}, + {"violation", gate.Result{Violations: []gate.Violation{{}}}, nil, 1}, + {"error alone", gate.Result{}, errors.New("boom"), 2}, + {"error beats violation", gate.Result{Violations: []gate.Violation{{}}}, errors.New("boom"), 2}, + } + for _, tt := range tests { + t.Run(tt.name, func(t *testing.T) { + require.Equal(t, tt.want, exitCode(tt.res, tt.err)) + }) + } +} + +func writeTempAlerts(t *testing.T) string { + t.Helper() + path := t.TempDir() + "/alerts.txt" + require.NoError(t, os.WriteFile(path, []byte("Some Alert\n"), 0o644)) + return path +} + +// The flag-validation matrix: every one of these must fail before any network +// call, because gate.Config.validate() runs first — an unreachable GRAFANA_URL +// succeeding or timing out is a different test than these, which check pure +// input validation. +func TestRunCheck_FlagValidation(t *testing.T) { + tests := []struct { + name string + env bool + args func(t *testing.T) []string + wantErr string + }{ + {"missing env", false, func(t *testing.T) []string { + return []string{"--to", "2026-01-01T00:00:00Z", "--alerts", writeTempAlerts(t)} + }, "GRAFANA_URL"}, + {"missing to", true, func(t *testing.T) []string { + return []string{"--alerts", writeTempAlerts(t)} + }, "--to"}, + {"bad to", true, func(t *testing.T) []string { + return []string{"--to", "not-a-time", "--alerts", writeTempAlerts(t)} + }, "--to"}, + {"bad from", true, func(t *testing.T) []string { + return []string{"--to", "2026-01-01T00:00:00Z", "--from", "not-a-time", "--alerts", writeTempAlerts(t)} + }, "--from"}, + {"bad output", true, func(t *testing.T) []string { + return []string{"--to", "2026-01-01T00:00:00Z", "--output", "xml", "--alerts", writeTempAlerts(t)} + }, "--output"}, + {"bad states", true, func(t *testing.T) []string { + return []string{"--to", "2026-01-01T00:00:00Z", "--states", "bogus", "--alerts", writeTempAlerts(t)} + }, "--states"}, + {"states normal is rejected", true, func(t *testing.T) []string { + // normal is the good state, never a state to classify AS bad: + // accepting it would make --states normal fail every healthy + // instance. + return []string{"--to", "2026-01-01T00:00:00Z", "--states", "normal", "--alerts", writeTempAlerts(t)} + }, "--states"}, + {"bad preexisting", true, func(t *testing.T) []string { + return []string{"--to", "2026-01-01T00:00:00Z", "--preexisting", "bogus", "--alerts", writeTempAlerts(t)} + }, "--preexisting"}, + {"alerts with in", true, func(t *testing.T) []string { + return []string{"--to", "2026-01-01T00:00:00Z", "--in", "some.jsonl", "--alerts", writeTempAlerts(t)} + }, "refused"}, + {"no alerts no in", true, func(t *testing.T) []string { + return []string{"--to", "2026-01-01T00:00:00Z"} + }, "no alert names"}, + } + for _, tt := range tests { + t.Run(tt.name, func(t *testing.T) { + if tt.env { + t.Setenv("GRAFANA_URL", "http://example.invalid") + t.Setenv("GRAFANA_TOKEN", "test-token") + } else { + t.Setenv("GRAFANA_URL", "") + t.Setenv("GRAFANA_TOKEN", "") + } + var stdout, stderr bytes.Buffer + args := append([]string{"check"}, tt.args(t)...) + code := run(args, &stdout, &stderr) + require.Equal(t, 2, code) + require.Contains(t, stderr.String(), tt.wantErr) + }) + } +} + +// A `to` already in the past with no recorded log cannot be classified from +// anything, because nothing ever observed the window. +func TestRunCheck_ToInPastNoLog(t *testing.T) { + t.Setenv("GRAFANA_URL", "http://example.invalid") + t.Setenv("GRAFANA_TOKEN", "test-token") + + var stdout, stderr bytes.Buffer + code := run([]string{"check", + "--from", "1999-01-01T00:00:00Z", "--to", "2000-01-01T00:00:00Z", + "--alerts", writeTempAlerts(t), + }, &stdout, &stderr) + require.Equal(t, 2, code) + require.Contains(t, stderr.String(), "already passed") +} + +// --output json never writes to stdout when Check was never reached, because +// there is no Result to encode — only the table (on stderr) can report a +// configuration failure. +func TestRunCheck_NoResultOnConfigError(t *testing.T) { + t.Setenv("GRAFANA_URL", "") + t.Setenv("GRAFANA_TOKEN", "") + var stdout, stderr bytes.Buffer + code := run([]string{"check", "--to", "2026-01-01T00:00:00Z", "--output", "json"}, &stdout, &stderr) + require.Equal(t, 2, code) + require.Empty(t, stdout.String()) +} diff --git a/grafana-alertcheck/cmd/common.go b/grafana-alertcheck/cmd/common.go new file mode 100644 index 000000000..3c56cd63f --- /dev/null +++ b/grafana-alertcheck/cmd/common.go @@ -0,0 +1,107 @@ +package main + +import ( + "bufio" + "flag" + "fmt" + "io" + "os" + "strings" + + "github.com/smartcontractkit/chainlink-testing-framework/grafana-alertcheck/internal/gate" +) + +// commonFlags is registerCommon's result: the exactly three flags watch and +// check share. Connection details are never flags, and states / poll-interval +// are deliberately NOT here — states is check-only because recording is +// unfiltered, and poll-interval is watch-only because check reads the cadence +// from the log header. Putting either here would give both commands an opinion +// about a value only one of them may set. +type commonFlags struct { + folder *string + concurrency *int + alerts *string +} + +func registerCommon(fs *flag.FlagSet) *commonFlags { + return &commonFlags{ + folder: fs.String("folder", "", "default folder to scope an unqualified alert name to"), + concurrency: fs.Int("concurrency", 1, "maximum concurrent requests to Grafana"), + alerts: fs.String("alerts", "", "path to a file of alert names, one per line, or - for stdin"), + } +} + +// readAlerts reads alert names, one per line, from a file or from +// stdin when path is "-". An empty path is not an error here — watch and +// check each decide for themselves whether an empty list is allowed +// (log mode never wants one; single-step / record mode always does). +func readAlerts(stdin io.Reader, path string) ([]string, error) { + if path == "" { + return nil, nil + } + var r io.Reader + if path == "-" { + r = stdin + } else { + f, err := os.Open(path) + if err != nil { + return nil, fmt.Errorf("read --alerts %s: %w", path, err) + } + defer f.Close() + r = f + } + var lines []string + sc := bufio.NewScanner(r) + for sc.Scan() { + lines = append(lines, sc.Text()) + } + if err := sc.Err(); err != nil { + return nil, fmt.Errorf("read --alerts %s: %w", path, err) + } + return lines, nil +} + +// parseStates parses check's --states flag: a comma-separated list of the +// "bad" state vocabulary Config.States matches against (classify.go's +// badStateSet). An empty string is not resolved here — it means "use the +// library default of {firing}" — so this returns nil, nil for "" rather than +// an error. +// +// normal is deliberately NOT accepted. The vocabulary is fixed to +// firing | pending | nodata | error precisely because "normal" is the good +// state, never a bad one to classify against: --states normal would turn every +// healthy instance into a violation and fail every healthy fleet. +func parseStates(s string) ([]gate.State, error) { + if strings.TrimSpace(s) == "" { + return nil, nil + } + var out []gate.State + for _, part := range strings.Split(s, ",") { + part = strings.TrimSpace(part) + if part == "" { + continue + } + switch gate.State(part) { + case gate.StateFiring, gate.StatePending, gate.StateNodata, gate.StateError: + out = append(out, gate.State(part)) + default: + return nil, fmt.Errorf("--states: unknown state %q (want any of: firing, pending, nodata, error)", part) + } + } + if len(out) == 0 { + return nil, fmt.Errorf("--states: %q named no state", s) + } + return out, nil +} + +// parsePreexisting parses check's --preexisting flag. +func parsePreexisting(s string) (gate.PreexistingPolicy, error) { + switch gate.PreexistingPolicy(s) { + case "": + return gate.PreexistingFailUnlessRecovered, nil + case gate.PreexistingFailUnlessRecovered, gate.PreexistingFail, gate.PreexistingIgnore: + return gate.PreexistingPolicy(s), nil + default: + return "", fmt.Errorf("--preexisting: unknown policy %q (want one of: fail-unless-recovered, fail, ignore)", s) + } +} diff --git a/grafana-alertcheck/cmd/grafana-alertcheck/env.go b/grafana-alertcheck/cmd/env.go similarity index 96% rename from grafana-alertcheck/cmd/grafana-alertcheck/env.go rename to grafana-alertcheck/cmd/env.go index e5702a6d1..02a125f1d 100644 --- a/grafana-alertcheck/cmd/grafana-alertcheck/env.go +++ b/grafana-alertcheck/cmd/env.go @@ -7,8 +7,7 @@ import ( // grafanaEnv reads the connection details from the environment only, never // from a flag — a flag value lands in the process argv and in CI logs, and -// the token must never be logged or otherwise surface in an error string -// (§20.2). +// the token must never be logged or otherwise surface in an error string. func grafanaEnv() (url, token string, err error) { url = os.Getenv("GRAFANA_URL") if url == "" { diff --git a/grafana-alertcheck/cmd/grafana-alertcheck/main_test.go b/grafana-alertcheck/cmd/grafana-alertcheck/main_test.go deleted file mode 100644 index 1d7906674..000000000 --- a/grafana-alertcheck/cmd/grafana-alertcheck/main_test.go +++ /dev/null @@ -1,60 +0,0 @@ -package main - -import ( - "bytes" - "strings" - "testing" -) - -func TestRun_NoArgs(t *testing.T) { - var stdout, stderr bytes.Buffer - code := run(nil, &stdout, &stderr) - if code != 2 { - t.Fatalf("code = %d, want 2", code) - } - if !strings.Contains(stderr.String(), "usage") { - t.Fatalf("stderr = %q, want a usage message", stderr.String()) - } -} - -func TestRun_Help(t *testing.T) { - for _, flag := range []string{"-h", "-help", "--help"} { - t.Run(flag, func(t *testing.T) { - var stdout, stderr bytes.Buffer - code := run([]string{flag}, &stdout, &stderr) - if code != 0 { - t.Fatalf("code = %d, want 0 (requested help is not a could-not-check condition)", code) - } - if !strings.Contains(stdout.String(), "usage") { - t.Fatalf("stdout = %q, want a usage message", stdout.String()) - } - if stderr.String() != "" { - t.Fatalf("stderr = %q, want empty — help goes to stdout", stderr.String()) - } - }) - } -} - -func TestRun_UnknownSubcommand(t *testing.T) { - var stdout, stderr bytes.Buffer - code := run([]string{"bogus"}, &stdout, &stderr) - if code != 2 { - t.Fatalf("code = %d, want 2", code) - } - if !strings.Contains(stderr.String(), `"bogus"`) { - t.Fatalf("stderr = %q, want it to name the unknown subcommand", stderr.String()) - } -} - -func TestRun_List_MissingEnv(t *testing.T) { - t.Setenv("GRAFANA_URL", "") - t.Setenv("GRAFANA_TOKEN", "") - var stdout, stderr bytes.Buffer - code := run([]string{"list"}, &stdout, &stderr) - if code != 2 { - t.Fatalf("code = %d, want 2", code) - } - if !strings.Contains(stderr.String(), "GRAFANA_URL") { - t.Fatalf("stderr = %q, want it to name the missing env var", stderr.String()) - } -} diff --git a/grafana-alertcheck/cmd/grafana-alertcheck/list.go b/grafana-alertcheck/cmd/list.go similarity index 86% rename from grafana-alertcheck/cmd/grafana-alertcheck/list.go rename to grafana-alertcheck/cmd/list.go index 687e5dd9f..c1d0e8532 100644 --- a/grafana-alertcheck/cmd/grafana-alertcheck/list.go +++ b/grafana-alertcheck/cmd/list.go @@ -11,11 +11,10 @@ import ( ) // runList reads every rule definition from the ruler endpoint and prints one -// line per rule: its kind, its Folder/Group/Title, and its uid. This is what -// makes the gate runnable end to end before any coverage logic exists (§9 -// rule 4) — it validates auth, the ruler parse, and the shapes Resolve -// matches against, all against a real Grafana. It is also the "did you mean" -// surface §17.2's no-match error points operators at. +// line per rule: its kind, its Folder/Group/Title, and its uid. It validates +// auth, the ruler parse, and the shapes Resolve matches against, all against a +// real Grafana, and it is the surface Resolve's no-match error points operators +// at. func runList(args []string, stdout, stderr io.Writer) int { if len(args) != 0 { fmt.Fprintf(stderr, "list takes no arguments, got %v\n", args) @@ -34,7 +33,7 @@ func runList(args []string, stdout, stderr io.Writer) int { // always terminates. It can still take minutes end-to-end under repeated // transient failures (5 retries * up to 30s backoff each, per call) — an // acceptable wait for an interactive `list`, not for `watch`/`check`, - // which get their own deadlines from `--until`/`to` in P10. + // which get their own deadlines from `--until`/`--to`. src := gate.NewHTTPSource(url, token, gate.SystemClock{}) version, err := src.Version(context.Background()) if err != nil { diff --git a/grafana-alertcheck/cmd/grafana-alertcheck/list_test.go b/grafana-alertcheck/cmd/list_test.go similarity index 73% rename from grafana-alertcheck/cmd/grafana-alertcheck/list_test.go rename to grafana-alertcheck/cmd/list_test.go index 119326c2c..62aa200d2 100644 --- a/grafana-alertcheck/cmd/grafana-alertcheck/list_test.go +++ b/grafana-alertcheck/cmd/list_test.go @@ -5,8 +5,9 @@ import ( "fmt" "net/http" "net/http/httptest" - "strings" "testing" + + "github.com/stretchr/testify/require" ) const rulerBody = `{ @@ -60,19 +61,11 @@ func TestRunList_HappyPath(t *testing.T) { var stdout, stderr bytes.Buffer code := run([]string{"list"}, &stdout, &stderr) - if code != 0 { - t.Fatalf("code = %d, want 0; stderr = %q", code, stderr.String()) - } + require.Equal(t, 0, code) out := stdout.String() - if !strings.Contains(out, "rule0000006a") { - t.Errorf("stdout = %q, want it to list rule0000006a", out) - } - if !strings.Contains(out, "Example No Gateways Available") { - t.Errorf("stdout = %q, want it to list the rule title", out) - } - if !strings.Contains(out, "grafana-managed") { - t.Errorf("stdout = %q, want it to name the rule kind", out) - } + require.Contains(t, out, "rule0000006a") + require.Contains(t, out, "Example No Gateways Available") + require.Contains(t, out, "grafana-managed") } func TestRunList_UnsupportedVersion(t *testing.T) { @@ -82,12 +75,8 @@ func TestRunList_UnsupportedVersion(t *testing.T) { var stdout, stderr bytes.Buffer code := run([]string{"list"}, &stdout, &stderr) - if code != 2 { - t.Fatalf("code = %d, want 2", code) - } - if !strings.Contains(stderr.String(), "12.5.0") { - t.Fatalf("stderr = %q, want it to name the unsupported version", stderr.String()) - } + require.Equal(t, 2, code) + require.Contains(t, stderr.String(), "12.5.0") } func TestRunList_RejectsArgs(t *testing.T) { @@ -96,7 +85,5 @@ func TestRunList_RejectsArgs(t *testing.T) { var stdout, stderr bytes.Buffer code := run([]string{"list", "extra"}, &stdout, &stderr) - if code != 2 { - t.Fatalf("code = %d, want 2", code) - } + require.Equal(t, 2, code) } diff --git a/grafana-alertcheck/cmd/grafana-alertcheck/main.go b/grafana-alertcheck/cmd/main.go similarity index 65% rename from grafana-alertcheck/cmd/grafana-alertcheck/main.go rename to grafana-alertcheck/cmd/main.go index d7968f5a8..7ab5d6397 100644 --- a/grafana-alertcheck/cmd/grafana-alertcheck/main.go +++ b/grafana-alertcheck/cmd/main.go @@ -1,8 +1,5 @@ -// Command grafana-alertcheck is the CLI entry point for the gate. P3 wires -// only the `list` subcommand — enough to validate auth, the ruler parse, and -// resolution against a real Grafana before any coverage logic exists (§9 rule -// 4, "reach runnable at PR 5"). P10 extends this file with `watch` and -// `check`. +// Command grafana-alertcheck is the CLI entry point for the gate: `list`, +// `watch` (record) and `check` (classify). package main import ( @@ -15,13 +12,13 @@ func main() { os.Exit(run(os.Args[1:], os.Stdout, os.Stderr)) } -const usage = "usage: grafana-alertcheck " +const usage = "usage: grafana-alertcheck " // run is the whole of main's testable surface: parse the subcommand, dispatch, // return the process exit code. Exit codes below 2 (pass/violations) belong to -// `check` alone (§20.3, P10); every failure reachable from here — a missing -// subcommand, a bad flag, a transport or auth failure — is a could-not-check -// condition and maps to 2, never to 0 or 1 (H7). +// `check` alone; every failure reachable from here — a missing subcommand, a +// bad flag, a transport or auth failure — is a could-not-check condition and +// maps to 2, never to 0 or 1. // // Requested help (-h/--help) is not a failure — it is the one exception to // that rule. Convention (and every stdlib flag.FlagSet default) is exit 0 to @@ -36,6 +33,10 @@ func run(args []string, stdout, stderr io.Writer) int { switch args[0] { case "list": return runList(args[1:], stdout, stderr) + case "watch": + return runWatch(args[1:], os.Stdin, stdout, stderr) + case "check": + return runCheck(args[1:], os.Stdin, stdout, stderr) case "-h", "-help", "--help": fmt.Fprintln(stdout, usage) return 0 diff --git a/grafana-alertcheck/cmd/main_test.go b/grafana-alertcheck/cmd/main_test.go new file mode 100644 index 000000000..34dd3b2d2 --- /dev/null +++ b/grafana-alertcheck/cmd/main_test.go @@ -0,0 +1,43 @@ +package main + +import ( + "bytes" + "testing" + + "github.com/stretchr/testify/require" +) + +func TestRun_NoArgs(t *testing.T) { + var stdout, stderr bytes.Buffer + code := run(nil, &stdout, &stderr) + require.Equal(t, 2, code) + require.Contains(t, stderr.String(), "usage") +} + +func TestRun_Help(t *testing.T) { + for _, flag := range []string{"-h", "-help", "--help"} { + t.Run(flag, func(t *testing.T) { + var stdout, stderr bytes.Buffer + code := run([]string{flag}, &stdout, &stderr) + require.Equal(t, 0, code, "requested help is not a could-not-check condition") + require.Contains(t, stdout.String(), "usage") + require.Empty(t, stderr.String(), "help goes to stdout") + }) + } +} + +func TestRun_UnknownSubcommand(t *testing.T) { + var stdout, stderr bytes.Buffer + code := run([]string{"bogus"}, &stdout, &stderr) + require.Equal(t, 2, code) + require.Contains(t, stderr.String(), `"bogus"`) +} + +func TestRun_List_MissingEnv(t *testing.T) { + t.Setenv("GRAFANA_URL", "") + t.Setenv("GRAFANA_TOKEN", "") + var stdout, stderr bytes.Buffer + code := run([]string{"list"}, &stdout, &stderr) + require.Equal(t, 2, code) + require.Contains(t, stderr.String(), "GRAFANA_URL") +} diff --git a/grafana-alertcheck/cmd/style.go b/grafana-alertcheck/cmd/style.go new file mode 100644 index 000000000..bb46fc2ec --- /dev/null +++ b/grafana-alertcheck/cmd/style.go @@ -0,0 +1,125 @@ +package main + +import ( + "bytes" + "io" + "os" + "strings" +) + +// ANSI SGR codes for the human-facing notes and table footer. The colours are +// applied only when the destination is a terminal (see colorEnabled); a pipe, +// file or CI log gets plain text, so stdout stays reserved for --output json +// and no machine reader ever sees escape sequences. +const ( + ansiReset = "\x1b[0m" + ansiRed = "\x1b[31m" + ansiGreen = "\x1b[32m" + ansiYellow = "\x1b[33m" + ansiCyan = "\x1b[36m" + // Orange has no entry in the base-16 palette; 256-colour 208 is a legible + // orange used for warnings, distinct from the yellow used for notes. + ansiOrange = "\x1b[38;5;208m" +) + +// colorEnabled reports whether ANSI colour should be written to w. Colour is +// written only when three things hold: NO_COLOR is unset, w is a real *os.File +// (so text/tabwriter buffers, strings.Builder and bytes.Buffer tests all stay +// plain), and that file is a character device (a terminal, not a redirect). +func colorEnabled(w io.Writer) bool { + if os.Getenv("NO_COLOR") != "" { + return false + } + f, ok := w.(*os.File) + if !ok { + return false + } + fi, err := f.Stat() + if err != nil { + return false + } + return fi.Mode()&os.ModeCharDevice != 0 +} + +// styleLine applies the note vocabulary's colour to one line when enabled. The +// colour wraps the text only; the terminating newline is written uncoloured so +// the terminal's line discipline is never inside the escape sequence. +func styleLine(line string, enabled bool) string { + if !enabled { + return line + } + content := strings.TrimRight(line, "\n") + var color string + switch { + case strings.HasPrefix(content, "warning:"): + color = ansiOrange + case strings.HasPrefix(content, "note:"): + color = ansiYellow + case strings.HasPrefix(content, "drain wait:"): + color = ansiCyan + } + if color == "" { + return line + } + return color + content + ansiReset + "\n" +} + +// noteStyler wraps the gate package's Notes stream — a presentation seam that +// keeps colour out of the library. It colourises each line by its known prefix +// and separates the collection countdown from the setup phase with a single +// blank line before the first "collecting:" line. The gate keeps emitting plain +// prose; only the CLI lays it out. +type noteStyler struct { + w io.Writer + enabled bool + pending []byte + sawCollecting bool +} + +func newNoteStyler(w io.Writer) *noteStyler { + return ¬eStyler{w: w, enabled: colorEnabled(w)} +} + +// startsSection reports whether a line opens a new phase of the stream and so +// deserves a blank line above it. "collecting:" opens the countdown (once — +// later countdown lines follow on from the first), and "drain wait:" opens the +// drain phase. The setup lines (planned run time, warning, min-observed, notes) +// are one contiguous block and are not separated from each other. +func (s *noteStyler) startsSection(line string) bool { + switch { + case strings.HasPrefix(line, "warning:"): + return true + case strings.HasPrefix(line, "drain wait:"): + return true + case strings.HasPrefix(line, "collecting:"): + if s.sawCollecting { + return false + } + s.sawCollecting = true + return true + } + return false +} + +func (s *noteStyler) Write(p []byte) (int, error) { + n := len(p) + s.pending = append(s.pending, p...) + for { + i := bytes.IndexByte(s.pending, '\n') + if i < 0 { + break + } + line := string(s.pending[:i+1]) + s.pending = s.pending[i+1:] + + if s.startsSection(line) { + if _, err := io.WriteString(s.w, "\n"); err != nil { + return n, err + } + } + if _, err := io.WriteString(s.w, styleLine(line, s.enabled)); err != nil { + return n, err + } + } + return n, nil +} diff --git a/grafana-alertcheck/cmd/table.go b/grafana-alertcheck/cmd/table.go new file mode 100644 index 000000000..39ec75cbc --- /dev/null +++ b/grafana-alertcheck/cmd/table.go @@ -0,0 +1,173 @@ +package main + +import ( + "fmt" + "io" + "sort" + "text/tabwriter" + "time" + + "github.com/smartcontractkit/chainlink-testing-framework/grafana-alertcheck/internal/gate" +) + +// renderTable is the human table. It always writes to the writer it is given, +// which the caller (runCheck) always points at stderr — stdout is reserved for +// the machine-readable --output json. +// +// Three titled tables, in order (the name column is RULE in all of them — one +// row is one resolved alert rule, never a firing instance): +// +// 1. RESULTS, one line per rule: outcome, BadFor, pollEvery, proved-or-not +// with the largest gap; +// 2. VIOLATIONS, one line per Violation (only when any): a rule's worst-of +// outcome does not carry the State/Health of the instance that actually +// caused it — Violation does — so this is also where those two columns +// appear, sorted after the result table rather than folded into it, and it +// is the only place an operator running WITHOUT --output json sees the +// --allow-paused hint that Violation.Note already carries (classify.go); +// 3. THRESHOLDS, the numbers that answer "why" on exit 2: each non-skipped +// rule's maxGap/healthGrace/evalStaleAfter, followed by the global +// transitionGrace and drainTimeout, and the largest measured clock skew +// alongside its own error bound (RTT/2) — SkewHardLimit is a separate, +// fixed input threshold and is reported next to it, never as if it were +// that bound. +func renderTable(w io.Writer, res gate.Result) error { + alertOf := make(map[string]string, len(res.Verdicts)) + for _, v := range res.Verdicts { + alertOf[v.RuleUID] = v.Alert + } + + enabled := colorEnabled(w) + // A blank line separates the result table from the notes the gate streamed + // before it (planned run time, warning, min-observed, collecting, drain + // wait), so the verdict reads as its own section rather than the tail of a + // wall of progress text. + fmt.Fprintln(w) + + fmt.Fprintln(w, "RESULTS") + tw := tabwriter.NewWriter(w, 0, 4, 2, ' ', 0) + fmt.Fprintln(tw, "RULE\tOUTCOME\tBADFOR\tPOLLEVERY\tPROVED\tNOTE") + for _, v := range sortedVerdicts(res.Verdicts) { + fmt.Fprintf(tw, "%s\t%s\t%s\t%s\t%s\t%s\n", + v.Alert, v.Outcome, v.BadFor.Round(time.Second), v.PollEvery.Round(time.Second), + provedLabel(res.Coverage[v.RuleUID]), v.Note) + } + if err := tw.Flush(); err != nil { + return fmt.Errorf("render table: %w", err) + } + + if len(res.Violations) > 0 { + fmt.Fprintln(w, "\nVIOLATIONS") + vtw := tabwriter.NewWriter(w, 0, 4, 2, ' ', 0) + fmt.Fprintln(vtw, "RULE\tOUTCOME\tSTATE\tHEALTH\tNOTE") + for _, v := range sortedViolations(res.Violations) { + fmt.Fprintf(vtw, "%s\t%s\t%s\t%s\t%s\n", alertLabel(v, alertOf), v.Outcome, v.State, v.Health, v.Note) + } + if err := vtw.Flush(); err != nil { + return fmt.Errorf("render table: %w", err) + } + } + + // The per-rule thresholds answer "why" on exit 2: a table, not the prose + // "rule NAME: maxGap=... healthGrace=... evalStaleAfter=..." that repeated + // the rule name a fourth time. It is separated from the result above by a + // blank line. + fmt.Fprintln(w) + fmt.Fprintln(w, "THRESHOLDS") + ttw := tabwriter.NewWriter(w, 0, 4, 2, ' ', 0) + fmt.Fprintln(ttw, "RULE\tMAXGAP\tHEALTHGRACE\tEVALSTALEAFTER") + for _, uid := range sortedThresholdUIDs(res.Thresholds, alertOf) { + t := res.Thresholds[uid] + fmt.Fprintf(ttw, "%s\t%s\t%s\t%s\n", + alertOr(uid, alertOf), t.MaxGap, t.HealthGrace, t.EvalStaleAfter) + } + if err := ttw.Flush(); err != nil { + return fmt.Errorf("render table: %w", err) + } + + fmt.Fprintln(w) + fmt.Fprintf(w, "global: transitionGrace=%s (source: %s) drainTimeout=%s\n", + res.Global.TransitionGrace, res.Global.GraceSource, res.Global.DrainTimeout) + fmt.Fprintf(w, "largest measured clock skew: %s (bound ±%s, hard limit %s), grafana %s\n", + res.ClockSkew.Round(time.Millisecond), res.ClockSkewBound.Round(time.Millisecond), + gate.SkewHardLimit, res.GrafanaVersion) + // The verdict — the single number a terminal operator reads last — sits on + // its own line at the very bottom, separated from the diagnostics above and + // from the shell prompt below. + fmt.Fprintf(w, "\n%s\n\n", violationsLabel(len(res.Violations), enabled)) + return nil +} + +// violationsLabel colours the "violations: N" prefix of the footer: green for a +// clean run, red otherwise. The rest of the line is written uncoloured. +func violationsLabel(n int, enabled bool) string { + s := fmt.Sprintf("violations: %d", n) + if !enabled { + return s + } + if n == 0 { + return ansiGreen + s + ansiReset + } + return ansiRed + s + ansiReset +} + +// provedLabel is the table's PROVED column: "yes" for a clean coverage +// proof, "no" with the reason and largest gap for an unobservable rule, and +// "-" for a rule decide never asked proveCoverage about at all (skipped — +// paused before the window opened). +func provedLabel(cov gate.CoverageResult) string { + if cov.Reason == "" && !cov.Unobservable && !cov.Proved { + return "-" + } + if cov.Unobservable { + if cov.LargestGap > 0 { + return fmt.Sprintf("no (%s; largest gap %s at %s)", cov.Reason, + cov.LargestGap.Round(time.Second), cov.LargestGapAt.Format(time.RFC3339)) + } + return fmt.Sprintf("no (%s)", cov.Reason) + } + return "yes" +} + +// alertLabel resolves a Violation's alert name. Most violations already +// carry it directly; the synthetic MinObserved-shortfall entry with no named +// rule (classify.go) has an empty Alert and an empty RuleUID, so alertOf +// cannot resolve it either — "-" says plainly that this row is not about a +// specific rule. +func alertLabel(v gate.Violation, alertOf map[string]string) string { + if v.Alert != "" { + return v.Alert + } + if a, ok := alertOf[v.RuleUID]; ok { + return a + } + return "-" +} + +func alertOr(uid string, alertOf map[string]string) string { + if a, ok := alertOf[uid]; ok { + return a + } + return uid +} + +func sortedVerdicts(in []gate.RuleVerdict) []gate.RuleVerdict { + out := append([]gate.RuleVerdict(nil), in...) + sort.Slice(out, func(i, j int) bool { return out[i].Alert < out[j].Alert }) + return out +} + +func sortedViolations(in []gate.Violation) []gate.Violation { + out := append([]gate.Violation(nil), in...) + sort.SliceStable(out, func(i, j int) bool { return out[i].Alert < out[j].Alert }) + return out +} + +func sortedThresholdUIDs(thresholds map[string]gate.RuleThresholds, alertOf map[string]string) []string { + uids := make([]string, 0, len(thresholds)) + for uid := range thresholds { + uids = append(uids, uid) + } + sort.Slice(uids, func(i, j int) bool { return alertOr(uids[i], alertOf) < alertOr(uids[j], alertOf) }) + return uids +} diff --git a/grafana-alertcheck/cmd/table_test.go b/grafana-alertcheck/cmd/table_test.go new file mode 100644 index 000000000..86c80a19b --- /dev/null +++ b/grafana-alertcheck/cmd/table_test.go @@ -0,0 +1,88 @@ +package main + +import ( + "bytes" + "testing" + "time" + + "github.com/smartcontractkit/chainlink-testing-framework/grafana-alertcheck/internal/gate" + "github.com/stretchr/testify/require" +) + +// The golden table test: a fixed Result renders a deterministic, ordered rule +// table, a violations section and a footer carrying the per-rule and global +// thresholds plus the skew and its bound — with no live Check involved. +func TestRenderTable(t *testing.T) { + gapAt := time.Date(2026, 1, 1, 12, 0, 0, 0, time.UTC) + res := gate.Result{ + GrafanaVersion: "13.1.0", + ClockSkew: 1500 * time.Millisecond, + ClockSkewBound: 250 * time.Millisecond, + Verdicts: []gate.RuleVerdict{ + {Alert: "Zebra Alert", RuleUID: "uid-z", Outcome: gate.OutcomeClean, PollEvery: 30 * time.Second}, + {Alert: "Ape Alert", RuleUID: "uid-a", Outcome: gate.OutcomeUnobservable, + PollEvery: 30 * time.Second, Note: "gap of 5m0s starting at 2026-01-01T12:00:00Z exceeds maxGap 1m0s"}, + {Alert: "Paused Alert", RuleUID: "uid-p", Outcome: gate.OutcomeSkipped, + Note: "paused before the window opened; counts against --min-observed unless --allow-paused is set"}, + }, + Violations: []gate.Violation{ + {Alert: "Ape Alert", RuleUID: "uid-a", Outcome: gate.OutcomeUnobservable, State: gate.StateFiring, Health: "error", Note: "unobservable"}, + {Alert: "Paused Alert", RuleUID: "uid-p", Outcome: gate.OutcomeSkipped, + Note: "paused before the window opened; counts against --min-observed unless --allow-paused is set"}, + }, + Coverage: map[string]gate.CoverageResult{ + "uid-z": {Proved: true}, + "uid-a": {Unobservable: true, Reason: gate.ReasonHeartbeatGap, LargestGap: 5 * time.Minute, LargestGapAt: gapAt}, + }, + Thresholds: map[string]gate.RuleThresholds{ + "uid-z": {MaxGap: time.Minute, HealthGrace: time.Minute, EvalStaleAfter: time.Minute}, + "uid-a": {MaxGap: time.Minute, HealthGrace: 2 * time.Minute, EvalStaleAfter: time.Minute}, + }, + Global: gate.GlobalThresholds{ + TransitionGrace: 5 * time.Minute, + GraceSource: `Ape Alert (for=5m)`, + DrainTimeout: 2 * time.Minute, + }, + } + + var buf bytes.Buffer + require.NoError(t, renderTable(&buf, res)) + out := buf.String() + + // Rule table: Ape sorts before Zebra sorts before... Paused is skipped and + // carries no coverage entry, so it renders "-" for PROVED. + require.Contains(t, out, "Ape Alert") + require.Contains(t, out, "unobservable") + require.Contains(t, out, "heartbeat_gap") + require.Contains(t, out, "largest gap 5m0s") + require.Contains(t, out, "Zebra Alert") + require.Contains(t, out, "clean") + + // The violations section must show up even without --output json, and must + // carry the --allow-paused hint text verbatim. + require.Contains(t, out, "VIOLATIONS") + require.Contains(t, out, "--allow-paused") + require.Contains(t, out, "STATE") + require.Contains(t, out, "HEALTH") + require.Contains(t, out, string(gate.StateFiring)) + require.Contains(t, out, "error") + + // The footer: per-rule thresholds are a table (RULE/MAXGAP/HEALTHGRACE/ + // EVALSTALEAFTER) rather than prose, followed by the global thresholds and + // the violations count with the skew and its own bound rather than the + // fixed hard limit. + require.Contains(t, out, "MAXGAP") + require.Contains(t, out, "HEALTHGRACE") + require.Contains(t, out, "EVALSTALEAFTER") + require.Contains(t, out, "global: transitionGrace=5m0s (source: Ape Alert (for=5m)) drainTimeout=2m0s") + require.Contains(t, out, "largest measured clock skew: 1.5s (bound ±250ms, hard limit 1m0s)") + require.Contains(t, out, "violations: 2") + require.Contains(t, out, "13.1.0") +} + +// The "-" case: a rule decide never asked proveCoverage about (paused before +// the window opened) has an empty CoverageResult and must not be reported as +// either proved or unobservable. +func TestProvedLabel_Skipped(t *testing.T) { + require.Equal(t, "-", provedLabel(gate.CoverageResult{})) +} diff --git a/grafana-alertcheck/cmd/watch.go b/grafana-alertcheck/cmd/watch.go new file mode 100644 index 000000000..69b585298 --- /dev/null +++ b/grafana-alertcheck/cmd/watch.go @@ -0,0 +1,153 @@ +package main + +import ( + "context" + "errors" + "flag" + "fmt" + "io" + "time" + + "github.com/smartcontractkit/chainlink-testing-framework/grafana-alertcheck/internal/gate" +) + +const watchUsage = "usage: grafana-alertcheck watch --out [--pidfile F] [--daemon-log F] " + + "--alerts [--folder F] [--poll-interval D] [--concurrency N] [--until RFC3339]" + +// runWatch is the record step's entire CLI surface, split in two by one flag +// set — gate.DaemonChildFlag ("--daemon-child") and gate.ReadyFDFlag +// ("--ready-fd") select which side of the parent/child split this invocation +// is: +// +// - without them: the record command an operator types. It parses --out, +// --alerts and the rest, builds a gate.WatchConfig and calls gate.Watch, +// which resolves, records the first observation of every rule, and +// detaches the recorder before returning. +// - with them: the detached recorder itself. gate.Watch's own childArgs +// (watch_unix.go) is the only thing that ever sets them — an operator +// never types "--daemon-child" and it does not appear in watchUsage — and +// this dispatches straight to gate.RunDaemonChild. +// +// Both flags live in the SAME flag set as the operator-facing ones rather +// than a second, hidden set: the child is started with childArgs' exact +// argv, e.g. "watch --daemon-child --out log.jsonl --ready-fd 3 +// [--until ...] [--concurrency ...]", and a second parser would have to stay +// byte-for-byte in sync with that slice to accept it. +func runWatch(args []string, stdin io.Reader, stdout, stderr io.Writer) int { + fs := flag.NewFlagSet("watch", flag.ContinueOnError) + fs.SetOutput(stderr) + fs.Usage = func() { fmt.Fprintln(stderr, watchUsage) } + + common := registerCommon(fs) + out := fs.String("out", "", "JSONL log path to record to") + pidfile := fs.String("pidfile", "", "pidfile path (default .pid)") + daemonLog := fs.String("daemon-log", "", "stdout/stderr sink for the detached recorder (default .daemon.log)") + until := fs.String("until", "", "optional hard stop, RFC3339 (default: run until check stops it)") + pollInterval := fs.String("poll-interval", "", "override every rule's poll cadence (default: half its own evaluation interval)") + + // Hidden: never in watchUsage, never typed by an operator (see doc comment). + daemonChild := fs.Bool(gate.DaemonChildFlag[2:], false, "") + readyFD := fs.Int(gate.ReadyFDFlag[2:], 0, "") + + if err := fs.Parse(args); err != nil { + if errors.Is(err, flag.ErrHelp) { + return 0 + } + return 2 + } + if fs.NArg() != 0 { + fmt.Fprintf(stderr, "watch: unexpected arguments %v\n", fs.Args()) + return 2 + } + + if *daemonChild { + return runDaemonChild(*out, *until, *common.concurrency, *readyFD, stderr) + } + + url, token, err := grafanaEnv() + if err != nil { + fmt.Fprintln(stderr, err) + return 2 + } + + alerts, err := readAlerts(stdin, *common.alerts) + if err != nil { + fmt.Fprintln(stderr, err) + return 2 + } + + cfg := gate.WatchConfig{ + URL: url, + Token: token, + Alerts: alerts, + Folder: *common.folder, + Out: *out, + PidFile: *pidfile, + DaemonLog: *daemonLog, + Concurrency: *common.concurrency, + Clock: gate.SystemClock{}, + Notes: newNoteStyler(stderr), + } + if *until != "" { + t, err := time.Parse(time.RFC3339, *until) + if err != nil { + fmt.Fprintf(stderr, "--until: %v\n", err) + return 2 + } + cfg.Until = t + } + if *pollInterval != "" { + d, err := time.ParseDuration(*pollInterval) + if err != nil { + fmt.Fprintf(stderr, "--poll-interval: %v\n", err) + return 2 + } + cfg.PollEvery = d + } + + if err := gate.Watch(context.Background(), cfg); err != nil { + fmt.Fprintln(stderr, err) + return 2 + } + return 0 +} + +// runDaemonChild is the detached recorder's whole entry point. Its stdout and +// stderr are already the daemon log file — spawnChild (watch_unix.go) +// redirects both before Start — so writing to stderr here lands exactly where +// waitForChildReady's failure path quotes from. +func runDaemonChild(out, until string, concurrency, readyFD int, stderr io.Writer) int { + url, token, err := grafanaEnv() + if err != nil { + fmt.Fprintln(stderr, err) + return 2 + } + cfg := gate.DaemonChildConfig{ + URL: url, + Token: token, + Out: out, + Concurrency: concurrency, + Clock: gate.SystemClock{}, + ReadyFD: readyFD, + } + if until != "" { + t, err := time.Parse(time.RFC3339, until) + if err != nil { + fmt.Fprintf(stderr, "--until: %v\n", err) + return 2 + } + cfg.Until = t + } + if err := gate.RunDaemonChild(context.Background(), cfg); err != nil { + fmt.Fprintln(stderr, err) + return 2 + } + return 0 +} + +// The daemon-child and ready-fd flags registered above must keep matching +// watch_unix.go's childArgs, which names exactly --daemon-child, --out, +// --ready-fd, --until and --concurrency and nothing else: that function +// builds this process's own argv when it re-execs itself as the detached +// recorder, so a flag added to one side without the other means the child +// fails on its very first flag.Parse. diff --git a/grafana-alertcheck/cmd/watch_test.go b/grafana-alertcheck/cmd/watch_test.go new file mode 100644 index 000000000..5c8f53c55 --- /dev/null +++ b/grafana-alertcheck/cmd/watch_test.go @@ -0,0 +1,82 @@ +package main + +import ( + "bytes" + "os" + "testing" + + "github.com/stretchr/testify/require" +) + +// The record step's flag-validation matrix. Every case fails inside +// gate.WatchConfig.validate() or before it, so none needs a reachable Grafana. +func TestRunWatch_FlagValidation(t *testing.T) { + tests := []struct { + name string + env bool + args func(t *testing.T) []string + wantErr string + }{ + {"missing env", false, func(t *testing.T) []string { + return []string{"--out", t.TempDir() + "/log.jsonl", "--alerts", writeTempAlerts(t)} + }, "GRAFANA_URL"}, + {"missing out", true, func(t *testing.T) []string { + return []string{"--alerts", writeTempAlerts(t)} + }, "no log path"}, + {"missing alerts", true, func(t *testing.T) []string { + return []string{"--out", t.TempDir() + "/log.jsonl"} + }, "no alert names"}, + {"bad until format", true, func(t *testing.T) []string { + return []string{"--out", t.TempDir() + "/log.jsonl", "--alerts", writeTempAlerts(t), "--until", "not-a-time"} + }, "--until"}, + {"until in the past", true, func(t *testing.T) []string { + return []string{"--out", t.TempDir() + "/log.jsonl", "--alerts", writeTempAlerts(t), "--until", "2000-01-01T00:00:00Z"} + }, "not in the future"}, + {"bad poll-interval", true, func(t *testing.T) []string { + return []string{"--out", t.TempDir() + "/log.jsonl", "--alerts", writeTempAlerts(t), "--poll-interval", "not-a-duration"} + }, "--poll-interval"}, + } + for _, tt := range tests { + t.Run(tt.name, func(t *testing.T) { + if tt.env { + t.Setenv("GRAFANA_URL", "http://example.invalid") + t.Setenv("GRAFANA_TOKEN", "test-token") + } else { + t.Setenv("GRAFANA_URL", "") + t.Setenv("GRAFANA_TOKEN", "") + } + var stdout, stderr bytes.Buffer + args := append([]string{"watch"}, tt.args(t)...) + code := run(args, &stdout, &stderr) + require.Equal(t, 2, code) + require.Contains(t, stderr.String(), tt.wantErr) + }) + } +} + +// Seeing gate.DaemonChildFlag must dispatch to gate.RunDaemonChild, and the +// flag must never appear in watchUsage (an operator never types it). +func TestRunWatch_DaemonChildDispatch(t *testing.T) { + t.Setenv("GRAFANA_URL", "http://example.invalid") + t.Setenv("GRAFANA_TOKEN", "test-token") + + var stdout, stderr bytes.Buffer + // No log at this path: RunDaemonChild fails trying to read it, which is + // enough to prove dispatch happened without needing a real recording. + missing := os.DevNull + ".missing" + code := run([]string{"watch", "--daemon-child", "--out", missing, "--ready-fd", "0"}, &stdout, &stderr) + require.Equal(t, 2, code) + require.Contains(t, stderr.String(), missing) + require.NotContains(t, watchUsage, "daemon-child") + require.NotContains(t, watchUsage, "ready-fd") +} + +func TestRunWatch_DaemonChild_MissingEnv(t *testing.T) { + t.Setenv("GRAFANA_URL", "") + t.Setenv("GRAFANA_TOKEN", "") + + var stdout, stderr bytes.Buffer + code := run([]string{"watch", "--daemon-child", "--out", "log.jsonl"}, &stdout, &stderr) + require.Equal(t, 2, code) + require.Contains(t, stderr.String(), "GRAFANA_URL") +} diff --git a/grafana-alertcheck/docs/_category_.yaml b/grafana-alertcheck/docs/_category_.yaml new file mode 100644 index 000000000..3cbc431c4 --- /dev/null +++ b/grafana-alertcheck/docs/_category_.yaml @@ -0,0 +1,8 @@ +position: 1 +label: 'Grafana Alertcheck' +collapsible: true +collapsed: false +link: + type: generated-index + slug: /platform-services/devex/cicd/grafana-alertcheck/index + description: 'CD quality gate for Grafana alerts: watch, classify, and gate releases.' diff --git a/grafana-alertcheck/docs/advanced.md b/grafana-alertcheck/docs/advanced.md new file mode 100644 index 000000000..073292cd6 --- /dev/null +++ b/grafana-alertcheck/docs/advanced.md @@ -0,0 +1,41 @@ +--- +id: grafana-alertcheck-advanced +title: Check budget and scheduling +sidebar_label: Budget and scheduling +sidebar_position: 2 +description: Why grafana-alertcheck schedules per rule, how the request budget works, and why it never queries state history. +--- + +# Check budget and scheduling + +## Per-rule schedules, never a global cycle + +Each rule polls at its **own** cadence, `--poll-interval` (default: half the rule's own evaluation interval). There is deliberately no single global minimum-interval cycle. + +One rule at `intervalSeconds=10` beside twenty at `300` keeps a 5 s cadence for itself and 150 s for the other twenty — not a 5 s cycle for all of them, which would be a 60× request bloat at ~1.8 s per request and would fail to start on a reasonable fleet. + +The scheduler staggers each rule's initial next-due time across its cadence, and serves due rules **earliest-due-first**, so a tight rule never queues behind slack ones. + +## The check budget + +The gate records one observation of every rule up front and checks the schedule against those **measured** latencies (payload sizes vary ~230× across rules, so a fixed estimate is meaningless). It errors at start — before waiting — if any of three conditions hold: + +- **Utilization** — total request rate exceeds `--concurrency`. +- **Per-rule** — one rule's request can't fit its own cadence. +- **Burst bound** — the slowest request exceeds the fleet's tightest cadence, which can open a mid-run gap. + +The error names the three levers only: raise `--concurrency`, raise `--poll-interval`, or watch fewer alerts. It never prescribes a single interval. + +## Why the gate never queries state history + +Querying Grafana's alert state history after the fact fails closed *in the wrong direction* — it returns "pass" when the truth is unknown: + +- History stores **transitions**, not states. An alert firing through the whole window has its only record *before* the window. +- The annotations API **does not serve Loki-backed history** at all. +- An empty result is indistinguishable from a healthy one: no alert fired, the backend differs, retention removed data, the token lacked permission — all look identical. +- There is **no coverage signal** — nothing proves the history is complete to time T. +- Artifact transitions (`Paused`, `RuleDeleted`, `Updated`, `MissingSeries`) look like recoveries. + +Instead, `watch` records its own evidence live and the log becomes the source of truth. The trade-off: the gate can miss an episode shorter than a rule's poll interval, though `activeAt` still surfaces sub-interval onsets for instances still active at a poll. + +A corollary of recording fresh: there is no replay. Re-running a failed job is a new deploy with a new `from` and a new recording — never a re-classification of old evidence. diff --git a/grafana-alertcheck/docs/architecture.md b/grafana-alertcheck/docs/architecture.md new file mode 100644 index 000000000..ca493220e --- /dev/null +++ b/grafana-alertcheck/docs/architecture.md @@ -0,0 +1,68 @@ +--- +id: grafana-alertcheck-architecture +title: Architecture +sidebar_label: Architecture +sidebar_position: 3 +description: The design invariants, pure-function seam, and recorder lifecycle of grafana-alertcheck, for maintainers. +--- + +# Architecture + +This page documents the invariants and seams a maintainer must not break. It exists because most of them are the difference between a gate that fails closed and one that silently passes broken windows. + +## Fail-closed invariants + +The gate must stop the release if it cannot get an answer. Every rule below is a specific instance of that: + +- **An error is never a pass.** A pass is exactly `len(Violations) == 0 && err == nil`. Every error path leaves `err` non-nil, and the CLI maps that to exit `2` unconditionally. +- **Inability beats violation.** Any `unobservable` rule is exit `2`, even alongside a real violation found first. +- **Absent never means normal.** An instance that leaves the bad set is looked up in the *same* response: present as `normal` → cleared; absent (or `MissingSeries`) → vanished (a discontinuity, not a recovery). +- **Staleness is absolute.** `grafana_now − lastEvaluation` is compared against a threshold, never "did it increase since the last poll" — a delta check reports stale on ~half the polls of a healthy rule. +- **`grafana_now` is the response `Date` header.** Never the runner clock, in any comparison against a Grafana timestamp. +- **No early exit.** `check` collects to `to + transitionGrace` before classifying once. +- **No replay.** No run-id key, no artifact download, no state between attempts. A retry is a new deploy. + +## The pure-function seam + +All correctness lives in two phases written as **pure functions** over a flat list of polls — no HTTP, no files, no clock, no goroutines: + +``` +HTTP ──> Source ──> []StateRule ──> reduce ──> []Poll ──> proveCoverage ──> decide ──> Result + │ + JSONL log ──> ReadLog ──┘ +``` + +- `proveCoverage` (the nine coverage checks) and `decide` (the instance timelines and outcomes) are pure; tests drive them with `[]Poll` literals and a fake `Clock`, with no sleeping or fixture server. +- `Check`/`Watch` are I/O shells: HTTP, signals, the pidfile, file reads, the countdown print. The only test doubles needed are the `Source` and `Clock` interfaces. +- `Policy` is the narrowed view of `Config` that reaches the pure layer — classification knobs and the window, no URL and no token. The token must never cross that line, which is the cheapest guarantee it never lands in an error string or a result. + +## Strict parsing as the version guard + +Both API responses are parsed strictly: a missing or unparseable **required** field (`health`, `state`, `lastEvaluation`, `interval`) is an error, never a zero value. Optional keys (`alerts`, `totals`, `labels`, `keepFiringFor`) are absent-tolerant, and unknown keys are ignored — so Grafana can add fields without breaking the parser, but removing one fails loudly. + +This, plus the declared supported range (Grafana >= 13.0.0, < 14.0.0), is how a deprecation or schema change is caught instead of silently misread. + +## The recorder lifecycle + +`watch` detaches a background recorder so observation survives the step boundary: + +1. Parent resolves names, writes the header, observes every non-paused rule once, checks the budget. +2. Parent re-execs itself as the child (`--daemon-child`) under a new session/process group, stdout/stderr to the daemon log. +3. Child re-reads the header, reopens the log `O_APPEND`, takes the exclusive `flock`, and writes one readiness byte on `--ready-fd`. +4. Parent writes the pidfile **after** the readiness report, then returns. + +Two authorities, only one of which is evidence: + +- The **pidfile** says a recording ever started (written only after ready, removed on failure). It can go stale — a pid gets reused. +- The **flock** says a writer exists *now*. The kernel drops it on exit, so the lock is always authoritative. + +On a clean stop (SIGTERM/SIGINT/`--until`) the child finishes the in-flight write, appends the `stopped` sentinel, fsyncs, and exits. A hard error writes no sentinel — so a recorder that died reads exactly like a coverage gap, because it is one. + +`check` signals via the pidfile, waits for the **lock** to release (never the pid), and only then reads the log once. Reading while a writer can still append can only produce a shorter window than was recorded. + +## The log is the source of truth + +`watch` records raw evidence, so nothing trusts a state that could become unreachable. Two consequences a maintainer must preserve: + +- The **header is authoritative for recording facts** (the cadence actually used, the URL, the alert set); the ruler API is authoritative for **rule facts** (`for`, `intervalSeconds`, kind). `check` always re-resolves definitions fresh and never reconstructs them from the header — the header duplicates `for`/`interval` only so the uploaded artifact is self-describing. +- The **cadence authority** is the header's `poll_every_seconds`, not the definitions. Re-deriving it would compare gaps recorded at an override cadence against default-cadence thresholds — fail-open in the faster-override direction. diff --git a/grafana-alertcheck/docs/how-alerts-are-evaluated.md b/grafana-alertcheck/docs/how-alerts-are-evaluated.md new file mode 100644 index 000000000..7c9cb3b6a --- /dev/null +++ b/grafana-alertcheck/docs/how-alerts-are-evaluated.md @@ -0,0 +1,85 @@ +--- +id: grafana-alertcheck-evaluation +title: How alerts are evaluated +sidebar_label: How alerts are evaluated +sidebar_position: 1 +description: The verdict model, instance timelines, and coverage proof behind grafana-alertcheck. +--- + +# How alerts are evaluated + +Both `watch`+`check` (recorder mode) and `check` alone (single-step mode) converge on the same input: a flat list of polls. Everything below runs over that list; the mode only changes where the polls came from. + +## Instance states + +Grafana reports instance states in two vocabularies (`Alerting`/`Normal` at instance level, `firing`/`inactive` at rule level). The gate normalizes every instance to one canonical set: + +| Canonical | Meaning | +| --------- | ------- | +| `normal` | Healthy | +| `firing` | The condition is true and `for` has elapsed | +| `pending` | The condition is true, `for` has not elapsed | +| `nodata` | The query returned no series (synthetic instance) | +| `error` | The query failed (synthetic instance) | + +A rule's **rule-level** `state` and `health` are kept verbatim and only reported — they are never classified. The **instance** state is what the classifier reasons about. + +A "bad" instance is one whose canonical state is in `--states` (default `firing`). `pending` and `nodata` are excluded by default. + +## Verdict model + +For each instance the gate builds a timeline of bad spans over `[from, to]`, then takes the worst outcome across a rule's instances as the rule's outcome. + +| Outcome | Shape | Exit | +| ------- | ----- | ---- | +| `clean` | Good throughout, observed throughout | → 0 | +| `newly_bad` | Entered a bad state **inside** the window | → 1 | +| `persistently_bad` | Bad at `from`, still bad at `to` | → 1 | +| `recovered` | Bad at `from`, cleared before `to`, stayed clear | → 0 | +| `flapping` | Cleared, then became bad again | → 1 | +| `skipped` | Paused **before** the window opened | reported, not observable | +| `unobservable` | Coverage gap / sustained `health=error` / stale / absent | → 2 | + +`recovered` has **no deadline** — an alert that clears at minute 58 of a 60-minute window still passes. The total bad time is reported as `BadFor`; the removed deadline is replaced by that measured value rather than a derived limit. + +### Preexisting policy + +For an instance already bad when `from` opened, `--preexisting` decides: + +- `fail-unless-recovered` (default) — clears and stays clear → pass; never clears → fail. +- `fail` — any preexisting instance fails, recovered or not. +- `ignore` — preexisting instances are disregarded; only new episodes fail. + +## Cleared vs vanished + +When an instance leaves the bad set, the gate looks it up **in the same response**: + +- Present as `normal` → `cleared` (a real recovery). +- Absent, or present as `normal (MissingSeries)` → `vanished` (a discontinuity, **not** a recovery). + +A vanished instance that was bad stays `persistently_bad`. A metric that stops being emitted is not evidence of health — this is deliberate and can surprise users whose fix is to remove a metric rather than drive it to a good value. + +## Coverage proof + +Before classifying, `check` must **prove** continuous coverage of `[from, to]` for each alert. Nine checks run; any failure makes the rule `unobservable`: + +1. **Sentinel** — a clean recorder stop, timestamped at or after `to + transitionGrace`. A recorder that died mid-window looks exactly like a coverage gap and is one. +2. **`from` bounds** — `from` earlier than the recording start is unprovable. +3. **Heartbeat gap** — any gap larger than `maxGap` (= 2 × poll cadence) inside the window. Data at both ends with a hole between is not enough. +4. **`health=error`** — a contiguous run longer than `healthGrace` consumes coverage; a short blip is a note. +5. **`health=nodata`** — a note, never fatal (unless `--nodata-is-unobservable`). +6. **Liveness** — `grafana_now − lastEvaluation` must not exceed `evalStaleAfter`. This is an **absolute** check, never a "did it increase since the last poll" delta. +7. **In-window pause** — a poll reporting `isPaused` mid-window is `unobservable` (the primary pause detector). +8. **Rule absent** — an authoritative `2xx` with no matching rule. +9. **`KeepLast`** — a note naming a stale-state blind spot. + +## Health: `error` vs `nodata` + +- `health=error` means the query **failed** — a malfunction. Sustained past `healthGrace`, it makes the rule `unobservable`. +- `health=nodata` means the query **ran and returned no series** — indistinguishable from a quiet system. It is not fatal by default; most of a fleet runs `no_data_state: OK`. + +## The drain wait and `transitionGrace` + +A condition that arises just before `to` becomes `firing` only at the first evaluation after its `for` elapses. `transitionGrace` (derived from the watched rules' `for` values) extends the classification bound past `to` so such a surfacing condition is caught. After collection, a **drain wait** polls until each rule has evaluated through `to + transitionGrace` (bounded by `drainTimeout`); a rule that never does is `unobservable`. + +Run time = `(to − from) + transitionGrace + drainTimeout`. This is printed at start, and the grace is warned about when it exceeds a quarter of the window — the window may be too short for the alert's `for`. diff --git a/grafana-alertcheck/docs/index.md b/grafana-alertcheck/docs/index.md new file mode 100644 index 000000000..95d413772 --- /dev/null +++ b/grafana-alertcheck/docs/index.md @@ -0,0 +1,87 @@ +--- +id: grafana-alertcheck-index +title: Grafana Alertcheck +sidebar_label: Overview +sidebar_position: 0 +description: A CD quality gate that bookends a release with alert-state observation and answers whether any watched Grafana alert was bad during the release window. +--- + +# Grafana Alertcheck + +`grafana-alertcheck` is a CD quality gate for Grafana alerts. It bookends a release with two commands — `watch` (record) and `check` (classify) — and answers one question: + +> A release finished at time T. Was any of these Grafana alerts in a bad state during the next N minutes? + +The contract is `watch → your work → check`. Between the two you run whatever you want (deploy, tests, migration); the gate only observes, then classifies. + +It **fails closed**: if it cannot get an answer, it stops the release. It never passes an unproven window. + +## How it works, in one paragraph + +`watch` starts a background recorder that polls each named alert and appends snapshots to a JSONL log. Your work then emits two RFC3339 timestamps — `from` (when the change landed) and `to` (when the work ended). `check` proves continuous coverage of `[from, to]`, builds a state timeline per alert, classifies it, and exits `0`, `1`, or `2`. + +## Install + +```bash +go install github.com/smartcontractkit/chainlink-testing-framework/grafana-alertcheck/cmd@latest +``` + +Connection details come from the environment — the token is env-only, never a flag: + +```bash +export GRAFANA_URL=https://grafana.example.com +export GRAFANA_TOKEN=… +``` + +Requires Grafana >= 13.0.0 and < 14.0.0. Outside that range the gate exits `2`. + +## Quickstart — recorder mode + +```bash +grafana-alertcheck watch --out /tmp/run.jsonl --alerts alerts.txt +./deploy.sh # emits deployed_at= when the rollout is stable +./verify.sh # emits finished_at= when the work is done +grafana-alertcheck check --in /tmp/run.jsonl --from "$deployed_at" --to "$finished_at" +``` + +`alerts.txt` holds one alert name per line. See [Naming alerts](./reference/cli#naming-alerts). + +`watch` returns only after the recorder has observed every named, non-paused alert once and reported ready — so auth, name-resolution, and parse failures surface **before** your deploy runs. + +## Quickstart — single-step mode + +Skip the recorder and observe the window inline, from inside `check` itself: + +```bash +grafana-alertcheck check --alerts alerts.txt --to "$finished_at" +``` + +In single-step mode the window starts at `check`'s first observation; if you give no `--from`, the interval before that first observation is declared as a blind spot with a warning (not an error). + +## Exit codes + +| Code | Meaning | +| ---- | ------- | +| `0` | Pass — no violations | +| `1` | Violations (including a paused-rule-only `--min-observed` shortfall) | +| `2` | The gate could not check — config, auth, resolution, a coverage gap, health/staleness, the drain limit, transport, … | + +An error is never a pass: `2` wins over any violation found alongside it. + +## Common surprises + +- A **paused rule fails by default** — even one someone else paused. Use `--allow-paused`. +- A fix that **stops emitting a metric is not a recovery** — the instance vanishes, which is a discontinuity, not health. +- The gate checks alert **state and health**, not notification delivery — a silenced alert that still fires fails. +- `recovered` has **no deadline** — a bad-at-`from` alert that clears by `to` passes; set `--preexisting fail` to forbid it. +- A **retry is a new deploy**, not a replay — re-running the job re-records against a new `from`. +- `watch` and `check` must run in **one job, one runner, one filesystem** — nothing persists across jobs or attempts. +- The gate **never exits early** — a violation at minute 2 still holds the runner to `to + transitionGrace + drainTimeout`; size the job timeout to the planned run time the gate prints at start. + +## More + +- [How alerts are evaluated](./how-alerts-are-evaluated) — the verdict model and coverage proof +- [Check budget and scheduling](./advanced) — why the schedule and budget look the way they do, and why history isn't queried +- [Architecture](./architecture) — design invariants and the recorder lifecycle, for maintainers +- [CLI reference](./reference/cli) — every subcommand and flag +- [Log format](./reference/log-format) — the JSONL log schema, for debugging artifacts diff --git a/grafana-alertcheck/docs/reference/_category_.yaml b/grafana-alertcheck/docs/reference/_category_.yaml new file mode 100644 index 000000000..2e5d50946 --- /dev/null +++ b/grafana-alertcheck/docs/reference/_category_.yaml @@ -0,0 +1,8 @@ +position: 3 +label: Reference +collapsible: true +collapsed: false +link: + type: generated-index + slug: /platform-services/devex/cicd/grafana-alertcheck/reference + description: 'CLI reference for grafana-alertcheck.' diff --git a/grafana-alertcheck/docs/reference/cli.md b/grafana-alertcheck/docs/reference/cli.md new file mode 100644 index 000000000..1530c508d --- /dev/null +++ b/grafana-alertcheck/docs/reference/cli.md @@ -0,0 +1,91 @@ +--- +id: grafana-alertcheck-cli +title: CLI reference +sidebar_label: CLI reference +sidebar_position: 0 +description: Full reference for the grafana-alertcheck CLI: watch, check, list, environment, naming, and output. +--- + +# CLI reference + +``` +grafana-alertcheck +``` + +Connection details are always from the environment: `GRAFANA_URL` and `GRAFANA_TOKEN`. The token is never a flag and never logged. + +## `list` + +Lists every rule from the ruler endpoint — kind, folder, group, title, uid. Useful to check auth and to find `uid:` names. + +```bash +grafana-alertcheck list +``` + +## `watch` — record + +```bash +grafana-alertcheck watch --out [--pidfile F] [--daemon-log F] \ + --alerts [--folder F] [--poll-interval D] [--concurrency N] [--until RFC3339] +``` + +| Flag | Default | Meaning | +| ---- | ------- | ------- | +| `--out` | — | JSONL log path (required) | +| `--pidfile` | `.pid` | Where the recorder's pid is written | +| `--daemon-log` | `.daemon.log` | stdout/stderr sink for the detached recorder | +| `--alerts` | — | File of alert names, one per line, or `-` for stdin (required) | +| `--folder` | — | Default folder to scope unqualified names | +| `--poll-interval` | half the rule's interval | Override every rule's cadence (never clamped) | +| `--concurrency` | `1` | Max concurrent requests to Grafana | +| `--until` | run until signalled | Optional hard stop | + +`watch` writes the header, observes every non-paused rule once, checks the budget, then detaches a background recorder and returns. Recording is **unfiltered** — there is no `--states` here, so the same log can be re-classified later under different `--states` without re-recording. + +## `check` — classify + +```bash +grafana-alertcheck check [--in ] [--pidfile F] --from RFC3339 --to RFC3339 \ + [--alerts ...] [--folder F] [--states ...] [--preexisting ...] [--min-observed N] \ + [--allow-paused] [--nodata-is-unobservable] [--concurrency N] [--output json] +``` + +| Flag | Default | Meaning | +| ---- | ------- | ------- | +| `--in` | — | Log recorded by `watch`; empty selects single-step mode | +| `--pidfile` | `.pid` | Recorder to stop before reading `--in` | +| `--from` | see below | Moment the deploy finished | +| `--to` | — | End of the window (required) | +| `--alerts` | — | Required **without** `--in`; refused **with** `--in` | +| `--states` | `firing` | Comma-separated bad states: `firing,pending,nodata,error` | +| `--preexisting` | `fail-unless-recovered` | `fail-unless-recovered` \| `fail` \| `ignore` | +| `--min-observed` | every resolved rule | Minimum rules that must be observed | +| `--allow-paused` | `false` | Don't count pre-window-paused rules against `--min-observed` | +| `--nodata-is-unobservable` | `false` | Treat sustained `health=nodata` as unobservable | +| `--concurrency` | `1` | Max concurrent requests | +| `--output` | `table` | `json` also writes the machine-readable result to stdout | + +`--from` and `--to` are RFC3339 with an explicit offset and must come from your work — `from` from the deploy step, `to` from the step that finishes. In recorder mode an absent `--from` is a hard error; in single-step mode it falls back (with a warning) to the start of the step. + +## Naming alerts + +Alert names take one of four forms: + +| Form | Meaning | +| ---- | ------- | +| `HighErrorRate` | Title only, scoped by `--folder` | +| `Platform/HighErrorRate` | Folder + title | +| `Platform/api/HighErrorRate` | Folder + group + title (always unique) | +| `uid:abc123` | Exact uid (present on both endpoints) | + +Datasource-managed and recording rules are refused with a specific error. A name matching multiple rules errors listing every candidate with the copyable `Folder/Group/Title` and its `uid:` form. A no-match errors with case-insensitive substring suggestions and points at `list`. Duplicate names that resolve to the same uid collapse to one (a note, not an error). + +## Output and exit codes + +The human table goes to **stderr**: `RESULTS` (one row per rule), `VIOLATIONS` (one per violation), and `THRESHOLDS` (each rule's `maxGap`/`healthGrace`/`evalStaleAfter` plus global `transitionGrace`/`drainTimeout` and the largest measured clock skew). `--output json` writes the result to stdout. + +| Code | Meaning | +| ---- | ------- | +| `0` | Pass | +| `1` | Violations | +| `2` | Could not check — every library error, never a pass | diff --git a/grafana-alertcheck/docs/reference/log-format.md b/grafana-alertcheck/docs/reference/log-format.md new file mode 100644 index 000000000..66e071770 --- /dev/null +++ b/grafana-alertcheck/docs/reference/log-format.md @@ -0,0 +1,98 @@ +--- +id: grafana-alertcheck-log-format +title: Log format +sidebar_label: Log format +sidebar_position: 1 +description: The JSONL log schema written by watch and read by check, for debugging the forensic artifact. +--- + +# Log format + +`watch` records evidence to a JSONL log — one JSON object per line. A poll record *is* the heartbeat; there is no separate heartbeat type. + +## Record types + +Exactly three: + +| `type` | Meaning | +| ------ | ------- | +| `header` | Line 1 — identity and the alert set | +| `poll` | One reduced observation of one rule | +| `stopped` | The sentinel, written on a clean stop only | + +The header must be line 1, appear once, and carry `schema_version` `1` (any other value is a read error). Any unparseable line — including the last, or one after the sentinel — makes the log unreadable: a truncated log is evidence the recorder was killed, and must not pass. + +## Header + +```json +{ + "type": "header", + "schema_version": 1, + "url": "https://grafana.example.com", + "grafana_version": "13.1.0", + "started_at": "2026-09-07T10:00:00Z", + "rules": [ + { + "uid": "rule0000001", + "title": "HighErrorRate", + "folder": "Platform", + "group": "api", + "for_seconds": 300, + "interval_seconds": 60, + "is_paused": false, + "no_data_state": "OK", + "exec_err_state": "OK", + "poll_every_seconds": 30 + } + ] +} +``` + +- `url` and `rules` are the log's identity — `check` validates them against the current environment and a fresh ruler read. +- `is_paused` records the pause state at record start (the moment `skipped` means). +- `poll_every_seconds` is the cadence the recording **actually used** (after any `--poll-interval` override). `check` derives `maxGap` from it, never from `interval_seconds`. +- `for_seconds`, `interval_seconds`, `no_data_state`, `exec_err_state` are forensic only — `check` re-resolves definitions and never reads them back. + +## Poll + +```json +{ + "type": "poll", + "rule_uid": "rule0000001", + "grafana_now": "2026-09-07T10:00:30Z", + "skew_ms": 20, + "skew_bound_ms": 40, + "latency_ms": 123, + "found": true, + "state": "inactive", + "health": "ok", + "last_evaluation": "2026-09-07T10:00:28Z", + "is_paused": false, + "histogram": { "alerting": 0, "normal": 2004 }, + "reasons": { "NoData": 1091 }, + "abnormal": [ { "labels": { "env": "prod" }, "state": "firing", "active_at": "2026-09-07T09:50:00Z", "value": "1.5" } ], + "cleared": [ "env=prod\u0001..." ], + "vanished": [] +} +``` + +Field notes: + +- `grafana_now` is the response's `Date` header — never the runner clock. +- `skew_ms`/`skew_bound_ms` are the per-poll clock-skew estimate and its uncertainty (RTT/2), in milliseconds for compactness only. +- `found: false` is an authoritative `2xx` in which this rule was absent — a transport failure is retried and never becomes a poll. +- `state`, `health`, `last_error` are raw rule-level strings, reporting-only. +- `histogram` is a verbatim copy of the response `totals`; written, never analysed. +- `reasons` counts non-empty instance reasons (`NoData`, `Error`, `KeepLast`, …); composite states stay visible only here. +- `abnormal` holds only instances whose **canonical** state is not `normal`. +- `cleared`/`vanished` are instance keys that left the bad set, resolved against the same response: `cleared` = a real recovery; `vanished` = a discontinuity, never a recovery. + +Instance keys are a sorted `k=v\n` join of labels, so they correlate across polls without hashing. + +## Stopped + +```json +{ "type": "stopped", "at": "2026-09-07T10:10:30Z" } +``` + +`at` is the recorder's own stop time. `check` compares it against `to + transitionGrace`; absent or earlier is `unobservable` — never a pass. diff --git a/grafana-alertcheck/go.mod b/grafana-alertcheck/go.mod index b0c8511ce..d5e0c88be 100644 --- a/grafana-alertcheck/go.mod +++ b/grafana-alertcheck/go.mod @@ -1,3 +1,7 @@ module github.com/smartcontractkit/chainlink-testing-framework/grafana-alertcheck go 1.26.6 + +require github.com/stretchr/testify v1.12.1 + +require go.yaml.in/yaml/v3 v3.0.5 // indirect diff --git a/grafana-alertcheck/go.sum b/grafana-alertcheck/go.sum new file mode 100644 index 000000000..c2336837e --- /dev/null +++ b/grafana-alertcheck/go.sum @@ -0,0 +1,4 @@ +github.com/stretchr/testify v1.12.1 h1:EuwCh5fleGS7H32xRwO3wRGT7DxrDhLAT6FF8MpWDWE= +github.com/stretchr/testify v1.12.1/go.mod h1:MDEgiDPPsNp5cuIrHPPCyornHKgEVbtFUmoNlxoYthg= +go.yaml.in/yaml/v3 v3.0.5 h1:N6y/pJk8buWs9NY5ERU2HSMfm+IuD/OtfdAnq6kESPw= +go.yaml.in/yaml/v3 v3.0.5/go.mod h1:HVTZu1O7/Vkt2N+BFy8Zza+lnLsABggaTM2ZpNIGuKg= diff --git a/grafana-alertcheck/internal/gate/check.go b/grafana-alertcheck/internal/gate/check.go new file mode 100644 index 000000000..ef6543b71 --- /dev/null +++ b/grafana-alertcheck/internal/gate/check.go @@ -0,0 +1,878 @@ +package gate + +import ( + "context" + "errors" + "fmt" + "io" + "os" + "sort" + "strings" + "time" +) + +// Check returns (Result, error) and no exit code: the code is a presentation +// decision the CLI makes. err != nil is exit 2 unconditionally, even alongside +// real violations; violations with err == nil is exit 1; neither is exit 0. +// Check never reads the environment — the CLI reads URL/token and passes them +// in, and the token must never reach a *flag.FlagSet. + +// countdownEvery is how often the collection loop reports what it is waiting +// for; a silent wait is indistinguishable from a hung process. +const countdownEvery = 30 * time.Second + +// recorderStopTimeout bounds the wait for the recorder's exit after SIGTERM. +// Everything after the signal is local (finish the in-flight write, sentinel, +// fsync), so this is loose; it stays a hard error because a log a writer still +// holds cannot be read. +const recorderStopTimeout = 30 * time.Second + +// recorderStopPoll is how often the wait re-checks the lock. With no wait(2) +// on a detached session leader, its exit is observable only by polling. +const recorderStopPoll = 100 * time.Millisecond + +// Config is check's whole input. It is the CLI's view of a run, and it is +// deliberately wider than Policy: Policy is the narrowed, pure-layer subset +// that reaches decide (classify.go), and the token is the field that must +// never cross that line. +type Config struct { + // URL and Token are the connection details, read from the environment by + // the CLI and never registered as flags. Token never enters the pure layer, + // an error string, or a Result. + URL, Token string + + // Alerts is REQUIRED in single-step mode and must be EMPTY in log mode: + // with a log, the header IS the alert set, and there is nothing to compare + // a second list against. + Alerts []string + Folder string + + States []State + Preexisting PreexistingPolicy + MinObserved int + AllowPaused bool + NodataIsUnobservable bool + + // From is the moment the deploy finished and To is the end of the work. + // They are different moments and both come from the work. In recorder mode + // an absent From is a hard error; in single-step mode it falls back to the + // start of this step, with a blind-interval warning. + From, To time.Time + + // Log is the path of a recording made by watch; "" selects single-step + // mode. PidFile defaults to .pid, the convention watch's parent + // writes and the only way check can reach the recorder it must stop before + // it may read the log. + Log string + PidFile string + + // There is deliberately NO PollEvery here, and `check` has no + // --poll-interval flag. In log mode the cadence comes from the header — + // the cadence the recording actually used — and a second authority would + // let an operator silently widen maxGap over evidence that was recorded at + // a different rate; in single-step mode the same process records and + // classifies, so the default cadence is the only cadence there is. + Concurrency int + Clock Clock + + // Notes is where the shell prints what an operator has to see while the + // run is in progress: the planned run time, the grace and its source, the + // countdown, the blind-interval warning. nil discards them. The library + // renders no table — the CLI owns presentation. + Notes io.Writer +} + +func (cfg Config) withDefaults() Config { + if cfg.Clock == nil { + cfg.Clock = SystemClock{} + } + if cfg.Notes == nil { + cfg.Notes = io.Discard + } + if cfg.Concurrency < 1 { + cfg.Concurrency = 1 + } + if cfg.PidFile == "" && cfg.Log != "" { + cfg.PidFile = cfg.Log + ".pid" + } + return cfg +} + +// namedAlerts returns the alert names that survive Resolve's trim-and-discard, +// so validation counts what Resolve will actually see rather than what the +// caller happened to pass (a file ending in a newline yields an empty line). +func (cfg Config) namedAlerts() []string { + out := make([]string, 0, len(cfg.Alerts)) + for _, a := range cfg.Alerts { + if strings.TrimSpace(a) != "" { + out = append(out, a) + } + } + return out +} + +// Check is the I/O shell: HTTP, signals, the pidfile, file reads, the +// countdown print. Every correctness question it touches is answered elsewhere +// (proveCoverage, decide — both pure), which is the most important seam in the +// project. A pass is exactly len(Violations) == 0 && err == nil; every error +// path leaves err non-nil. +func Check(ctx context.Context, cfg Config) (Result, error) { + cfg = cfg.withDefaults() + if err := cfg.validate(); err != nil { + return Result{}, err + } + // The Source is built here and injected into check() so every behaviour + // below is testable against a scripted fake — the same seam prepareWatch + // uses, and the reason this file needs no test-only setter. + return check(ctx, cfg, NewHTTPSource(cfg.URL, cfg.Token, cfg.Clock)) +} + +// validate runs before any network call, so a configuration mistake costs +// nothing and, more importantly, is never discovered after a ten-minute wait. +func (cfg Config) validate() error { + if cfg.URL == "" { + return errors.New("check: no grafana url") + } + if cfg.To.IsZero() { + return errors.New("check: no `to`: the end of the window is required") + } + + named := cfg.namedAlerts() + if cfg.Log == "" { + // An empty Alerts is an error — but only without a log. + if len(named) == 0 { + return errors.New("check: no alert names given and no recorded log to take them from") + } + } else if len(named) > 0 { + // The other direction: with a log, the alert set comes from the log. + // Accepting both would mean reconciling two sets, which the log being + // the one source removes entirely. + return fmt.Errorf("check: --alerts is refused with a recorded log: %s already names the alert set it recorded", cfg.Log) + } + + now := cfg.Clock.Now() + // from mirrors what check() will use, so the window checks below judge the + // window that will really be classified. + from := cfg.From + switch { + case from.IsZero() && cfg.Log != "": + // Never a warning-and-continue: falling back to the start of the check + // step reinstates exactly the blind interval the recorder exists to + // remove, which is the fail-open shape this design refuses. + return errors.New("check: no `from` in recorder mode: the deploy step must emit a completion timestamp") + case from.IsZero(): + // Single-step only. The caller sees the resulting blind interval named + // exactly, once the first observation has fixed its end. + from = now + } + + if cfg.To.Before(from) { + return fmt.Errorf("check: `to` %s is before `from` %s", cfg.To.Format(time.RFC3339), from.Format(time.RFC3339)) + } + if from.After(now.Add(fromFutureTolerance)) { + return fmt.Errorf("check: `from` %s is more than %s ahead of this runner's clock %s", + from.Format(time.RFC3339), fromFutureTolerance, now.Format(time.RFC3339)) + } + + // A `to` in the past is fine WITH a log (the collection loop is already + // done). Without one it is a request to prove a window nothing observed: + // every heartbeat gap would measure negative, and the run would report a + // proved window it never saw. + if cfg.Log == "" && !cfg.To.After(now) { + return fmt.Errorf("check: `to` %s has already passed and there is no recorded log: a window that ended before check started can only be classified from a recording", + cfg.To.Format(time.RFC3339)) + } + return nil +} + +// check is Check with the Source injected, and its body is one commented block +// per stage of a run, in the order a run performs them. +func check(ctx context.Context, cfg Config, src Source) (Result, error) { + // ---- Validate the configuration. -------------------------------------- + // Done by Check before this function is reached, except for the one part + // that needs a clock reading kept for later: the single-step fallback for + // an absent `from`. + from := cfg.From + if from.IsZero() { + from = cfg.Clock.Now() + fmt.Fprintf(cfg.Notes, "note: no `from` given; the window starts at the start of this step, %s\n", + from.Format(time.RFC3339)) + } + + // ---- Resolve the definitions from the ruler API. ---------------------- + // Unconditional, in BOTH modes. A log's header supplies the alert set as + // UIDs and the recording facts, never the rule facts: `for`, + // intervalSeconds and Kind always come from a fresh ruler read, which is + // why LoggedRule.ForSeconds is never converted back into a Definition. + version, err := src.Version(ctx) + if err != nil { + return Result{}, fmt.Errorf("read grafana version: %w", err) + } + if err := CheckGrafanaVersion(version); err != nil { + return Result{}, err + } + allDefs, err := src.Definitions(ctx) + if err != nil { + return Result{}, fmt.Errorf("read rule definitions: %w", err) + } + + // ---- With a log, validate its identity. ------------------------------- + // The header is read early — line 1 only, the one line a writer can never + // change — so a wrong URL or an unresolvable rule fails closed NOW. It is + // advisory: the authoritative header is re-read once collection ends and + // the writer has exited. + var ( + resolved []Definition + notes []string + earlyHdr Header + logHasHdr bool + rt map[string]ruleTimings + gt globalTimings + timingNote []string + ) + if cfg.Log != "" { + earlyHdr, err = ReadLogHeader(cfg.Log) + if err != nil { + return Result{}, fmt.Errorf("log identity: %w", err) + } + logHasHdr = true + resolved, notes, err = resolveFromLog(allDefs, earlyHdr, cfg) + if err == nil { + // Fail fast on a bound violation that can't change: StartedAt is + // immutable (line 1), so check 2's backstop still catches any bad + // advisory read — fail closed, never false-pass. Recorder mode only; + // single-step warns-and-passes (see below). + if from.Before(earlyHdr.StartedAt) { + return Result{}, fmt.Errorf("check: `from` %s is before recording started at %s", + from.Format(time.RFC3339), earlyHdr.StartedAt.Format(time.RFC3339)) + } + } + } else { + resolved, notes, err = Resolve(allDefs, cfg.namedAlerts(), cfg.Folder) + } + if err != nil { + return Result{}, err + } + if len(resolved) == 0 { + // Reachable only from a header with an empty rule list. Left to run, + // MinObserved would default to zero, no rule would be judged, and the + // gate would return a pass over nothing at all. + return Result{}, fmt.Errorf("check: no rules to classify") + } + for _, n := range notes { + fmt.Fprintf(cfg.Notes, "note: %s\n", n) + } + + // ---- Derive the timings, print them, fit the request budget. ---------- + if logHasHdr { + // The header is the authority for the cadence actually recorded at; + // re-deriving it from defs would compare gaps recorded at an override + // cadence against thresholds computed from the default — fail-open in + // the faster-override direction. + rt, gt, err = DeriveTimingsFromLog(earlyHdr, resolved) + if err != nil { + return Result{}, fmt.Errorf("log identity: %w", err) + } + } else { + rt, gt, timingNote = DeriveTimings(resolved, 0) + for _, n := range timingNote { + fmt.Fprintf(cfg.Notes, "note: %s\n", n) + } + } + summary, warning := StartupSummary(from, cfg.To, gt) + fmt.Fprintln(cfg.Notes, summary) + // MinObserved is printed with the plan, beside "planned run time", rather + // than after it: it is a fact about the run, not a diagnostic. Its default + // is the resolved rule count AFTER duplicate names collapse, which is + // len(resolved) by construction; decide defaults it identically, and it is + // resolved here rather than inferred from the verdict afterwards. + minObserved := cfg.MinObserved + if minObserved == 0 { + minObserved = len(resolved) + } + fmt.Fprintf(cfg.Notes, "min-observed: %d of %d resolved rule(s)\n", minObserved, len(resolved)) + if warning != "" { + fmt.Fprintf(cfg.Notes, "warning: %s\n", warning) + } + + // The measurement pass and budget check are single-step only: in recorder + // mode watch already measured and checked the budget before detaching. + var ( + header Header + initial []Poll + reducer = NewReducer() + ) + if !logHasHdr { + // StartedAt is fixed before the pass rather than after it, so the + // interval it claims to have observed can only be wider than the one + // it really saw — and the first heartbeat's own boundary gap is what + // proves that interval, not this timestamp. + startedAt := cfg.Clock.Now() + active := activeRules(resolved) + var measured map[string]time.Duration + initial, measured, err = firstObservations(ctx, src, active, reducer, cfg.Concurrency, cfg.Notes) + if err != nil { + return Result{}, err + } + if err := CheckBudget(activeTimingsOf(active, rt), measured, cfg.Concurrency); err != nil { + return Result{}, err + } + + // Single-step synthesis — how the pure layer stays unconditional. The + // shell builds the Header and later stamps the sentinel itself, so the + // sentinel and from-bounds coverage checks run exactly as they do over + // a recording and no mode flag ever reaches proveCoverage or decide. + header = Header{ + SchemaVersion: LogSchemaVersion, + URL: cfg.URL, + GrafanaVersion: version, + StartedAt: startedAt, + Rules: loggedRules(resolved, rt), + } + if from.Before(startedAt) { + // The declared blind interval: in single-step mode this is a + // warning and a pass, and ONLY here. Recorder mode keeps the + // from-bounds coverage check strict, because there the recorder + // was supposed to be watching and the gap means it was not. + fmt.Fprintf(cfg.Notes, "warning: cannot see [%s, %s) — %s before the first observation; the window is classified from %s\n", + from.Format(time.RFC3339), startedAt.Format(time.RFC3339), + startedAt.Sub(from).Round(time.Second), startedAt.Format(time.RFC3339)) + from = startedAt + } + } + + // ---- Collect the evidence. -------------------------------------------- + // Collect ONLY. No classification happens here and there is no early exit, + // even once a violation is certain: the loop always runs to + // to + transitionGrace, which is what makes "did the early exit lose the + // coverage proof?" a question that cannot be asked. + windowEnd := cfg.To.Add(gt.transitionGrace) + + var poller *livePoller + if !logHasHdr { + poller = newLivePoller(src, reducer, activeRules(resolved), rt, cfg.Concurrency, cfg.Clock.Now()) + } + collected, err := collectUntil(ctx, cfg, windowEnd, poller) + if err != nil { + // Nothing collected is classified; the count lets an operator tell a + // run that failed at once from one that failed at minute nine. + return Result{}, fmt.Errorf("collect evidence after %d poll(s): %w", len(collected), err) + } + + var ( + polls []Poll + sentinel *time.Time + ) + if logHasHdr { + // In this order and no other: signal the writer, wait for its exit, + // and only THEN read the log once. A log read while a writer can still + // append can only yield a shorter window than the one that was + // actually recorded. + heldLog, err := stopRecorder(ctx, cfg) + if err != nil { + return Result{}, err + } + header, polls, sentinel, err = ReadLog(cfg.Log) + // Held across the read so no writer can appear mid-read, then released + // (everything past here works from memory, and the drain wait is minutes). + _ = heldLog.Close() + if err != nil { + return Result{}, err + } + // The authoritative header wins: the advisory read was only a fail-fast. + resolved, _, err = resolveFromLog(allDefs, header, cfg) + if err != nil { + return Result{}, err + } + // rt is re-derived from the authoritative header. windowEnd is NOT + // recomputed: the loop already stopped at the earlier value, and moving + // it afterwards would prove a window this run did not collect. + if rt, gt, err = DeriveTimingsFromLog(header, resolved); err != nil { + return Result{}, fmt.Errorf("log identity: %w", err) + } + } else { + polls = make([]Poll, 0, len(initial)+len(collected)) + polls = append(polls, initial...) + polls = append(polls, collected...) + // The shell stamps the sentinel itself, when the collection loop + // exits: by construction that is at or after to + transitionGrace, so + // the sentinel check passes for the same reason a clean recorder stop + // does, and for no other. + stoppedAt := cfg.Clock.Now() + sentinel = &stoppedAt + } + + // ---- The drain wait. -------------------------------------------------- + // The last instance of the liveness check: did this rule evaluate through + // the end of the window? It is I/O and it is deliberately NOT part of + // proveCoverage — adding it there would put HTTP inside the pure layer and + // destroy the seam this design depends on. + drained, err := drainWait(ctx, cfg, src, resolved, header.pausedAtStart(), rt, polls, windowEnd, gt.drainTimeout) + if err != nil { + return Result{}, err + } + + // ---- Classify. -------------------------------------------------------- + pol := Policy{ + States: cfg.States, + Preexisting: cfg.Preexisting, + MinObserved: minObserved, + AllowPaused: cfg.AllowPaused, + NodataIsUnobservable: cfg.NodataIsUnobservable, + From: from, + To: cfg.To, + } + result, decideErr := decide(header, polls, sentinel, resolved, rt, gt, pol) + result, drainErr := mergeDrainTimeouts(result, drained) + + // ---- Return the Result. ----------------------------------------------- + // Both errors are joined rather than one shadowing the other: each names + // rules the other does not, and on exit 2 that list IS the answer to + // "why". + return result, errors.Join(decideErr, drainErr) +} + +// resolveFromLog is the log's identity check in practice: the URL must match +// and every header UID must still resolve against a fresh ruler read. The +// alert set is TAKEN from the log, never compared against --alerts (which the +// validator requires empty in log mode). Resolving through Resolve by uid: +// keeps one implementation of the resolution rules. +// +// Only the header-to-defs direction can fail: resolved is BUILT from the +// header, so no resolved definition can be absent from it. +func resolveFromLog(allDefs []Definition, h Header, cfg Config) ([]Definition, []string, error) { + if h.URL != cfg.URL { + return nil, nil, fmt.Errorf("log identity: %s recorded url %q but this run is configured for %q", + cfg.Log, h.URL, cfg.URL) + } + names := make([]string, 0, len(h.Rules)) + for _, lr := range h.Rules { + names = append(names, "uid:"+lr.UID) + } + resolved, notes, err := Resolve(allDefs, names, "") + if err != nil { + return nil, nil, fmt.Errorf("log identity: %s names a rule that no longer resolves: %w", cfg.Log, err) + } + return resolved, notes, nil +} + +// activeRules drops the rules whose DEFINITION says paused. They are skipped: +// never polled, never waited for, and reported from the definitions alone — a +// skipped rule has no poll records at all, so it has no heartbeats to prove +// and no IsPaused poll to detect. +func activeRules(defs []Definition) []Definition { + out := make([]Definition, 0, len(defs)) + for _, d := range defs { + if !d.IsPaused { + out = append(out, d) + } + } + return out +} + +// activeTimingsOf narrows the timings map to the rules that will actually be +// polled, which is what the request budget is spent on: a skipped rule +// consumes none of the capacity, so counting it would refuse schedules that +// fit. +func activeTimingsOf(active []Definition, rt map[string]ruleTimings) map[string]ruleTimings { + out := make(map[string]ruleTimings, len(active)) + for _, d := range active { + out[d.UID] = rt[d.UID] + } + return out +} + +// livePoller is single-step mode's collection engine: the same per-rule +// scheduler and the same Reducer the recorder uses, writing into memory +// instead of a log. Log mode has none — the recorder is doing this work in +// another process — and collectUntil takes a nil poller for it. +type livePoller struct { + src Source + reducer *Reducer + sched *Scheduler + titles map[string]string // uid -> title: poll by title, select by UID + concurrency int +} + +func newLivePoller(src Source, reducer *Reducer, active []Definition, rt map[string]ruleTimings, + concurrency int, now time.Time) *livePoller { + + titles := make(map[string]string, len(active)) + cadence := make(map[string]time.Duration, len(active)) + for _, d := range active { + titles[d.UID] = d.Title + cadence[d.UID] = rt[d.UID].pollEvery + } + return &livePoller{ + src: src, + reducer: reducer, + sched: NewScheduler(cadence, now), + titles: titles, + concurrency: concurrency, + } +} + +// poll runs one round of due rules and returns every poll that succeeded, +// alongside the first failure. +// +// The successes are NOT kept for the reason watchLoopConfig.pollBatch keeps +// its own: those go into a durable log that a later check will read, so +// dropping one would turn a single rule's transport failure into a coverage +// gap for the others. Here there is no later reader. A terminal failure during +// collection is exit 2 and check discards the whole collection, so these come +// back only to let the error say how far the run got before it stopped — +// which is the one part of it an operator can act on. +func (p *livePoller) poll(ctx context.Context, uids []string) ([]Poll, error) { + observed, obsErr := observeAll(ctx, p.src, p.titles, uids, p.concurrency) + out := make([]Poll, 0, len(uids)) + for _, uid := range uids { + obs, ok := observed[uid] + if !ok { + continue + } + out = append(out, p.reducer.Reduce(uid, obs)) + } + return out, obsErr +} + +// collectUntil is the collection loop, shared by both modes. With a poller it +// polls each rule on its own cadence; with nil it only waits, because in +// recorder mode the evidence is being written by another process. Both print +// the same countdown, because both are the same silence to an operator +// watching a job. +// +// It never classifies and never exits early. +func collectUntil(ctx context.Context, cfg Config, deadline time.Time, p *livePoller) ([]Poll, error) { + var ( + polls []Poll + lastPrint time.Time + ) + for { + now := cfg.Clock.Now() + if !now.Before(deadline) { + return polls, nil + } + if lastPrint.IsZero() || now.Sub(lastPrint) >= countdownEvery { + fmt.Fprintf(cfg.Notes, "collecting: %s until the window closes at %s\n", + deadline.Sub(now).Round(time.Second), deadline.Format(time.RFC3339)) + lastPrint = now + } + + wait := min(deadline.Sub(now), countdownEvery) + if p != nil { + // Mark before polling, against the batch's own `now`: the next poll + // is one cadence after this one was DUE, not after it returned, so + // request latency cannot make the heartbeat spacing drift towards + // maxGap (the same rule watchLoop follows). + due := p.sched.Due(now) + for _, uid := range due { + p.sched.Mark(uid, now) + } + if len(due) > 0 { + batch, err := p.poll(ctx, due) + polls = append(polls, batch...) + if err != nil { + return polls, err + } + } + if next, ok := p.sched.earliestDue(); ok { + wait = min(wait, next.Sub(cfg.Clock.Now())) + } + } + + select { + case <-ctx.Done(): + return polls, ctx.Err() + case <-cfg.Clock.After(max(wait, 0)): + } + } +} + +// stopRecorder signals the recorder and waits for it to go; the log may not be +// read until the writer has provably gone, so every failure is a hard error. +// It returns the log held under an exclusive flock, which the caller must keep +// open across ReadLog — the lock is the proof that no writer exists. +// +// Two authorities, only one of which is evidence: +// +// - the PIDFILE says whether a recording ever started (it is written only +// after the child reports ready, and removed on failure). +// - the FLOCK says whether a writer exists right now. A pidfile can go stale +// — nothing removes it on a clean --until stop, so it may name a pid +// somebody else now owns — but the kernel drops a flock when the holder +// exits, so the lock is always authoritative. +// +// So: read the pidfile to learn a recording happened, ask the lock whether it +// is still running, and signal only if it is. +func stopRecorder(ctx context.Context, cfg Config) (*os.File, error) { + pid, err := ReadPidFile(cfg.PidFile) + if err != nil { + return nil, fmt.Errorf("cannot stop the recorder: %w; a pidfile is written only once a recorder reports that it is running, so an unreadable one means the recording never started", err) + } + + log, err := os.Open(cfg.Log) + if err != nil { + return nil, fmt.Errorf("open %s to check for a writer: %w", cfg.Log, err) + } + + held, err := tryLockExclusive(log) + if err != nil { + log.Close() + return nil, err + } + if held { + // No writer. Send no signal, whatever the pidfile says — the pid may + // belong to somebody else entirely by now. Which of --until, a clean + // stop and a death ended the recording is the sentinel's question, + // answered by the coverage proof over the log this unblocks. + fmt.Fprintf(cfg.Notes, "note: no writer holds %s; the recorder has already finished\n", cfg.Log) + return log, nil + } + + // The lock is held, so a writer is alive and the pidfile's pid cannot be + // stale — the recorder that took the lock is the one the parent recorded. + gone, err := signalRecorder(pid) + if err != nil { + log.Close() + return nil, err + } + if gone { + // A live writer holds the log and the pidfile names a process that + // does not exist. That is a broken contract, not a case to reason + // around: signalling the real holder would mean guessing who it is. + log.Close() + return nil, fmt.Errorf("a writer holds %s but pidfile %s names pid %d, which does not exist: the pidfile does not name the process that holds the log", + cfg.Log, cfg.PidFile, pid) + } + + // Wait on the LOCK, not on the pid: its release is the kernel-guaranteed + // writer-is-gone event, and it carries no pid-reuse hazard. + deadline := cfg.Clock.Now().Add(recorderStopTimeout) + for { + select { + case <-ctx.Done(): + log.Close() + return nil, ctx.Err() + case <-cfg.Clock.After(recorderStopPoll): + } + + held, err := tryLockExclusive(log) + if err != nil { + log.Close() + return nil, err + } + if held { + return log, nil + } + if !cfg.Clock.Now().Before(deadline) { + log.Close() + return nil, fmt.Errorf("recorder pid %d still holds %s %s after SIGTERM; refusing to read a log a writer can still append to", + pid, cfg.Log, recorderStopTimeout) + } + } +} + +// drainVerdict is what the drain wait concluded about one rule it could not +// clear. It carries the reason as well as the prose because the two outcomes +// are genuinely different faults: drain_timeout means the rule is still there +// and still behind, rule_absent means it is gone. Collapsing both into +// drain_timeout would name the wait instead of the fault, and Reason is a +// published vocabulary that reaches the JSON output. +type drainVerdict struct { + reason UnobservableReason + note string +} + +// drainWait is the final liveness check: did each rule evaluate through the +// end of the window? A rule that cannot answer within drainTimeout is +// unobservable, never a pass. It returns one verdict per rule it could not +// clear (keyed by UID); an error only for a hard failure of the wait itself. +// +// Two kinds of rule are excluded up front because draining them could not +// change a verdict: a rule the HEADER says was paused at the window open (the +// header, not the late-resolved definitions — see Header.pausedAtStart), and a +// rule whose last poll says Found == false (already unobservable via rule_absent). +func drainWait(ctx context.Context, cfg Config, src Source, defs []Definition, pausedAtStart map[string]bool, + rt map[string]ruleTimings, polls []Poll, windowEnd time.Time, timeout time.Duration) (map[string]drainVerdict, error) { + + pending := make(map[string]string) // uid -> title, the shape observeAll wants + for _, d := range defs { + if pausedAtStart[d.UID] { + continue + } + rulePolls := pollsForRule(polls, d.UID) + if n := len(rulePolls); n > 0 && !rulePolls[n-1].Found { + continue + } + // Evidence already in hand can satisfy the wait outright: a rule whose + // recorded evaluations already reach past the end of the window has + // answered the question, and polling it again asks nothing new. + if !anyPollEvaluatedThrough(rulePolls, windowEnd) { + pending[d.UID] = d.Title + } + } + if len(pending) == 0 { + return nil, nil + } + + fmt.Fprintf(cfg.Notes, "drain wait: %d rule(s) have not yet evaluated through %s (limit %s)\n", + len(pending), windowEnd.Format(time.RFC3339), timeout) + + deadline := cfg.Clock.Now().Add(timeout) + verdicts := make(map[string]drainVerdict) + for { + uids := make([]string, 0, len(pending)) + for uid := range pending { + uids = append(uids, uid) + } + sort.Strings(uids) // deterministic request order and message order + + observed, err := observeAll(ctx, src, pending, uids, cfg.Concurrency) + if err != nil { + return nil, fmt.Errorf("drain wait: %w", err) + } + for _, uid := range uids { + obs, ok := observed[uid] + if !ok { + continue + } + rule := stateRuleByUID(obs.Rules, uid) + if rule == nil { + // A 2xx that parsed and carries no matching rule is an + // authoritative "the rule is gone" — the transport retried + // every transient failure long before this Observation + // existed. It is knowable on the FIRST poll, so waiting the + // rest of drainTimeout would spend two minutes to reach the + // same verdict under a name that describes the wait rather + // than the fault. + verdicts[uid] = drainVerdict{ + reason: ReasonRuleAbsent, + note: fmt.Sprintf("rule %q: absent from the state endpoint during the drain wait; there is no evaluation to wait for", + pending[uid]), + } + delete(pending, uid) + continue + } + if rule.IsPaused { + // A paused rule does not evaluate, so this one can never catch + // up and the rest of drainTimeout would buy nothing. The reason + // stays drain_timeout: UnobservableReason is a published + // vocabulary that reaches the JSON output, and the prose below + // is where the detail belongs. + verdicts[uid] = drainVerdict{ + reason: ReasonDrainTimeout, + note: fmt.Sprintf("rule %q: paused before it evaluated through %s, so it never will", + pending[uid], windowEnd.Format(time.RFC3339)), + } + delete(pending, uid) + continue + } + if evaluatedThrough(rule.LastEvaluation, obs.Skew, obs.SkewBound, windowEnd) { + delete(pending, uid) + } + } + if len(pending) == 0 { + return verdicts, nil + } + + now := cfg.Clock.Now() + if !now.Before(deadline) { + for uid, title := range pending { + verdicts[uid] = drainVerdict{ + reason: ReasonDrainTimeout, + note: fmt.Sprintf("rule %q: did not evaluate through %s within the %s drain limit", + title, windowEnd.Format(time.RFC3339), timeout), + } + } + return verdicts, nil + } + + // Re-ask no faster than the tightest cadence among the rules still + // pending: a rule evaluating every 60s cannot answer differently 200ms + // later, and hammering it would spend the run's request budget on + // nothing. + wait := deadline.Sub(now) + for uid := range pending { + if every := rt[uid].pollEvery; every > 0 { + wait = min(wait, every) + } + } + select { + case <-ctx.Done(): + return nil, ctx.Err() + case <-cfg.Clock.After(max(wait, 0)): + } + } +} + +// anyPollEvaluatedThrough reports whether any recorded poll of a rule already +// proves it evaluated through windowEnd. +func anyPollEvaluatedThrough(polls []Poll, windowEnd time.Time) bool { + for _, p := range polls { + if !p.Found { + continue + } + if evaluatedThrough(p.LastEvaluation, p.Skew(), p.SkewBound(), windowEnd) { + return true + } + } + return false +} + +// evaluatedThrough is the drain wait's cross-domain comparison: a Grafana +// lastEvaluation is translated by its poll's skew, and the bound is SUBTRACTED +// (the pessimistic end) so an evaluation that only *might* have reached the +// window end is not counted as having reached it. A zero lastEvaluation never +// satisfies the wait. +func evaluatedThrough(lastEval time.Time, skew, bound time.Duration, windowEnd time.Time) bool { + if lastEval.IsZero() { + return false + } + return !lastEval.Add(-skew).Add(-bound).Before(windowEnd) +} + +// mergeDrainTimeouts folds the drain wait's I/O verdicts into the pure Result, +// running after decide so that function stays pure of its arguments. It returns +// its own error rather than mutating decide's so neither hides the other: a run +// faulted by both the coverage proof and the drain wait must name both. +func mergeDrainTimeouts(res Result, drained map[string]drainVerdict) (Result, error) { + if len(drained) == 0 { + return res, nil + } + if res.Coverage == nil { + res.Coverage = make(map[string]CoverageResult, len(drained)) + } + + var names []string + for i := range res.Verdicts { + uid := res.Verdicts[i].RuleUID + verdict, ok := drained[uid] + if !ok { + continue + } + cov := res.Coverage[uid] + cov.Unobservable = true + cov.Proved = false + if cov.Reason == "" { + // The FIRST reason wins, as it does inside proveCoverage: a rule + // the coverage proof already faulted keeps the fault it was + // actually caught by. + cov.Reason = verdict.reason + } + cov.Notes = append(cov.Notes, verdict.note) + res.Coverage[uid] = cov + + if res.Verdicts[i].Outcome != OutcomeUnobservable { + names = append(names, fmt.Sprintf("%s (%s)", res.Verdicts[i].Alert, verdict.reason)) + } + res.Verdicts[i].Outcome = OutcomeUnobservable + res.Verdicts[i].Note = strings.Join(cov.Notes, "; ") + } + if len(names) == 0 { + // Every drained rule was already unobservable for an earlier reason, + // so decide's own error already stops the run. Adding a second error + // saying the same thing would only make the message longer. + return res, nil + } + return res, fmt.Errorf("gate: %d rule(s) unobservable at the drain wait: %s", len(names), strings.Join(names, "; ")) +} diff --git a/grafana-alertcheck/internal/gate/check_process.go b/grafana-alertcheck/internal/gate/check_process.go new file mode 100644 index 000000000..86dd73148 --- /dev/null +++ b/grafana-alertcheck/internal/gate/check_process.go @@ -0,0 +1,29 @@ +package gate + +import ( + "errors" + "fmt" + "syscall" +) + +// signalRecorder asks the recorder to stop. +// +// The caller must have established that a writer is alive — by taking the +// log's flock and being refused — before it calls this. Nothing removes the +// pidfile when a recorder exits cleanly, so a pid read without that proof can +// name any same-user process that has since inherited it. +// +// gone reports ESRCH. Given the lock proof, that is a broken contract rather +// than a clean stop, and stopRecorder treats it as one; the value is reported +// instead of raised here because this function knows the errno and not what +// it means. +func signalRecorder(pid int) (gone bool, err error) { + switch err := syscall.Kill(pid, syscall.SIGTERM); { + case err == nil: + return false, nil + case errors.Is(err, syscall.ESRCH): + return true, nil + default: + return false, fmt.Errorf("signal recorder pid %d: %w", pid, err) + } +} diff --git a/grafana-alertcheck/internal/gate/check_test.go b/grafana-alertcheck/internal/gate/check_test.go new file mode 100644 index 000000000..780184bd3 --- /dev/null +++ b/grafana-alertcheck/internal/gate/check_test.go @@ -0,0 +1,1268 @@ +package gate + +import ( + "bufio" + "context" + "encoding/json" + "errors" + "fmt" + "os" + "os/exec" + "path/filepath" + "strings" + "sync" + "syscall" + "testing" + "time" + + "github.com/stretchr/testify/require" +) + +// The one rule every test in this file watches, unless it says otherwise: a +// 60s evaluation interval, no `for`, not paused. Every derived value follows +// from those three numbers, and the tests assert against them by name rather +// than by magic constant: +// +// pollEvery 30s (intervalSeconds/2) +// maxGap 60s (2 x pollEvery) +// healthGrace 60s (max(maxGap, interval)) +// evalStaleAfter 120s (2 x interval) +// transitionGrace 60s (for + interval) +// drainTimeout 2m (max(2 x interval, 2m)) +const ( + checkUID = "rule-one" + checkTitle = "Rule One" + + checkPollEvery = 30 * time.Second + checkGrace = 60 * time.Second + checkDrainLimit = 2 * time.Minute +) + +func checkDef() Definition { + return Definition{ + UID: checkUID, Title: checkTitle, Folder: "F", Group: "G", + IntervalSeconds: 60, NoDataState: "OK", ExecErrState: "OK", + Kind: KindGrafanaManaged, + } +} + +// checkSource is a Source whose state answers depend on virtual time and on +// the call count, which is what a collection-loop test needs: fakeSource's +// static script cannot express "healthy for the whole window" without +// scripting every poll, and loopSource (watch_test.go) deliberately refuses +// Version and Definitions because the recorder's child never reads them. +type checkSource struct { + mu sync.Mutex + + version string + versionErr error + defs []Definition + defsErr error + + calls map[string]int + respond func(title string, call int) (Observation, error) +} + +func newCheckSource(respond func(title string, call int) (Observation, error)) *checkSource { + return &checkSource{ + version: "13.1.0", + defs: []Definition{checkDef()}, + calls: map[string]int{}, + respond: respond, + } +} + +func (s *checkSource) Version(context.Context) (string, error) { return s.version, s.versionErr } + +func (s *checkSource) Definitions(context.Context) ([]Definition, error) { return s.defs, s.defsErr } + +// RuleState answers from the responder. A nil responder means the test +// expects no state read at all — it fails with a message rather than a nil +// dereference, because "this path must not poll" is an assertion several tests +// here make on purpose. +func (s *checkSource) RuleState(_ context.Context, title string) (Observation, error) { + s.mu.Lock() + s.calls[title]++ + call := s.calls[title] + s.mu.Unlock() + if s.respond == nil { + return Observation{}, fmt.Errorf("checkSource: this test expects no state read, but %q was polled", title) + } + return s.respond(title, call) +} + +func (s *checkSource) callCount(title string) int { + s.mu.Lock() + defer s.mu.Unlock() + return s.calls[title] +} + +var _ Source = (*checkSource)(nil) + +// checkStateRule builds one state-endpoint rule whose totals agree with the +// instances it carries. That agreement is load-bearing: a totals map claiming +// normal instances that the instance list does not contain fails +// VerifyNormalInstancesVisible, which is a different failure from the one most +// of these tests are about. +func checkStateRule(lastEval time.Time, insts ...Instance) StateRule { + totals := map[string]int{} + for _, i := range insts { + totals[string(i.State)]++ + } + return StateRule{ + UID: checkUID, Title: checkTitle, Folder: "F", Group: "G", + Interval: time.Minute, State: "inactive", Health: "ok", + LastEvaluation: lastEval, Totals: totals, Instances: insts, + } +} + +// healthyObservation is a poll of a rule that evaluated at this instant, with +// no skew at all — every test that is not ABOUT skew uses zero so its +// arithmetic reads directly off the timestamps. +func healthyObservation(now time.Time, insts ...Instance) Observation { + return Observation{ + Rules: []StateRule{checkStateRule(now, insts...)}, + GrafanaNow: now, + Latency: 200 * time.Millisecond, + } +} + +// baseConfig is a single-step run over [now, now+5m]: window 5m, grace 60s, so +// the collection loop ends at now+6m. +func baseConfig(t *testing.T, clock Clock) Config { + t.Helper() + now := clock.Now() + return Config{ + URL: "https://grafana.example.com", + Alerts: []string{"uid:" + checkUID}, + From: now, + To: now.Add(5 * time.Minute), + Clock: clock, + Notes: &strings.Builder{}, + }.withDefaults() +} + +func notesOf(cfg Config) string { return cfg.Notes.(*strings.Builder).String() } + +// --------------------------------------------------------------------------- +// Configuration validation +// --------------------------------------------------------------------------- + +func TestCheckValidateRejectsBadConfigurations(t *testing.T) { + clock := newFakeClock(testNow) + base := func() Config { + return Config{ + URL: "https://grafana.example.com", + From: testNow, + To: testNow.Add(5 * time.Minute), + Clock: clock, + } + } + + tests := []struct { + name string + mutate func(*Config) + wantErr string + }{ + { + name: "no url", + mutate: func(c *Config) { c.URL = ""; c.Alerts = []string{"A"} }, + wantErr: "no grafana url", + }, + { + name: "no to", + mutate: func(c *Config) { c.To = time.Time{}; c.Alerts = []string{"A"} }, + wantErr: "no `to`", + }, + { + // An empty Alerts is an error — but only without a log. + name: "single-step without alerts", + mutate: func(c *Config) {}, + wantErr: "no alert names given", + }, + { + // An alerts file ending in a newline must not read as a named alert. + name: "single-step with only blank alert lines", + mutate: func(c *Config) { c.Alerts = []string{"", " "} }, + wantErr: "no alert names given", + }, + { + // The other direction: with a log, the log names the alert set. + name: "log mode with alerts", + mutate: func(c *Config) { c.Log = "log.jsonl"; c.Alerts = []string{"A"} }, + wantErr: "--alerts is refused with a recorded log", + }, + { + // Never a warning-and-continue. + name: "log mode without from", + mutate: func(c *Config) { c.Log = "log.jsonl"; c.From = time.Time{} }, + wantErr: "the deploy step must emit a completion timestamp", + }, + { + name: "from beyond the future tolerance", + mutate: func(c *Config) { + c.Alerts = []string{"A"} + c.From = testNow.Add(2 * time.Minute) + c.To = testNow.Add(10 * time.Minute) + }, + wantErr: "ahead of this runner's clock", + }, + { + name: "to before from", + mutate: func(c *Config) { c.Alerts = []string{"A"}; c.To = testNow.Add(-time.Minute) }, + wantErr: "is before `from`", + }, + { + // A past `to` is only "not a special mode" WITH a log: without one + // the coverage window ends before the first observation exists. + name: "single-step with a to already past", + mutate: func(c *Config) { + c.Alerts = []string{"A"} + c.From = testNow.Add(-10 * time.Minute) + c.To = testNow.Add(-time.Minute) + }, + wantErr: "can only be classified from a recording", + }, + } + + for _, tc := range tests { + t.Run(tc.name, func(t *testing.T) { + cfg := base() + tc.mutate(&cfg) + err := cfg.withDefaults().validate() + require.Errorf(t, err, "validate()") + require.Contains(t, err.Error(), tc.wantErr) + }) + } +} + +// A past `to` WITH a log is not a special mode: the collection loop's condition +// is already true and the evidence classifies immediately. No branch, and no +// refusal. +func TestCheckValidateAcceptsAPastToWithALog(t *testing.T) { + cfg := Config{ + URL: "https://grafana.example.com", + Log: "log.jsonl", + From: testNow.Add(-10 * time.Minute), + To: testNow.Add(-time.Minute), + Clock: newFakeClock(testNow), + }.withDefaults() + + require.NoError(t, cfg.validate()) + require.Equal(t, "log.jsonl.pid", cfg.PidFile) +} + +// --------------------------------------------------------------------------- +// Single-step mode +// --------------------------------------------------------------------------- + +func TestCheckSingleStepCleanWindowPasses(t *testing.T) { + clock := newVirtualClock(testNow) + cfg := baseConfig(t, clock) + src := newCheckSource(func(_ string, _ int) (Observation, error) { + return healthyObservation(clock.Now()), nil + }) + + res, err := check(context.Background(), cfg, src) + require.NoError(t, err) + // A pass is exactly this shape. + require.Empty(t, res.Violations) + require.Len(t, res.Verdicts, 1) + require.Equal(t, OutcomeClean, res.Verdicts[0].Outcome) + cov := res.Coverage[checkUID] + require.True(t, cov.Proved) + require.False(t, cov.Unobservable) + + // The collection loop ran to to+transitionGrace and no further. + windowEnd := cfg.To.Add(checkGrace) + require.False(t, clock.Now().Before(windowEnd)) + // One measurement-pass poll plus one every 30s across the 6-minute + // collection, plus the drain wait's own polls. The exact count depends on + // the scheduler's random stagger, so assert the order of magnitude a full + // window implies rather than an exact number. + require.GreaterOrEqual(t, src.callCount(checkTitle), 12) + require.Contains(t, notesOf(cfg), "planned run time") +} + +// resolve_test.go proves the collapse-note-plus-satisfied-MinObserved path at +// Resolve() directly; this drives the same shape through check() end to end — +// the two input names must collapse to one verdict, the run must pass, and the +// collapse note must reach the run's own notes, not just Resolve()'s return +// value. +func TestCheckSingleStepDuplicateAlertNamesCollapseWithNote(t *testing.T) { + clock := newVirtualClock(testNow) + cfg := baseConfig(t, clock) + cfg.Alerts = []string{"uid:" + checkUID, checkTitle} // the same rule, named two different ways + src := newCheckSource(func(_ string, _ int) (Observation, error) { + return healthyObservation(clock.Now()), nil + }) + + res, err := check(context.Background(), cfg, src) + require.NoError(t, err) + require.Len(t, res.Verdicts, 1, "the duplicate must collapse to a single rule") + require.Empty(t, res.Violations, "MinObserved must be satisfied by the post-collapse count of 1") + require.Contains(t, notesOf(cfg), "counted once") +} + +// A rule with health=error for the whole window is unobservable, exit 2 — +// driven from the real "[JD] No Job Proposals" capture (testdata/README.md), +// not a synthetic Poll table, so a change in how the real payload shapes +// health/lastError cannot slip past a hand-built fixture that happens to still +// look right. +func TestCheckSingleStepContinuousHealthErrorIsUnobservable(t *testing.T) { + body := readFixture(t, "state_health_error.json") + rules, err := ParseState(body) + require.NoError(t, err) + base := rules[0] + def := Definition{ + UID: base.UID, Title: base.Title, Folder: base.Folder, Group: base.Group, + IntervalSeconds: int(base.Interval / time.Second), NoDataState: "OK", ExecErrState: "OK", + Kind: KindGrafanaManaged, + } + + clock := newVirtualClock(testNow) + cfg := Config{ + URL: "https://grafana.example.com", Alerts: []string{"uid:" + def.UID}, + From: testNow, To: testNow.Add(5 * time.Minute), Clock: clock, Notes: &strings.Builder{}, + }.withDefaults() + + src := newCheckSource(func(_ string, _ int) (Observation, error) { + // Every field but LastEvaluation stays exactly as the real capture + // shaped it (health=error, the real lastError text, the real Error + // instance); LastEvaluation tracks the poll so staleness — a + // different coverage check — never becomes the actual cause. + r := base + r.LastEvaluation = clock.Now() + return Observation{Rules: []StateRule{r}, GrafanaNow: clock.Now(), Latency: 200 * time.Millisecond}, nil + }) + src.defs = []Definition{def} + + res, err := check(context.Background(), cfg, src) + require.Error(t, err, "continuous health=error must be unobservable") + require.Len(t, res.Verdicts, 1) + require.Equal(t, OutcomeUnobservable, res.Verdicts[0].Outcome) + require.Equal(t, ReasonHealthError, res.Coverage[def.UID].Reason) +} + +// A certain violation does not release the runner early, and it does not stop +// the gate reporting exit-1 shape — violations with a nil error. +func TestCheckSingleStepFiringInstanceReportsWithoutExitingEarly(t *testing.T) { + clock := newVirtualClock(testNow) + cfg := baseConfig(t, clock) + firing := Instance{ + Labels: map[string]string{"alertname": "Rule One", "instance": "a"}, + State: StateFiring, + ActiveAt: testNow.Add(-10 * time.Minute), // bad before the window opened + } + src := newCheckSource(func(_ string, _ int) (Observation, error) { + return healthyObservation(clock.Now(), firing), nil + }) + + res, err := check(context.Background(), cfg, src) + require.NoError(t, err, "a violation is exit 1, not an error") + require.Len(t, res.Violations, 1) + require.Equal(t, OutcomePersistentlyBad, res.Violations[0].Outcome) + require.False(t, clock.Now().Before(cfg.To.Add(checkGrace)), "exited early; collection must run to to+grace") +} + +// A newly_bad instance at from+30s gives exit 1, but ONLY after +// to+transitionGrace. The test above covers a rule already bad before the +// window opened (persistently_bad); this covers a fresh onset just inside the +// window, which must not release the runner the instant it is first observed. +func TestCheckSingleStepNewOnsetDoesNotExitEarly(t *testing.T) { + clock := newVirtualClock(testNow) + cfg := baseConfig(t, clock) + onset := testNow.Add(30 * time.Second) + src := newCheckSource(func(_ string, _ int) (Observation, error) { + now := clock.Now() + if now.Before(onset) { + return healthyObservation(now), nil + } + firing := Instance{ + Labels: map[string]string{"alertname": checkTitle, "instance": "a"}, + State: StateFiring, + ActiveAt: onset, + } + return healthyObservation(now, firing), nil + }) + + res, err := check(context.Background(), cfg, src) + require.NoError(t, err) + require.Len(t, res.Violations, 1) + require.Equal(t, OutcomeNewlyBad, res.Violations[0].Outcome) + require.False(t, clock.Now().Before(cfg.To.Add(checkGrace)), + "exited early; collection must run to to+grace even for a fresh onset at from+30s") +} + +// An ABSENT `from` in single-step mode (as opposed to recorder mode, which +// hard-errors — TestCheckValidateRejectsBadConfigurations's "log mode without +// from") falls back to the start of this check step, with the same +// declared-blind-interval warning as an explicit early `from`. +func TestCheckSingleStepAbsentFromFallsBackToStepStart(t *testing.T) { + clock := newVirtualClock(testNow) + cfg := baseConfig(t, clock) + cfg.From = time.Time{} + src := newCheckSource(func(_ string, _ int) (Observation, error) { + return healthyObservation(clock.Now()), nil + }) + + res, err := check(context.Background(), cfg, src) + require.NoError(t, err, "an absent `from` in single-step mode is a fallback, not an error") + require.Contains(t, notesOf(cfg), "no `from` given") + require.True(t, res.From.Equal(testNow)) +} + +// In single-step mode an explicit `from` earlier than the first observation is +// a DECLARED blind interval — a warning and a pass, naming the exact interval +// it cannot see. Recorder mode keeps the from-bounds coverage check strict. +func TestCheckSingleStepFromBeforeFirstObservationWarnsAndPasses(t *testing.T) { + clock := newVirtualClock(testNow) + cfg := baseConfig(t, clock) + cfg.From = testNow.Add(-2 * time.Minute) + src := newCheckSource(func(_ string, _ int) (Observation, error) { + return healthyObservation(clock.Now()), nil + }) + + res, err := check(context.Background(), cfg, src) + require.NoError(t, err) + notes := notesOf(cfg) + require.Contains(t, notes, "cannot see [") + require.Contains(t, notes, testNow.Format(time.RFC3339)) + // The classified window is the clamped one, and Result says so rather than + // reporting a window the run never proved. + require.True(t, res.From.Equal(testNow)) +} + +// The failure limit was exceeded. The measurement pass succeeds and the +// collection loop then hits a terminal failure, so this exercises the path a +// live run really takes. +func TestCheckFailClosedOnExhaustedRetries(t *testing.T) { + clock := newVirtualClock(testNow) + cfg := baseConfig(t, clock) + src := newCheckSource(func(_ string, call int) (Observation, error) { + if call > 1 { + return Observation{}, &RetryExhaustedError{Failures: 6, Cause: errors.New("connection refused")} + } + return healthyObservation(clock.Now()), nil + }) + + res, err := check(context.Background(), cfg, src) + require.Error(t, err, "the collection failure to fail closed") + require.Contains(t, err.Error(), "collect evidence") + require.Empty(t, res.Violations, "an error must never be reported as a verdict") +} + +// The resolution of the definitions failed. Both shapes — the ruler read +// itself failing, and a name that resolves to nothing. +func TestCheckFailClosedOnDefinitionResolution(t *testing.T) { + t.Run("ruler read fails", func(t *testing.T) { + clock := newVirtualClock(testNow) + cfg := baseConfig(t, clock) + src := newCheckSource(nil) + src.defsErr = errors.New("502 bad gateway") + + _, err := check(context.Background(), cfg, src) + require.Error(t, err) + require.Contains(t, err.Error(), "read rule definitions") + }) + + t.Run("unknown alert name", func(t *testing.T) { + clock := newVirtualClock(testNow) + cfg := baseConfig(t, clock) + cfg.Alerts = []string{"No Such Rule"} + src := newCheckSource(nil) + + _, err := check(context.Background(), cfg, src) + require.Error(t, err) + require.Contains(t, err.Error(), "no rule matched") + }) +} + +// The version gate: an unsupported Grafana is exit 2 before anything else is +// attempted. +func TestCheckRefusesUnsupportedGrafanaVersion(t *testing.T) { + clock := newVirtualClock(testNow) + cfg := baseConfig(t, clock) + src := newCheckSource(nil) + src.version = "12.4.0" + + _, err := check(context.Background(), cfg, src) + require.Error(t, err) + require.Contains(t, err.Error(), "unsupported grafana version") +} + +// The budget is checked against the latencies the measurement pass actually +// measured, and a schedule that cannot fit errors at START rather than +// producing a gap-riddled recording nobody can classify. +func TestCheckSingleStepRefusesAScheduleThatDoesNotFit(t *testing.T) { + clock := newVirtualClock(testNow) + cfg := baseConfig(t, clock) + src := newCheckSource(func(_ string, _ int) (Observation, error) { + obs := healthyObservation(clock.Now()) + obs.Latency = 45 * time.Second // longer than the rule's own 30s cadence + return obs, nil + }) + + _, err := check(context.Background(), cfg, src) + require.Error(t, err, "the budget check to refuse the schedule") + for _, want := range []string{"raising concurrency", "raising poll-interval", "watching fewer alerts"} { + require.Contains(t, err.Error(), want) + } +} + +// --------------------------------------------------------------------------- +// Recorder mode +// --------------------------------------------------------------------------- + +// recordedLog writes a log the way watch would have: a header, one poll every +// 30s over [start, end], and a stopped sentinel at sentinelAt. lastEvalLag is +// how far behind each poll's own GrafanaNow its lastEvaluation sits, which is +// what the drain-wait tests vary. +func recordedLog(t *testing.T, dir string, url string, startedAt, start, end, sentinelAt time.Time, lastEvalLag time.Duration) string { + t.Helper() + path := filepath.Join(dir, "log.jsonl") + clock := newFakeClock(sentinelAt) + w, err := NewWriter(path, clock) + require.NoError(t, err) + header := Header{ + URL: url, + GrafanaVersion: "13.1.0", + StartedAt: startedAt, + Rules: []LoggedRule{{ + UID: checkUID, Title: checkTitle, Folder: "F", Group: "G", + IntervalSeconds: 60, NoDataState: "OK", ExecErrState: "OK", + PollEverySeconds: checkPollEvery.Seconds(), + }}, + } + require.NoError(t, w.WriteHeader(header)) + for at := start; !at.After(end); at = at.Add(checkPollEvery) { + require.NoError(t, w.WritePoll(Poll{ + RuleUID: checkUID, GrafanaNow: at, Found: true, + State: "inactive", Health: "ok", LastEvaluation: at.Add(-lastEvalLag), + })) + } + require.NoError(t, w.Stop()) + return path +} + +// deadPid returns a pid that is guaranteed to have exited — the normal state +// of a recorder by the time check signals it, since a recorder given --until +// (or one that finished cleanly) is already gone. +func deadPid(t *testing.T) int { + t.Helper() + cmd := exec.Command("/bin/sh", "-c", "exit 0") + require.NoError(t, cmd.Start()) + pid := cmd.Process.Pid + require.NoError(t, cmd.Wait()) + return pid +} + +func writePid(t *testing.T, path, contents string) { + t.Helper() + require.NoError(t, os.WriteFile(path, []byte(contents), 0o644)) +} + +// recorderConfig points check at a recording of [testNow-1m, windowEnd+30s] +// over the window [testNow, testNow+5m]. +func recorderConfig(t *testing.T, clock Clock, logPath string) Config { + t.Helper() + return Config{ + URL: "https://grafana.example.com", + Log: logPath, + From: testNow, + To: testNow.Add(5 * time.Minute), + Clock: clock, + Notes: &strings.Builder{}, + }.withDefaults() +} + +func TestCheckRecorderModeCleanWindowPasses(t *testing.T) { + dir := t.TempDir() + windowEnd := testNow.Add(5*time.Minute + checkGrace) + logPath := recordedLog(t, dir, "https://grafana.example.com", + testNow.Add(-time.Minute), testNow.Add(-time.Minute), windowEnd.Add(30*time.Second), windowEnd.Add(30*time.Second), 0) + writePid(t, logPath+".pid", fmt.Sprintf("%d\n", deadPid(t))) + + clock := newVirtualClock(testNow.Add(time.Minute)) + cfg := recorderConfig(t, clock, logPath) + // The drain wait is satisfied from the log's own evidence, so the source + // must never be asked for a state — asserted by the nil responder. + src := newCheckSource(func(title string, _ int) (Observation, error) { + require.Fail(t, fmt.Sprintf("the drain wait polled %q although the log already proves the evaluations", title)) + return Observation{}, errors.New("unexpected poll") + }) + + res, err := check(context.Background(), cfg, src) + require.NoError(t, err) + require.Empty(t, res.Violations) + require.Len(t, res.Verdicts, 1) + require.Equal(t, OutcomeClean, res.Verdicts[0].Outcome) + require.Equal(t, "13.1.0", res.GrafanaVersion) + // The collection loop still waited out to+transitionGrace even though the + // recorder had already finished. + require.False(t, clock.Now().Before(windowEnd)) +} + +// The identity of the log is not correct. The check runs against the header +// read EARLY, so it fails before the window's wait rather than after it. +func TestCheckFailClosedOnWrongLogIdentity(t *testing.T) { + t.Run("different url", func(t *testing.T) { + dir := t.TempDir() + windowEnd := testNow.Add(5*time.Minute + checkGrace) + logPath := recordedLog(t, dir, "https://other.example.com", + testNow.Add(-time.Minute), testNow.Add(-time.Minute), windowEnd, windowEnd.Add(30*time.Second), 0) + + clock := newVirtualClock(testNow) + cfg := recorderConfig(t, clock, logPath) + _, err := check(context.Background(), cfg, newCheckSource(nil)) + require.Error(t, err) + require.Contains(t, err.Error(), "log identity") + require.True(t, clock.Now().Equal(testNow), "it must fail before the wait") + }) + + t.Run("rule no longer resolves", func(t *testing.T) { + dir := t.TempDir() + windowEnd := testNow.Add(5*time.Minute + checkGrace) + logPath := recordedLog(t, dir, "https://grafana.example.com", + testNow.Add(-time.Minute), testNow.Add(-time.Minute), windowEnd, windowEnd.Add(30*time.Second), 0) + + cfg := recorderConfig(t, newVirtualClock(testNow), logPath) + src := newCheckSource(nil) + src.defs = []Definition{{UID: "somebody-else", Title: "Other", Kind: KindGrafanaManaged, IntervalSeconds: 60}} + + _, err := check(context.Background(), cfg, src) + require.Error(t, err) + require.Contains(t, err.Error(), "log identity") + }) +} + +// `from` before the recording's StartedAt is statically knowable from the +// header (immutable line 1), so check fails closed on it BEFORE the window's +// wait — exactly like the identity check above — rather than surfacing a +// from_before_record verdict only after the drain. +func TestCheckFailFastWhenFromPrecedesRecordStart(t *testing.T) { + dir := t.TempDir() + startedAt := testNow.Add(time.Minute) // the recording opened a minute AFTER `from` + windowEnd := testNow.Add(5*time.Minute + checkGrace) + // The poll range is irrelevant to the assertion: the fail-fast reads + // StartedAt from the header alone, before any polling would matter. + logPath := recordedLog(t, dir, "https://grafana.example.com", + startedAt, startedAt, windowEnd.Add(30*time.Second), windowEnd.Add(30*time.Second), 0) + + clock := newVirtualClock(testNow) + cfg := recorderConfig(t, clock, logPath) // From = testNow, before StartedAt + _, err := check(context.Background(), cfg, newCheckSource(nil)) + require.Error(t, err) + require.Contains(t, err.Error(), "before recording started") + require.True(t, clock.Now().Equal(testNow), "it must fail before the wait") +} + +// A whole-second `from` in the same second as the recording's sub-second +// StartedAt is not a blind interval: the whole-second comparison lets the run +// proceed to a clean pass instead of the fail-fast above. +func TestCheckRecorderModeFromSameSecondAsStartedAtPasses(t *testing.T) { + dir := t.TempDir() + windowEnd := testNow.Add(5*time.Minute + checkGrace) + // StartedAt is 500ms after `from` (testNow via recorderConfig) — the same + // whole second. Polls still cover the whole window. + logPath := recordedLog(t, dir, "https://grafana.example.com", + testNow.Add(500*time.Millisecond), testNow.Add(-time.Minute), windowEnd.Add(30*time.Second), windowEnd.Add(30*time.Second), 0) + writePid(t, logPath+".pid", fmt.Sprintf("%d\n", deadPid(t))) + + clock := newVirtualClock(testNow.Add(time.Minute)) + cfg := recorderConfig(t, clock, logPath) + src := newCheckSource(func(title string, _ int) (Observation, error) { + require.Fail(t, fmt.Sprintf("the drain wait polled %q although the log already proves the evaluations", title)) + return Observation{}, errors.New("unexpected poll") + }) + + res, err := check(context.Background(), cfg, src) + require.NoError(t, err) + require.Empty(t, res.Violations) + require.Len(t, res.Verdicts, 1) + require.Equal(t, OutcomeClean, res.Verdicts[0].Outcome) +} + +// The coverage proof failed: a hole in the middle of the recording is not +// saved by healthy data at both ends. +func TestCheckFailClosedOnCoverageGap(t *testing.T) { + dir := t.TempDir() + windowEnd := testNow.Add(5*time.Minute + checkGrace) + path := filepath.Join(dir, "log.jsonl") + clock := newFakeClock(windowEnd.Add(30 * time.Second)) + w, err := NewWriter(path, clock) + require.NoError(t, err) + require.NoError(t, w.WriteHeader(Header{ + URL: "https://grafana.example.com", GrafanaVersion: "13.1.0", StartedAt: testNow.Add(-time.Minute), + Rules: []LoggedRule{{UID: checkUID, Title: checkTitle, IntervalSeconds: 60, PollEverySeconds: checkPollEvery.Seconds()}}, + })) + for at := testNow.Add(-time.Minute); !at.After(windowEnd.Add(30 * time.Second)); at = at.Add(checkPollEvery) { + // A three-minute hole in the middle of the window. + if at.After(testNow.Add(time.Minute)) && at.Before(testNow.Add(4*time.Minute)) { + continue + } + require.NoError(t, w.WritePoll(Poll{ + RuleUID: checkUID, GrafanaNow: at, Found: true, State: "inactive", Health: "ok", LastEvaluation: at, + })) + } + require.NoError(t, w.Stop()) + writePid(t, path+".pid", fmt.Sprintf("%d\n", deadPid(t))) + + cfg := recorderConfig(t, newVirtualClock(testNow), path) + res, err := check(context.Background(), cfg, newCheckSource(nil)) + require.Error(t, err, "the coverage gap to fail closed") + require.Equal(t, ReasonHeartbeatGap, res.Coverage[checkUID].Reason) + require.Equal(t, OutcomeUnobservable, res.Verdicts[0].Outcome) +} + +// An episode fully between the deploy and the start of the check: recorder +// mode must find this at the LEADING edge of the window too, right after `from` +// (the deploy's completion), not only in the middle +// (TestCheckFailClosedOnCoverageGap above). No poll exists for [from, from+3m), +// so whatever happened there is invisible to every per-poll check and only the +// coverage gap itself can catch it — the reason the recorder exists at all. +func TestCheckRecorderModeFindsAGapImmediatelyAfterTheDeploy(t *testing.T) { + dir := t.TempDir() + windowEnd := testNow.Add(5*time.Minute + checkGrace) + path := filepath.Join(dir, "log.jsonl") + clock := newFakeClock(windowEnd.Add(30 * time.Second)) + w, err := NewWriter(path, clock) + require.NoError(t, err) + require.NoError(t, w.WriteHeader(Header{ + URL: "https://grafana.example.com", GrafanaVersion: "13.1.0", StartedAt: testNow.Add(-time.Minute), + Rules: []LoggedRule{{UID: checkUID, Title: checkTitle, IntervalSeconds: 60, PollEverySeconds: checkPollEvery.Seconds()}}, + })) + gapEnd := testNow.Add(3 * time.Minute) // nothing recorded from `from` (testNow) to here + for at := gapEnd; !at.After(windowEnd.Add(30 * time.Second)); at = at.Add(checkPollEvery) { + require.NoError(t, w.WritePoll(Poll{ + RuleUID: checkUID, GrafanaNow: at, Found: true, State: "inactive", Health: "ok", LastEvaluation: at, + })) + } + require.NoError(t, w.Stop()) + writePid(t, path+".pid", fmt.Sprintf("%d\n", deadPid(t))) + + cfg := recorderConfig(t, newVirtualClock(testNow), path) + res, err := check(context.Background(), cfg, newCheckSource(nil)) + require.Error(t, err, "a hole right after the deploy hides whatever happened there") + require.Len(t, res.Verdicts, 1) + require.Equal(t, OutcomeUnobservable, res.Verdicts[0].Outcome, "never clean") +} + +// The drain limit passed. The recording itself is clean, so this isolates the +// drain wait — the rule simply never evaluates through the end of the window, +// and a rule that cannot answer that question is unobservable. +func TestCheckFailClosedOnDrainTimeout(t *testing.T) { + dir := t.TempDir() + windowEnd := testNow.Add(5*time.Minute + checkGrace) + // A 45s lag keeps every poll inside evalStaleAfter (120s), so the liveness + // coverage check is silent and only the drain wait can fail. + logPath := recordedLog(t, dir, "https://grafana.example.com", + testNow.Add(-time.Minute), testNow.Add(-time.Minute), windowEnd, windowEnd.Add(30*time.Second), 45*time.Second) + writePid(t, logPath+".pid", fmt.Sprintf("%d\n", deadPid(t))) + + clock := newVirtualClock(testNow) + cfg := recorderConfig(t, clock, logPath) + frozen := windowEnd.Add(-45 * time.Second) + src := newCheckSource(func(_ string, _ int) (Observation, error) { + now := clock.Now() + return Observation{ + Rules: []StateRule{checkStateRule(frozen)}, + GrafanaNow: now, + }, nil + }) + + res, err := check(context.Background(), cfg, src) + require.Error(t, err, "the drain limit to fail closed") + require.Equal(t, ReasonDrainTimeout, res.Coverage[checkUID].Reason) + require.Equal(t, OutcomeUnobservable, res.Verdicts[0].Outcome) + require.Contains(t, res.Verdicts[0].Note, "drain limit") + require.GreaterOrEqual(t, clock.Now().Sub(windowEnd), checkDrainLimit, + "the rule never evaluates through the window, so the drain wait must run its full limit") +} + +// A rule the state endpoint no longer serves is knowable on the FIRST drain +// poll, and the answer is rule_absent — the fault — rather than +// drain_timeout, which would only name the wait. It must not spend the whole +// drain limit to reach it. +func TestCheckDrainWaitNamesADeletedRuleAtOnce(t *testing.T) { + dir := t.TempDir() + windowEnd := testNow.Add(5*time.Minute + checkGrace) + logPath := recordedLog(t, dir, "https://grafana.example.com", + testNow.Add(-time.Minute), testNow.Add(-time.Minute), windowEnd, windowEnd.Add(30*time.Second), 45*time.Second) + writePid(t, logPath+".pid", fmt.Sprintf("%d\n", deadPid(t))) + + clock := newVirtualClock(testNow) + cfg := recorderConfig(t, clock, logPath) + // An authoritative 2xx that parsed and carries no matching rule. The + // transport retried every transient failure long before an Observation + // exists, so this is a deletion, not a hiccup. + src := newCheckSource(func(_ string, _ int) (Observation, error) { + return Observation{GrafanaNow: clock.Now()}, nil + }) + + res, err := check(context.Background(), cfg, src) + require.Error(t, err, "a deleted rule to fail closed") + require.Equal(t, ReasonRuleAbsent, res.Coverage[checkUID].Reason, "the fault, not the wait") + require.Equal(t, 1, src.callCount(checkTitle), "the absence is knowable on the first poll") + require.Less(t, clock.Now().Sub(windowEnd), checkDrainLimit) +} + +// --------------------------------------------------------------------------- +// `skipped` comes from the header, not from a definition read after the window +// --------------------------------------------------------------------------- + +// pausedAfterWindowLog records a rule that was ACTIVE at record start and that +// fired inside the window. The caller then tells check that the rule's current +// definition says paused — the state somebody set after the fact. +func pausedAfterWindowLog(t *testing.T, dir string, firesAt time.Time, end, sentinelAt time.Time) string { + t.Helper() + path := filepath.Join(dir, "log.jsonl") + w, err := NewWriter(path, newFakeClock(sentinelAt)) + require.NoError(t, err) + require.NoError(t, w.WriteHeader(Header{ + URL: "https://grafana.example.com", GrafanaVersion: "13.1.0", StartedAt: testNow.Add(-time.Minute), + Rules: []LoggedRule{{ + UID: checkUID, Title: checkTitle, IntervalSeconds: 60, + IsPaused: false, PollEverySeconds: checkPollEvery.Seconds(), + }}, + })) + firing := Instance{ + Labels: map[string]string{"alertname": checkTitle, "instance": "a"}, + State: StateFiring, + ActiveAt: firesAt, + } + for at := testNow.Add(-time.Minute); !at.After(end); at = at.Add(checkPollEvery) { + p := Poll{RuleUID: checkUID, GrafanaNow: at, Found: true, State: "inactive", Health: "ok", LastEvaluation: at} + if !at.Before(firesAt) { + p.State = "firing" + p.Abnormal = []Instance{firing} + } + require.NoError(t, w.WritePoll(p)) + } + require.NoError(t, w.Stop()) + return path +} + +// pausedAfterWindowCheck runs the timeline above. The recording reaches past +// to + transitionGrace, which for this 60s rule is to + 60s: the fresh +// definition says paused, but that no longer shrinks the grace — the header +// does, and the header says the rule was active (deriveGlobalTimings). +func pausedAfterWindowCheck(t *testing.T, allowPaused bool) (Result, error, Config) { + t.Helper() + dir := t.TempDir() + to := testNow.Add(5 * time.Minute) + end := to.Add(checkGrace + 30*time.Second) + logPath := pausedAfterWindowLog(t, dir, testNow.Add(2*time.Minute), end, end) + writePid(t, logPath+".pid", fmt.Sprintf("%d\n", deadPid(t))) + + cfg := recorderConfig(t, newVirtualClock(testNow), logPath) + cfg.AllowPaused = allowPaused + + src := newCheckSource(nil) + paused := checkDef() + paused.IsPaused = true // somebody paused it after the alert started paging + src.defs = []Definition{paused} + + res, err := check(context.Background(), cfg, src) + return res, err, cfg +} + +// The window's own evidence outranks a definition read after it closed: a rule +// that was active at record start is classified, whatever its pause state is +// by the time check resolves the definitions. +func TestCheckPausingARuleAfterTheWindowDoesNotMakeItSkipped(t *testing.T) { + res, err, _ := pausedAfterWindowCheck(t, false) + require.NoError(t, err) + require.Equal(t, OutcomeNewlyBad, res.Verdicts[0].Outcome, + "the rule was active for the whole window and fired inside it") + require.Len(t, res.Violations, 1) + require.Equal(t, OutcomeNewlyBad, res.Violations[0].Outcome) + require.NotContains(t, res.Verdicts[0].Note, "paused before the window opened") +} + +// The regression pin for the loophole this fix closed. Reading skipped from +// the post-window definition made the rule skipped; --allow-paused then made +// skipped free; and a window in which the alert fired reported exit 0. The +// default message names --allow-paused, so an operator was led straight to it. +func TestCheckAllowPausedCannotExcuseARulePausedAfterItFired(t *testing.T) { + res, err, _ := pausedAfterWindowCheck(t, true) + require.NoError(t, err) + require.NotEmpty(t, res.Violations, "the run passed over a window in which the alert fired") +} + +// The other direction, unchanged: a rule the HEADER says was paused when the +// recording opened is genuinely skipped. It has no polls, so no coverage is +// attempted for it, and --allow-paused behaves as it always did. +func TestCheckHeaderPausedRuleStaysSkipped(t *testing.T) { + dir := t.TempDir() + windowEnd := testNow.Add(5*time.Minute + checkGrace) + path := filepath.Join(dir, "log.jsonl") + w, err := NewWriter(path, newFakeClock(windowEnd.Add(30*time.Second))) + require.NoError(t, err) + // Named in the header, is_paused true, and no poll records at all — the + // shape watch writes for a rule paused before the window opened. + require.NoError(t, w.WriteHeader(Header{ + URL: "https://grafana.example.com", GrafanaVersion: "13.1.0", StartedAt: testNow.Add(-time.Minute), + Rules: []LoggedRule{{ + UID: checkUID, Title: checkTitle, IntervalSeconds: 60, + IsPaused: true, PollEverySeconds: checkPollEvery.Seconds(), + }}, + })) + require.NoError(t, w.Stop()) + writePid(t, path+".pid", fmt.Sprintf("%d\n", deadPid(t))) + + run := func(allowPaused bool) (Result, error) { + cfg := recorderConfig(t, newVirtualClock(testNow), path) + cfg.AllowPaused = allowPaused + // The definition is unpaused now; the header still decides. + src := newCheckSource(func(title string, _ int) (Observation, error) { + require.Fail(t, fmt.Sprintf("the drain wait polled skipped rule %q", title)) + return Observation{}, errors.New("unexpected poll") + }) + return check(context.Background(), cfg, src) + } + + res, err := run(false) + require.NoError(t, err, "a skipped rule is a known condition, not an inability") + require.Equal(t, OutcomeSkipped, res.Verdicts[0].Outcome) + _, ok := res.Coverage[checkUID] + require.False(t, ok, "a skipped rule has no coverage to prove") + require.Len(t, res.Violations, 1, "the MinObserved shortfall") + + res, err = run(true) + require.NoError(t, err) + require.Empty(t, res.Violations, "with --allow-paused: want a pass") +} + +// A paused rule does not evaluate, so it can never catch up: the drain wait +// must conclude on the first poll instead of spending the whole limit. +func TestCheckDrainWaitConcludesAtOnceOnAPausedRule(t *testing.T) { + dir := t.TempDir() + windowEnd := testNow.Add(5*time.Minute + checkGrace) + logPath := recordedLog(t, dir, "https://grafana.example.com", + testNow.Add(-time.Minute), testNow.Add(-time.Minute), windowEnd, windowEnd.Add(30*time.Second), 45*time.Second) + writePid(t, logPath+".pid", fmt.Sprintf("%d\n", deadPid(t))) + + clock := newVirtualClock(testNow) + cfg := recorderConfig(t, clock, logPath) + src := newCheckSource(func(_ string, _ int) (Observation, error) { + now := clock.Now() + rule := checkStateRule(windowEnd.Add(-45 * time.Second)) + rule.IsPaused = true + return Observation{Rules: []StateRule{rule}, GrafanaNow: now}, nil + }) + + res, err := check(context.Background(), cfg, src) + require.Error(t, err, "a rule that stopped evaluating must fail closed") + require.Equal(t, 1, src.callCount(checkTitle), "a paused rule can never catch up") + require.Equal(t, ReasonDrainTimeout, res.Coverage[checkUID].Reason, + "the vocabulary is published, so the detail goes in the note") + require.Contains(t, res.Verdicts[0].Note, "paused before it evaluated through") + require.Less(t, clock.Now().Sub(windowEnd), checkDrainLimit) +} + +// An absent or unparseable pidfile is never "there was nothing to stop". The +// parent writes the pidfile only once the child reports that it is recording, +// so a missing one means the recording never started — and the log must not be +// read at all. +func TestCheckRefusesToReadALogItCannotStop(t *testing.T) { + windowEnd := testNow.Add(5*time.Minute + checkGrace) + + tests := []struct { + name string + pidfile string // "" = do not create one + }{ + {name: "missing pidfile"}, + {name: "unparseable pidfile", pidfile: "not-a-pid\n"}, + {name: "empty pidfile", pidfile: ""}, + } + + for i, tc := range tests { + t.Run(tc.name, func(t *testing.T) { + dir := t.TempDir() + logPath := recordedLog(t, dir, "https://grafana.example.com", + testNow.Add(-time.Minute), testNow.Add(-time.Minute), windowEnd, windowEnd.Add(30*time.Second), 0) + if i != 0 { + writePid(t, logPath+".pid", tc.pidfile) + } + + cfg := recorderConfig(t, newVirtualClock(testNow), logPath) + _, err := check(context.Background(), cfg, newCheckSource(nil)) + require.Error(t, err) + require.Contains(t, err.Error(), "cannot stop the recorder") + }) + } +} + +// startLockHolder re-execs this test binary as a process that holds the log's +// flock and ignores SIGTERM, and returns its pid once the lock is genuinely +// held. See lockHolderEnv (watch_daemon_test.go) for why it must be a separate +// real process rather than a shell one-liner. +func startLockHolder(t *testing.T, logPath string) int { + t.Helper() + cmd := exec.Command(os.Args[0]) + cmd.Env = append(os.Environ(), lockHolderEnv+"="+logPath) + cmd.Stderr = os.Stderr + stdout, err := cmd.StdoutPipe() + require.NoError(t, err) + require.NoError(t, cmd.Start()) + t.Cleanup(func() { + _ = cmd.Process.Kill() + _ = cmd.Wait() + }) + _, err = bufio.NewReader(stdout).ReadString('\n') + require.NoError(t, err, "the lock holder never reported holding the lock") + return cmd.Process.Pid +} + +// A recorder that will not let go of the log means the log may still be +// appended to, and a log a writer can change cannot be read at all. +func TestCheckFailsWhenTheRecorderWillNotExit(t *testing.T) { + dir := t.TempDir() + windowEnd := testNow.Add(5*time.Minute + checkGrace) + logPath := recordedLog(t, dir, "https://grafana.example.com", + testNow.Add(-time.Minute), testNow.Add(-time.Minute), windowEnd, windowEnd.Add(30*time.Second), 0) + + writePid(t, logPath+".pid", fmt.Sprintf("%d\n", startLockHolder(t, logPath))) + + cfg := recorderConfig(t, newVirtualClock(testNow), logPath) + _, err := check(context.Background(), cfg, newCheckSource(nil)) + require.Error(t, err) + require.Contains(t, err.Error(), "still holds") +} + +// The regression pin for a stray SIGTERM. Nothing removes the pidfile when a +// recorder exits cleanly — the parent has returned and the child never learns +// the path — so after a --until run, a supported flow, the pidfile names a pid +// the operating system is free to hand to somebody else. The flock, not the +// pid, is what says whether a writer exists. +func TestCheckDoesNotSignalABystanderHoldingAReusedPid(t *testing.T) { + dir := t.TempDir() + windowEnd := testNow.Add(5*time.Minute + checkGrace) + logPath := recordedLog(t, dir, "https://grafana.example.com", + testNow.Add(-time.Minute), testNow.Add(-time.Minute), windowEnd.Add(30*time.Second), windowEnd.Add(30*time.Second), 0) + + // An innocent process that happens to hold the pid the finished recorder + // left behind. It does not hold the log's lock, because it is not a + // recorder. + bystander := exec.Command("sleep", "30") + require.NoError(t, bystander.Start()) + t.Cleanup(func() { + _ = bystander.Process.Kill() + _ = bystander.Wait() + }) + writePid(t, logPath+".pid", fmt.Sprintf("%d\n", bystander.Process.Pid)) + + cfg := recorderConfig(t, newVirtualClock(testNow), logPath) + _, err := check(context.Background(), cfg, newCheckSource(nil)) + require.NoError(t, err) + require.NoError(t, syscall.Kill(bystander.Process.Pid, 0), + "check signalled a process that was not the recorder") +} + +// A dead pidfile (the recorder process has already exited, holding no flock) +// with NO sentinel in the log — the shape a killed `watch` leaves behind — +// must not hang the stop wait: the flock is free immediately, so +// check reads the log at once, finds no sentinel, and fails closed. +func TestCheckDeadPidWithNoSentinelIsUnobservable(t *testing.T) { + dir := t.TempDir() + windowEnd := testNow.Add(5*time.Minute + checkGrace) + logPath := filepath.Join(dir, "log.jsonl") + w, err := NewWriter(logPath, newFakeClock(testNow)) + require.NoError(t, err) + require.NoError(t, w.WriteHeader(Header{ + URL: "https://grafana.example.com", GrafanaVersion: "13.1.0", StartedAt: testNow.Add(-time.Minute), + Rules: []LoggedRule{{ + UID: checkUID, Title: checkTitle, Folder: "F", Group: "G", + IntervalSeconds: 60, NoDataState: "OK", ExecErrState: "OK", + PollEverySeconds: checkPollEvery.Seconds(), + }}, + })) + // Healthy heartbeats all the way past windowEnd — evaluatedThrough is + // satisfied, so the drain wait needs no live re-poll — but no sentinel is + // ever written: the recorder died before it could call Stop. + for at := testNow.Add(-time.Minute); !at.After(windowEnd.Add(30 * time.Second)); at = at.Add(checkPollEvery) { + require.NoError(t, w.WritePoll(Poll{RuleUID: checkUID, GrafanaNow: at, Found: true, State: "inactive", Health: "ok", LastEvaluation: at})) + } + require.NoError(t, w.Close()) // no sentinel — a clean exit would call Stop + writePid(t, logPath+".pid", fmt.Sprintf("%d\n", deadPid(t))) + + cfg := recorderConfig(t, newVirtualClock(testNow), logPath) + res, err := check(context.Background(), cfg, newCheckSource(nil)) + require.Error(t, err, "no sentinel means the recorder never proved it ran to the end") + require.Len(t, res.Verdicts, 1) + require.Equal(t, OutcomeUnobservable, res.Verdicts[0].Outcome) +} + +// An incomplete last line gives exit 2. log_test.go's TestReadLogRejectsBadLogs +// pins ReadLog's own error and TestExitCode pins that any non-nil error maps to +// exit 2, but only this feeds a genuinely truncated log through check() itself: +// a raw file with a valid header and poll, then a torn JSON tail, exactly what +// a recorder killed mid-write leaves behind. +func TestCheckRecorderModeTruncatedLogFailsClosed(t *testing.T) { + dir := t.TempDir() + path := filepath.Join(dir, "log.jsonl") + + h := Header{ + SchemaVersion: LogSchemaVersion, + URL: "https://grafana.example.com", GrafanaVersion: "13.1.0", StartedAt: testNow.Add(-time.Minute), + Rules: []LoggedRule{{UID: checkUID, Title: checkTitle, IntervalSeconds: 60, PollEverySeconds: checkPollEvery.Seconds()}}, + } + hb, err := json.Marshal(headerRecord{Type: RecordHeader, Header: h}) + require.NoError(t, err) + pb, err := json.Marshal(pollRecord{Type: RecordPoll, Poll: Poll{ + RuleUID: checkUID, GrafanaNow: testNow, Found: true, State: "inactive", Health: "ok", LastEvaluation: testNow, + }}) + require.NoError(t, err) + content := string(hb) + "\n" + string(pb) + "\n" + `{"type":"poll","rule_ui` // torn mid-write + require.NoError(t, os.WriteFile(path, []byte(content), 0o644)) + writePid(t, path+".pid", fmt.Sprintf("%d\n", deadPid(t))) + + cfg := recorderConfig(t, newVirtualClock(testNow), path) + _, err = check(context.Background(), cfg, newCheckSource(nil)) + require.Error(t, err) + require.Contains(t, err.Error(), "unparseable") +} + +// One authority for the cadence, from check's side: maxGap comes from the +// cadence the header records, never from a re-derivation off intervalSeconds. +// The fail-open direction is the one asserted — a log recorded at 5s on a 60s +// rule must still fail on a hole a re-derived 30s maxGap would have forgiven. +func TestCheckDerivesMaxGapFromTheRecordedCadence(t *testing.T) { + dir := t.TempDir() + windowEnd := testNow.Add(5*time.Minute + checkGrace) + path := filepath.Join(dir, "log.jsonl") + w, err := NewWriter(path, newFakeClock(windowEnd.Add(30*time.Second))) + require.NoError(t, err) + require.NoError(t, w.WriteHeader(Header{ + URL: "https://grafana.example.com", GrafanaVersion: "13.1.0", StartedAt: testNow.Add(-time.Minute), + Rules: []LoggedRule{{UID: checkUID, Title: checkTitle, IntervalSeconds: 60, PollEverySeconds: 5}}, + })) + for at := testNow.Add(-time.Minute); !at.After(windowEnd.Add(30 * time.Second)); at = at.Add(5 * time.Second) { + // A 20s hole: under the recorded 5s cadence maxGap is 10s and this + // fails; under a cadence re-derived from intervalSeconds it would be + // 60s and the hole would pass unseen. + if at.After(testNow.Add(time.Minute)) && at.Before(testNow.Add(80*time.Second)) { + continue + } + require.NoError(t, w.WritePoll(Poll{ + RuleUID: checkUID, GrafanaNow: at, Found: true, State: "inactive", Health: "ok", LastEvaluation: at, + })) + } + require.NoError(t, w.Stop()) + writePid(t, path+".pid", fmt.Sprintf("%d\n", deadPid(t))) + + cfg := recorderConfig(t, newVirtualClock(testNow), path) + res, err := check(context.Background(), cfg, newCheckSource(nil)) + require.Error(t, err, "a 20s hole exceeds the 10s maxGap the recorded 5s cadence implies") + require.Equal(t, ReasonHeartbeatGap, res.Coverage[checkUID].Reason) +} + +// --------------------------------------------------------------------------- +// The pieces, in isolation +// --------------------------------------------------------------------------- + +// The drain wait's one comparison is cross-domain, and its uncertainty is +// spent in the fail-closed direction: an evaluation that only MIGHT have +// reached the end of the window does not count as one that did. +func TestEvaluatedThroughSpendsItsUncertaintyFailingClosed(t *testing.T) { + end := testNow + + tests := []struct { + name string + lastEval time.Time + skew, bound time.Duration + wantSatisfied bool + }{ + {name: "zero lastEvaluation never satisfies", lastEval: time.Time{}, wantSatisfied: false}, + {name: "exactly at the end, no skew", lastEval: end, wantSatisfied: true}, + {name: "one second short", lastEval: end.Add(-time.Second), wantSatisfied: false}, + { + name: "far enough past the end to absorb the bound", + // Grafana runs 10s fast; the reading translates back to end+5s and + // the 1s bound still leaves it past the end. + lastEval: end.Add(16 * time.Second), skew: 10 * time.Second, bound: time.Second, + wantSatisfied: true, + }, + { + name: "inside the bound is not proof", + // Translated it lands exactly on the end, so the bound can put it + // either side — which is not an evaluation THROUGH the end. + lastEval: end.Add(10 * time.Second), skew: 10 * time.Second, bound: time.Second, + wantSatisfied: false, + }, + } + + for _, tc := range tests { + t.Run(tc.name, func(t *testing.T) { + require.Equal(t, tc.wantSatisfied, evaluatedThrough(tc.lastEval, tc.skew, tc.bound, end)) + }) + } +} + +// A drain timeout on one rule and a coverage failure on another must both +// reach the message. Neither error may shadow the other. +func TestMergeDrainTimeoutsNamesEveryUnobservableRule(t *testing.T) { + res := Result{ + Coverage: map[string]CoverageResult{ + "a": {Proved: true}, + "b": {Unobservable: true, Reason: ReasonHeartbeatGap, Notes: []string{"rule \"B\": gap"}}, + }, + Verdicts: []RuleVerdict{ + {Alert: "A", RuleUID: "a", Outcome: OutcomeClean}, + {Alert: "B", RuleUID: "b", Outcome: OutcomeUnobservable}, + }, + } + + merged, err := mergeDrainTimeouts(res, map[string]drainVerdict{ + "a": {reason: ReasonDrainTimeout, note: "rule \"A\": did not evaluate through the end within the drain limit"}, + "b": {reason: ReasonDrainTimeout, note: "rule \"B\": did not evaluate through the end within the drain limit"}, + }) + require.Error(t, err, "naming the newly unobservable rule") + require.Contains(t, err.Error(), "unobservable at the drain wait") + // Only A is newly unobservable; B was already, so naming it twice would + // only lengthen the message. + require.Contains(t, err.Error(), "A ("+string(ReasonDrainTimeout)+")") + require.NotContains(t, err.Error(), "B (") + require.Equal(t, ReasonDrainTimeout, merged.Coverage["a"].Reason) + // B keeps the reason the coverage proof gave it — the FIRST reason wins, + // as it does inside proveCoverage. + require.Equal(t, ReasonHeartbeatGap, merged.Coverage["b"].Reason) + require.Equal(t, OutcomeUnobservable, merged.Verdicts[0].Outcome) +} + +// ReadLogHeader is the one read of a log a writer may still hold, so its +// refusals matter as much as its successes. +func TestReadLogHeader(t *testing.T) { + dir := t.TempDir() + + t.Run("reads line 1 while the log keeps growing", func(t *testing.T) { + path := filepath.Join(dir, "growing.jsonl") + w, err := NewWriter(path, newFakeClock(testNow)) + require.NoError(t, err) + defer w.Close() + require.NoError(t, w.WriteHeader(testHeader())) + require.NoError(t, w.WritePoll(Poll{RuleUID: "rule1", GrafanaNow: testNow, Found: true})) + + h, err := ReadLogHeader(path) + require.NoError(t, err) + require.Equal(t, testHeader().URL, h.URL) + require.Len(t, h.Rules, 1) + }) + + t.Run("a half-written header is not a header", func(t *testing.T) { + path := filepath.Join(dir, "torn.jsonl") + require.NoError(t, os.WriteFile(path, []byte(`{"type":"header","url":"htt`), 0o644)) + _, err := ReadLogHeader(path) + require.Error(t, err) + require.Contains(t, err.Error(), "no complete header") + }) + + t.Run("a wrong schema version is refused", func(t *testing.T) { + path := filepath.Join(dir, "old.jsonl") + require.NoError(t, os.WriteFile(path, []byte(`{"type":"header","schema_version":99,"url":"u"}`+"\n"), 0o644)) + _, err := ReadLogHeader(path) + require.Error(t, err) + require.Contains(t, err.Error(), "schema version 99") + }) +} diff --git a/grafana-alertcheck/internal/gate/classify.go b/grafana-alertcheck/internal/gate/classify.go new file mode 100644 index 000000000..d8d058a34 --- /dev/null +++ b/grafana-alertcheck/internal/gate/classify.go @@ -0,0 +1,610 @@ +package gate + +import ( + "fmt" + "slices" + "strings" + "time" +) + +// ReasonNodata is decide's own unobservable reason: proveCoverage never sets it +// — health=nodata is a note there, never fatal — because escalating it needs +// Policy.NodataIsUnobservable, which only decide (the Policy-holding seam) has. +const ReasonNodata UnobservableReason = "nodata" + +// Outcome is the verdict of one instance's timeline, and (after decide takes +// the worst across instances) of the rule. It is a published JSON output: the +// fail values stay distinct even though v1 maps them all to exit 1, so a later +// version can split them without breaking the interface. +type Outcome string + +const ( + OutcomeClean Outcome = "clean" + OutcomeNewlyBad Outcome = "newly_bad" + OutcomeRecovered Outcome = "recovered" + OutcomePersistentlyBad Outcome = "persistently_bad" + OutcomeFlapping Outcome = "flapping" + OutcomeSkipped Outcome = "skipped" + OutcomeUnobservable Outcome = "unobservable" +) + +// PreexistingPolicy governs only the ONE ambiguous case in the outcome table: +// an instance that was already bad when the window opened. A newly_bad or +// flapping instance is a fail under every policy, so this type only ever +// changes how `recovered` and `persistently_bad` are judged (isViolation +// below). +type PreexistingPolicy string + +const ( + // PreexistingFailUnlessRecovered is the default: a preexisting instance + // that clears and stays clear is a pass (`recovered`); one that never + // clears is still a fail (`persistently_bad`). + PreexistingFailUnlessRecovered PreexistingPolicy = "fail-unless-recovered" + // PreexistingFail makes ANY preexisting instance a fail, even one that + // recovers — for a user who wants no benefit of the doubt for a + // condition this release did not cause. + PreexistingFail PreexistingPolicy = "fail" + // PreexistingIgnore disregards a preexisting instance entirely, whether + // it recovers or stays bad for the whole window: only a genuinely NEW + // bad episode (newly_bad or flapping) can fail the rule. + PreexistingIgnore PreexistingPolicy = "ignore" +) + +// Violation is one instance whose timeline outcome counts against the run, +// after the preexisting policy has been applied (isViolation below). +type Violation struct { + Alert, RuleUID string + Outcome Outcome + State State + Health string // raw, reporting-only, like Poll.Health + LastError string + // FirstSeen is the episode's onset in the runner domain (translated by the + // poll's own skew), or `from` when preexisting — never a raw Grafana time. + FirstSeen time.Time + // ClearedAt is zero unless the episode closed via a genuine Cleared event. + ClearedAt time.Time + InstanceLabels map[string]string + // Note explains a Violation with no instance behind it — decide's synthetic + // MinObserved shortfall — and must not double as LastError (reporting-only + // rule state from a real poll). + Note string +} + +// RuleVerdict is one rule's worst-of outcome, present for every resolved rule +// (passes included) so the table shows every alert asked for. +type RuleVerdict struct { + Alert, RuleUID string + Outcome Outcome + BadFor time.Duration // total wall-clock time any instance was bad inside the window, overlaps merged + PollEvery time.Duration + Note string +} + +// Policy is decide's narrowed, pure-layer view of a Config: the classification +// knobs and the window, nothing else. No URL, no token, no I/O handles — those +// never reach the pure layer. +type Policy struct { + States []State + Preexisting PreexistingPolicy + MinObserved int + AllowPaused, NodataIsUnobservable bool + From, To time.Time +} + +// RuleThresholds is one non-skipped rule's resolved coverage thresholds, +// carried on Result so the CLI's table can print the numbers that answer "why" +// on exit 2 without decide exposing the unexported ruleTimings type itself. +type RuleThresholds struct { + MaxGap time.Duration + HealthGrace time.Duration + EvalStaleAfter time.Duration +} + +// GlobalThresholds is the run-wide half of the same information: +// transitionGrace and drainTimeout apply once, across every non-skipped +// watched rule, not per rule (globalTimings). +type GlobalThresholds struct { + TransitionGrace time.Duration + // GraceSource names, and already carries the `for` value of, the rule that + // set TransitionGrace — an operator has to see both. "none" when no rule + // contributed (TransitionGrace is then 0). + GraceSource string + DrainTimeout time.Duration +} + +// Result is decide's whole answer: everything the human table and the JSON +// output need. Coverage carries one CoverageResult per non-skipped rule — +// there is deliberately no separate Interval type anywhere in the project. +type Result struct { + From, To time.Time + GrafanaVersion string + ClockSkew time.Duration // the largest |skew| across every poll decide was given, not only the ones a rule's window actually used + // ClockSkewBound is the skew BOUND (RTT/2) of that SAME poll — not + // the largest bound seen overall, which would pair a wide bound from an + // unrelated slow request with the worst skew and misstate how tightly + // that skew is actually known. SkewHardLimit is a separate, fixed input + // validation threshold (source.go) and is not an error bound on this + // value; the CLI prints both, but must not conflate them. + ClockSkewBound time.Duration + Coverage map[string]CoverageResult + // Thresholds carries one RuleThresholds per rule Coverage also covers — + // every non-skipped rule, keyed by UID. A skipped rule has neither: it was + // never scheduled, so it has no maxGap/healthGrace/evalStaleAfter to + // report. + Thresholds map[string]RuleThresholds + Global GlobalThresholds + Verdicts []RuleVerdict + Violations []Violation +} + +// episode is one contiguous, policy-bad span of one instance's timeline, +// already resolved to the runner domain and clamped to [from, windowEnd]. It +// never crosses a genuine Cleared event: a Vanished marker freezes the state +// instead of closing the episode, which is what keeps a vanish from ever +// reading as a recovery. +type episode struct { + start, end time.Time + closedByRealClear bool +} + +// instanceTimeline accumulates one instance's walk across a rule's in-window +// polls. preexisting is decided once, the first time this key is seen bad: by +// the translated ActiveAt against `from`, never by which poll happened to +// report it first — a poll's own cadence is not evidence of when the condition +// actually began. +type instanceTimeline struct { + labels map[string]string + preexisting bool + seen bool + badOpen bool + episodeStart time.Time + lastState State + lastHealth string + lastError string + episodes []episode +} + +// runnerTime translates a Grafana-domain timestamp into the runner domain by +// undoing poll p's measured skew. The single implementation for the package — +// coverage.go's window-membership and heartbeat-boundary checks use it too. +func runnerTime(p Poll, grafanaDomain time.Time) time.Time { + return grafanaDomain.Add(-p.Skew()) +} + +// classifyRule builds every instance timeline for one rule across +// [from, windowEnd] and reduces them to the rule's worst outcome, merged +// BadFor, and the Violations the preexisting policy charges against the run. +// PURE: no I/O, no clock reads; polls need not be pre-filtered to this rule. +func classifyRule(def Definition, polls []Poll, from, windowEnd time.Time, badStates map[State]bool, pol PreexistingPolicy) (Outcome, time.Duration, []Violation) { + rulePolls := pollsForRule(polls, def.UID) + inWindow := inWindowPolls(rulePolls, from, windowEnd) + + timelines := make(map[string]*instanceTimeline) + order := make([]string, 0) + + // get backfills labels on the first real Instance: a bare Cleared/Vanished + // marker can create the timeline first (with no labels), and a later re-fire + // must not report an empty InstanceLabels. + get := func(key string, labels map[string]string) *instanceTimeline { + tl, ok := timelines[key] + if !ok { + tl = &instanceTimeline{labels: labels} + timelines[key] = tl + order = append(order, key) + return tl + } + if tl.labels == nil && labels != nil { + tl.labels = labels + } + return tl + } + + openEpisode := func(tl *instanceTimeline, start time.Time) { + tl.badOpen = true + tl.episodeStart = start + } + closeEpisode := func(tl *instanceTimeline, end time.Time, real bool) { + // inWindowPolls widens its boundary outward by the skew bound, so a + // translated end can land past windowEnd or before episodeStart; clamp + // both, otherwise mergeDurations gets an inverted span. + if end.After(windowEnd) { + end = windowEnd + } + if end.Before(tl.episodeStart) { + end = tl.episodeStart + } + tl.episodes = append(tl.episodes, episode{start: tl.episodeStart, end: end, closedByRealClear: real}) + tl.badOpen = false + } + // onsetOf resolves a fresh episode's start: the instance's own ActiveAt, + // translated to the runner domain by this poll's skew, clamped to + // [from, windowEnd]. + onsetOf := func(p Poll, inst Instance) time.Time { + start := runnerTime(p, inst.ActiveAt) + if start.Before(from) { + start = from + } + if start.After(windowEnd) { + start = windowEnd + } + return start + } + + for _, p := range inWindow { + byKey := make(map[string]Instance, len(p.Abnormal)) + for _, inst := range p.Abnormal { + byKey[instanceKey(inst.Labels)] = inst + } + + for key, inst := range byKey { + tl := get(key, inst.Labels) + bad := badStates[inst.State] + switch { + case !tl.seen: + tl.seen = true + if bad { + // Fail-closed: "preexisting" only when even the worst-case + // skew error places the onset at or before `from`; an onset + // that might be in-window must classify as a new episode. + activeAtRunner := runnerTime(p, inst.ActiveAt) + tl.preexisting = !activeAtRunner.Add(p.SkewBound()).After(from) + if tl.preexisting { + openEpisode(tl, from) + } else { + openEpisode(tl, onsetOf(p, inst)) + } + } + case bad && !tl.badOpen: + openEpisode(tl, onsetOf(p, inst)) + case !bad && tl.badOpen: + closeEpisode(tl, runnerTime(p, p.GrafanaNow), true) + } + tl.lastState, tl.lastHealth, tl.lastError = inst.State, p.Health, p.LastError + } + + for _, key := range p.Cleared { + tl := get(key, nil) + if !tl.seen { + // Cleared on first mention: the transition happened pre-window, + // with no in-window evidence it was ever bad. + tl.seen = true + continue + } + if tl.badOpen { + closeEpisode(tl, runnerTime(p, p.GrafanaNow), true) + } + tl.lastHealth, tl.lastError = p.Health, p.LastError + } + + // Vanished is a deliberate no-op: freeze badOpen/preexisting as-is, so a + // vanish while bad stays bad (never reading as a recovery). + for _, key := range p.Vanished { + tl := get(key, nil) + tl.seen = true + tl.lastHealth = p.Health + } + } + + // Map iteration order is nondeterministic; sort so Violations/BadFor output + // is stable for a given input (like log.go sorts Cleared/Vanished). + slices.Sort(order) + + var ( + outcome Outcome = OutcomeClean + badFor []episode + viols []Violation + ) + + for _, key := range order { + tl := timelines[key] + if tl.badOpen { + closeEpisode(tl, windowEnd, false) + } + if len(tl.episodes) == 0 { + continue + } + + var instOutcome Outcome + switch { + case len(tl.episodes) > 1: + instOutcome = OutcomeFlapping + case tl.preexisting: + if tl.episodes[0].closedByRealClear { + instOutcome = OutcomeRecovered + } else { + instOutcome = OutcomePersistentlyBad + } + default: + // A genuinely new onset fails whether or not it clears in-window; + // only a preexisting condition earns `recovered`. + instOutcome = OutcomeNewlyBad + } + + if outcomeRank(instOutcome) > outcomeRank(outcome) { + outcome = instOutcome + } + badFor = append(badFor, tl.episodes...) + + if isViolation(instOutcome, pol) { + var clearedAt time.Time + last := tl.episodes[len(tl.episodes)-1] + if last.closedByRealClear { + clearedAt = last.end + } + viols = append(viols, Violation{ + Alert: def.Title, + RuleUID: def.UID, + Outcome: instOutcome, + State: tl.lastState, + Health: tl.lastHealth, + LastError: tl.lastError, + FirstSeen: tl.episodes[0].start, + ClearedAt: clearedAt, + InstanceLabels: tl.labels, + }) + } + } + + return outcome, mergeDurations(badFor), viols +} + +// isViolation decides whether one instance's outcome counts against the run, +// once the preexisting policy is applied. newly_bad and flapping always do: +// both contain a genuinely new bad episode, so no policy forgives them. +// recovered and persistently_bad are, by classifyRule's construction, +// ALWAYS preexisting (a non-preexisting single episode is newly_bad instead, +// regardless of whether it clears) — so these are the only two policy can +// change, and isViolation needs no separate preexisting flag to know that. +func isViolation(o Outcome, pol PreexistingPolicy) bool { + switch o { + case OutcomeNewlyBad, OutcomeFlapping: + return true + case OutcomePersistentlyBad: + return pol != PreexistingIgnore + case OutcomeRecovered: + return pol == PreexistingFail + default: + return false + } +} + +// outcomeRank orders outcomes for classifyRule's worst-of reduction across a +// rule's instances: +// +// unobservable > {flapping, persistently_bad, newly_bad} > recovered > +// skipped > clean +// +// with unobservable and skipped applied outside this function (decide owns +// both: unobservable from CoverageResult, skipped from the log header). The +// three fail values are not ranked against each other by anything that reads +// this, so their relative order here is an arbitrary but fixed tie-break, not +// a claim that one is worse than another. +func outcomeRank(o Outcome) int { + switch o { + case OutcomeFlapping: + return 4 + case OutcomePersistentlyBad: + return 3 + case OutcomeNewlyBad: + return 2 + case OutcomeRecovered: + return 1 + default: // OutcomeClean + return 0 + } +} + +// mergeDurations sums the wall-clock time covered by a set of episodes, +// merging overlaps so a rule with several simultaneously-bad instances is +// not reported as bad for longer than it actually was. +func mergeDurations(eps []episode) time.Duration { + if len(eps) == 0 { + return 0 + } + sorted := slices.Clone(eps) + slices.SortStableFunc(sorted, func(a, b episode) int { return a.start.Compare(b.start) }) + + var total time.Duration + cur := sorted[0] + for _, e := range sorted[1:] { + if e.start.After(cur.end) { + total += cur.end.Sub(cur.start) + cur = e + continue + } + if e.end.After(cur.end) { + cur.end = e.end + } + } + total += cur.end.Sub(cur.start) + return total +} + +// pollsForRule filters polls to one rule and sorts them by GrafanaNow, the +// same selection proveCoverage uses (by UID, never by title) — stable, because +// two polls sharing a coarse Date header must not reorder nondeterministically +// in a pure function. This is the single filter+sort implementation for the +// package: proveCoverage calls it too, rather than keeping its own copy that +// could silently drift from this one's membership test. +func pollsForRule(polls []Poll, uid string) []Poll { + var out []Poll + for _, p := range polls { + if p.RuleUID == uid { + out = append(out, p) + } + } + slices.SortStableFunc(out, func(a, b Poll) int { return a.GrafanaNow.Compare(b.GrafanaNow) }) + return out +} + +// badStateSet turns Policy.States into a lookup set, defaulting to {firing} +// when the caller leaves States empty — decide applies the default itself so a +// test can pass a zero-value Policy and get the real default, rather than +// depending on the CLI to have filled it in. +func badStateSet(states []State) map[State]bool { + if len(states) == 0 { + states = []State{StateFiring} + } + set := make(map[State]bool, len(states)) + for _, s := range states { + set[s] = true + } + return set +} + +// decide is the pure seam between the collected evidence and the CLI's exit +// code, and carries nearly the whole test suite because of it. It combines +// proveCoverage's nine checks with classifyRule's timelines under one Policy, +// and owns the inability-beats-violation rule: any unobservable rule makes +// decide return a non-nil error, which the CLI maps to exit 2 unconditionally +// — never to 0 or 1, and never suppressed by a real violation found alongside +// it. +// +// Result is fully populated even when the returned error is non-nil. A caller +// must not use Violations to second-guess the error, but Result stays useful +// for the human table on exit 2. +func decide(h Header, polls []Poll, sentinel *time.Time, defs []Definition, + rt map[string]ruleTimings, gt globalTimings, pol Policy) (Result, error) { + + badStates := badStateSet(pol.States) + + result := Result{ + From: pol.From, + To: pol.To, + GrafanaVersion: h.GrafanaVersion, + Coverage: make(map[string]CoverageResult), + Thresholds: make(map[string]RuleThresholds), + Global: GlobalThresholds{ + TransitionGrace: gt.transitionGrace, + GraceSource: graceSourceOrNone(gt.graceSource), + DrainTimeout: gt.drainTimeout, + }, + } + skewSeen := false + for _, p := range polls { + s := p.Skew() + if s < 0 { + s = -s + } + // The bound travels with its own poll's skew (see Result.ClockSkewBound), + // overwritten in lockstep. >= rather than > so a bound is still assigned + // when every poll's skew is exactly 0. + if !skewSeen || s > result.ClockSkew { + result.ClockSkew = s + result.ClockSkewBound = p.SkewBound() + skewSeen = true + } + } + + minObserved := pol.MinObserved + if minObserved == 0 { + minObserved = len(defs) + } + + windowEnd := pol.To.Add(gt.transitionGrace) + + var ( + skippedRules []Definition + watchedCount int + anyUnobservable bool + unobservableNames []string + ) + + // `skipped` is decided from the header, never from defs: defs are resolved + // after the window closed, so Definition.IsPaused describes the present, + // while Header.pausedAtStart describes the window open — the only moment + // "paused before the window opened" can mean. + pausedAtStart := h.pausedAtStart() + + for _, def := range defs { + if pausedAtStart[def.UID] { + skippedRules = append(skippedRules, def) + result.Verdicts = append(result.Verdicts, RuleVerdict{ + Alert: def.Title, RuleUID: def.UID, Outcome: OutcomeSkipped, + PollEvery: rt[def.UID].pollEvery, + Note: "paused before the window opened", + }) + continue + } + watchedCount++ + + t := rt[def.UID] + cov := proveCoverage(h, polls, sentinel, t, def, pol.From, pol.To, gt.transitionGrace) + + if pol.NodataIsUnobservable && !cov.Unobservable { + inWindow := inWindowPolls(pollsForRule(polls, def.UID), pol.From, windowEnd) + if runLen, sawAny := longestHealthRun(inWindow, "nodata"); sawAny && runLen > t.healthGrace { + cov.Unobservable = true + cov.Proved = false + if cov.Reason == "" { + cov.Reason = ReasonNodata + } + cov.Notes = append(cov.Notes, fmt.Sprintf( + "rule %q: health=nodata for %s exceeds healthGrace %s and --nodata-is-unobservable is set", + def.Title, runLen, t.healthGrace)) + } + } + result.Coverage[def.UID] = cov + result.Thresholds[def.UID] = RuleThresholds{ + MaxGap: t.maxGap, + HealthGrace: t.healthGrace, + EvalStaleAfter: t.evalStaleAfter, + } + + outcome, badFor, viols := classifyRule(def, polls, pol.From, windowEnd, badStates, pol.Preexisting) + if cov.Unobservable { + outcome = OutcomeUnobservable + anyUnobservable = true + unobservableNames = append(unobservableNames, fmt.Sprintf("%s (%s)", def.Title, cov.Reason)) + } + result.Violations = append(result.Violations, viols...) + result.Verdicts = append(result.Verdicts, RuleVerdict{ + Alert: def.Title, RuleUID: def.UID, Outcome: outcome, BadFor: badFor, + PollEvery: t.pollEvery, Note: strings.Join(cov.Notes, "; "), + }) + } + + // MinObserved defaults to len(defs) (post-collapse). A shortfall counts + // toward exit 1, never exit 2, and surfaces through Violations — so it + // always produces at least one, even when no rule is paused. + counted := watchedCount + var attributable []Definition + if pol.AllowPaused { + counted += len(skippedRules) + } else { + attributable = skippedRules + } + if shortfall := minObserved - counted; shortfall > 0 { + attributed := 0 + for _, def := range attributable { + if attributed >= shortfall { + break + } + // The paused rule and --allow-paused must both be named to the + // user; both live in this one Violation, in Note — the renderer + // prints Note verbatim rather than re-deriving the hint, so the + // exact wording here is what an operator reads. + result.Violations = append(result.Violations, Violation{ + Alert: def.Title, RuleUID: def.UID, Outcome: OutcomeSkipped, + Note: "paused before the window opened; counts against --min-observed unless --allow-paused is set", + }) + attributed++ + } + for ; attributed < shortfall; attributed++ { + // No named rule explains this part of the deficit — e.g. an + // operator-supplied --min-observed above what could ever be + // resolved. Note, not LastError: LastError is reporting-only + // rule state read from a real poll, and this Violation never + // touched one. + result.Violations = append(result.Violations, Violation{ + Outcome: OutcomeSkipped, + Note: fmt.Sprintf("min-observed %d exceeds the %d rule(s) counted as observed", minObserved, counted), + }) + } + } + + if anyUnobservable { + return result, fmt.Errorf("gate: %d rule(s) unobservable: %s", len(unobservableNames), strings.Join(unobservableNames, "; ")) + } + return result, nil +} diff --git a/grafana-alertcheck/internal/gate/classify_test.go b/grafana-alertcheck/internal/gate/classify_test.go new file mode 100644 index 000000000..ad25a6ea0 --- /dev/null +++ b/grafana-alertcheck/internal/gate/classify_test.go @@ -0,0 +1,992 @@ +package gate + +import ( + "testing" + "time" + + "github.com/stretchr/testify/require" +) + +func lbl(name string) map[string]string { return map[string]string{"instance": name} } + +// abnormalPoll builds one Poll carrying a single abnormal instance, with the +// bookkeeping classifyRule needs (RuleUID, GrafanaNow, Health, Abnormal). +func abnormalPoll(uid string, at time.Time, state State, labels map[string]string, activeAt time.Time) Poll { + return Poll{ + RuleUID: uid, + GrafanaNow: at, + Found: true, + Health: "ok", + LastEvaluation: at, + Abnormal: []Instance{{Labels: labels, State: state, ActiveAt: activeAt}}, + } +} + +func clearedPoll(uid string, at time.Time, cleared ...string) Poll { + return Poll{RuleUID: uid, GrafanaNow: at, Found: true, Health: "ok", LastEvaluation: at, Cleared: cleared} +} + +func vanishedPoll(uid string, at time.Time, vanished ...string) Poll { + return Poll{RuleUID: uid, GrafanaNow: at, Found: true, Health: "ok", LastEvaluation: at, Vanished: vanished} +} + +func quietPoll(uid string, at time.Time) Poll { + return Poll{RuleUID: uid, GrafanaNow: at, Found: true, Health: "ok", LastEvaluation: at} +} + +var defaultBad = badStateSet(nil) // {firing} + +// pausedHeader builds the header decide reads `skipped` from: the pause state +// as of record start. Definition.IsPaused is deliberately NOT that authority +// — it comes from a ruler read taken after the window closed — so a test that +// wants a rule treated as skipped must say so HERE (Header.pausedAtStart). +func pausedHeader(startedAt time.Time, pausedUIDs ...string) Header { + h := Header{SchemaVersion: LogSchemaVersion, StartedAt: startedAt} + for _, uid := range pausedUIDs { + h.Rules = append(h.Rules, LoggedRule{UID: uid, IsPaused: true}) + } + return h +} + +// --- clean / newly_bad --- + +func TestClassifyRule_NoEvidenceIsClean(t *testing.T) { + from := time.Date(2026, 1, 1, 0, 0, 0, 0, time.UTC) + to := from.Add(10 * time.Minute) + def := Definition{UID: "r1", Title: "R1"} + + polls := []Poll{quietPoll("r1", from), quietPoll("r1", to)} + outcome, badFor, viols := classifyRule(def, polls, from, to, defaultBad, PreexistingFailUnlessRecovered) + require.Equal(t, OutcomeClean, outcome) + require.Zero(t, badFor) + require.Empty(t, viols) +} + +func TestClassifyRule_NewOnsetInsideWindowIsNewlyBad(t *testing.T) { + from := time.Date(2026, 1, 1, 0, 0, 0, 0, time.UTC) + to := from.Add(10 * time.Minute) + onset := from.Add(5 * time.Minute) + def := Definition{UID: "r1", Title: "R1"} + + polls := []Poll{ + quietPoll("r1", from), + abnormalPoll("r1", onset, StateFiring, lbl("a"), onset), + abnormalPoll("r1", to, StateFiring, lbl("a"), onset), + } + outcome, badFor, viols := classifyRule(def, polls, from, to, defaultBad, PreexistingFailUnlessRecovered) + require.Equal(t, OutcomeNewlyBad, outcome) + require.Equal(t, to.Sub(onset), badFor) + require.Len(t, viols, 1) + require.Equal(t, OutcomeNewlyBad, viols[0].Outcome) +} + +// A genuinely new bad episode fails even if it clears again before the window +// ends — only a PREEXISTING condition earns the benefit of `recovered`. +func TestClassifyRule_NewOnsetThatClearsStillFails(t *testing.T) { + from := time.Date(2026, 1, 1, 0, 0, 0, 0, time.UTC) + to := from.Add(10 * time.Minute) + onset := from.Add(2 * time.Minute) + clearAt := from.Add(3 * time.Minute) + def := Definition{UID: "r1", Title: "R1"} + + polls := []Poll{ + quietPoll("r1", from), + abnormalPoll("r1", onset, StateFiring, lbl("a"), onset), + clearedPoll("r1", clearAt, instanceKey(lbl("a"))), + quietPoll("r1", to), + } + outcome, _, viols := classifyRule(def, polls, from, to, defaultBad, PreexistingFailUnlessRecovered) + require.Equal(t, OutcomeNewlyBad, outcome, "even though it cleared") + require.Len(t, viols, 1) +} + +// --- recovered / persistently_bad (preexisting) --- + +func TestClassifyRule_PreexistingThatRecoversIsRecoveredAndNotAViolation(t *testing.T) { + from := time.Date(2026, 1, 1, 0, 0, 0, 0, time.UTC) + to := from.Add(10 * time.Minute) + clearAt := from.Add(8 * time.Minute) + def := Definition{UID: "r1", Title: "R1"} + key := instanceKey(lbl("a")) + + polls := []Poll{ + abnormalPoll("r1", from, StateFiring, lbl("a"), from.Add(-time.Hour)), + clearedPoll("r1", clearAt, key), + quietPoll("r1", to), + } + outcome, badFor, viols := classifyRule(def, polls, from, to, defaultBad, PreexistingFailUnlessRecovered) + require.Equal(t, OutcomeRecovered, outcome) + require.Equal(t, clearAt.Sub(from), badFor) + require.Empty(t, viols, "default policy passes a recovered preexisting instance") +} + +// The late condition: bad for 58 of a 60-minute window, clear at minute 58, +// still passes with a large BadFor — never a fail against some derived +// deadline (e.g. "must clear before 90% of the window"). +func TestClassifyRule_LateRecoveryPassesRegardlessOfHowLateItIs(t *testing.T) { + from := time.Date(2026, 1, 1, 0, 0, 0, 0, time.UTC) + to := from.Add(60 * time.Minute) + clearAt := from.Add(58 * time.Minute) + def := Definition{UID: "r1", Title: "R1"} + key := instanceKey(lbl("a")) + + polls := []Poll{ + abnormalPoll("r1", from, StateFiring, lbl("a"), from.Add(-time.Hour)), + clearedPoll("r1", clearAt, key), + quietPoll("r1", to), + } + outcome, badFor, viols := classifyRule(def, polls, from, to, defaultBad, PreexistingFailUnlessRecovered) + require.Equal(t, OutcomeRecovered, outcome, "even 58 minutes into a 60-minute window") + require.Equal(t, clearAt.Sub(from), badFor, "not a value clamped against a deadline") + require.Empty(t, viols, "there is no deadline a preexisting recovery must beat") +} + +func TestClassifyRule_PreexistingStillBadAtWindowEndIsPersistentlyBad(t *testing.T) { + from := time.Date(2026, 1, 1, 0, 0, 0, 0, time.UTC) + to := from.Add(10 * time.Minute) + def := Definition{UID: "r1", Title: "R1"} + + polls := []Poll{ + abnormalPoll("r1", from, StateFiring, lbl("a"), from.Add(-time.Hour)), + abnormalPoll("r1", to, StateFiring, lbl("a"), from.Add(-time.Hour)), + } + outcome, badFor, viols := classifyRule(def, polls, from, to, defaultBad, PreexistingFailUnlessRecovered) + require.Equal(t, OutcomePersistentlyBad, outcome) + require.Equal(t, to.Sub(from), badFor) + require.Len(t, viols, 1) + require.Equal(t, OutcomePersistentlyBad, viols[0].Outcome) +} + +// --- flapping --- + +func TestClassifyRule_ClearThenBadAgainIsFlapping(t *testing.T) { + from := time.Date(2026, 1, 1, 0, 0, 0, 0, time.UTC) + to := from.Add(10 * time.Minute) + def := Definition{UID: "r1", Title: "R1"} + key := instanceKey(lbl("a")) + + polls := []Poll{ + abnormalPoll("r1", from, StateFiring, lbl("a"), from), + clearedPoll("r1", from.Add(2*time.Minute), key), + abnormalPoll("r1", from.Add(5*time.Minute), StateFiring, lbl("a"), from.Add(5*time.Minute)), + quietPoll("r1", to), + } + outcome, _, viols := classifyRule(def, polls, from, to, defaultBad, PreexistingFailUnlessRecovered) + require.Equal(t, OutcomeFlapping, outcome) + require.Len(t, viols, 1) + require.Equal(t, OutcomeFlapping, viols[0].Outcome, "always a fail regardless of policy") +} + +// A clear and then a second bad state gives flapping, wherever the second bad +// state lands. A table over where the second onset falls — immediately after +// the clear, mid-window, and right at the +// last instant before windowEnd — closes the boundary this single fixed +// timing above cannot. +func TestClassifyRule_FlappingAtEveryTimingOfTheSecondOnset(t *testing.T) { + from := time.Date(2026, 1, 1, 0, 0, 0, 0, time.UTC) + to := from.Add(10 * time.Minute) + clearAt := from.Add(2 * time.Minute) + def := Definition{UID: "r1", Title: "R1"} + key := instanceKey(lbl("a")) + + tests := []struct { + name string + secondOnset time.Time + }{ + {"immediately after the clear", clearAt.Add(time.Second)}, + {"mid-window", from.Add(5 * time.Minute)}, + {"the last instant before windowEnd", to.Add(-time.Second)}, + } + for _, tc := range tests { + t.Run(tc.name, func(t *testing.T) { + polls := []Poll{ + abnormalPoll("r1", from, StateFiring, lbl("a"), from), + clearedPoll("r1", clearAt, key), + abnormalPoll("r1", tc.secondOnset, StateFiring, lbl("a"), tc.secondOnset), + quietPoll("r1", to), + } + outcome, _, viols := classifyRule(def, polls, from, to, defaultBad, PreexistingFailUnlessRecovered) + require.Equalf(t, OutcomeFlapping, outcome, "second onset at %s", tc.secondOnset) + require.Len(t, viols, 1) + require.Equal(t, OutcomeFlapping, viols[0].Outcome) + }) + } +} + +// --- vanished is a discontinuity, never a clear --- + +func TestClassifyRule_VanishedWhileBadStaysPersistentlyBad(t *testing.T) { + from := time.Date(2026, 1, 1, 0, 0, 0, 0, time.UTC) + to := from.Add(10 * time.Minute) + def := Definition{UID: "r1", Title: "R1"} + key := instanceKey(lbl("a")) + + polls := []Poll{ + abnormalPoll("r1", from, StateFiring, lbl("a"), from.Add(-time.Minute)), + vanishedPoll("r1", from.Add(5*time.Minute), key), + quietPoll("r1", to), + } + outcome, badFor, viols := classifyRule(def, polls, from, to, defaultBad, PreexistingFailUnlessRecovered) + require.Equal(t, OutcomePersistentlyBad, outcome, "a vanish must never read as a recovery") + require.Equal(t, to.Sub(from), badFor, "the freeze must hold the episode open to windowEnd") + require.Len(t, viols, 1) +} + +func TestClassifyRule_VanishedWhileNeverBadIsUninteresting(t *testing.T) { + from := time.Date(2026, 1, 1, 0, 0, 0, 0, time.UTC) + to := from.Add(10 * time.Minute) + def := Definition{UID: "r1", Title: "R1"} + key := instanceKey(lbl("a")) + + // Pending is abnormal (non-normal) but not in the default {firing} bad + // set, so its vanish must stay uninteresting too. + polls := []Poll{ + abnormalPoll("r1", from, StatePending, lbl("a"), from), + vanishedPoll("r1", from.Add(5*time.Minute), key), + quietPoll("r1", to), + } + outcome, badFor, viols := classifyRule(def, polls, from, to, defaultBad, PreexistingFailUnlessRecovered) + require.Equal(t, OutcomeClean, outcome) + require.Zero(t, badFor) + require.Empty(t, viols) +} + +// --- preexisting policy --- + +func TestClassifyRule_PreexistingPolicyFailFailsARecoveredInstance(t *testing.T) { + from := time.Date(2026, 1, 1, 0, 0, 0, 0, time.UTC) + to := from.Add(10 * time.Minute) + def := Definition{UID: "r1", Title: "R1"} + key := instanceKey(lbl("a")) + + polls := []Poll{ + abnormalPoll("r1", from, StateFiring, lbl("a"), from.Add(-time.Hour)), + clearedPoll("r1", from.Add(2*time.Minute), key), + quietPoll("r1", to), + } + outcome, _, viols := classifyRule(def, polls, from, to, defaultBad, PreexistingFail) + require.Equal(t, OutcomeRecovered, outcome, "the descriptive outcome does not change under policy=fail") + require.Len(t, viols, 1) + require.Equal(t, OutcomeRecovered, viols[0].Outcome, + "policy=fail gives no benefit of the doubt to a preexisting instance") +} + +func TestClassifyRule_PreexistingPolicyIgnoreForgivesPersistentlyBad(t *testing.T) { + from := time.Date(2026, 1, 1, 0, 0, 0, 0, time.UTC) + to := from.Add(10 * time.Minute) + def := Definition{UID: "r1", Title: "R1"} + + polls := []Poll{ + abnormalPoll("r1", from, StateFiring, lbl("a"), from.Add(-time.Hour)), + abnormalPoll("r1", to, StateFiring, lbl("a"), from.Add(-time.Hour)), + } + outcome, _, viols := classifyRule(def, polls, from, to, defaultBad, PreexistingIgnore) + require.Equal(t, OutcomePersistentlyBad, outcome, "the descriptive outcome does not change under policy=ignore") + require.Empty(t, viols, "policy=ignore disregards a preexisting instance even if it never recovers") +} + +func TestClassifyRule_PreexistingPolicyIgnoreStillFailsANewOnset(t *testing.T) { + from := time.Date(2026, 1, 1, 0, 0, 0, 0, time.UTC) + to := from.Add(10 * time.Minute) + onset := from.Add(5 * time.Minute) + def := Definition{UID: "r1", Title: "R1"} + + polls := []Poll{ + quietPoll("r1", from), + abnormalPoll("r1", onset, StateFiring, lbl("a"), onset), + abnormalPoll("r1", to, StateFiring, lbl("a"), onset), + } + outcome, _, viols := classifyRule(def, polls, from, to, defaultBad, PreexistingIgnore) + require.Equal(t, OutcomeNewlyBad, outcome) + require.Len(t, viols, 1, "ignore only forgives PREEXISTING badness") +} + +// --- worst-of across instances --- + +func TestClassifyRule_WorstOfMultipleInstancesWins(t *testing.T) { + from := time.Date(2026, 1, 1, 0, 0, 0, 0, time.UTC) + to := from.Add(10 * time.Minute) + def := Definition{UID: "r1", Title: "R1"} + + polls := []Poll{ + { + RuleUID: "r1", GrafanaNow: from, Found: true, Health: "ok", + Abnormal: []Instance{ + {Labels: lbl("a"), State: StateFiring, ActiveAt: from.Add(-time.Hour)}, // preexisting, will recover + {Labels: lbl("b"), State: StateFiring, ActiveAt: from}, // preexisting, will stay bad + }, + }, + clearedPoll("r1", from.Add(2*time.Minute), instanceKey(lbl("a"))), + { + RuleUID: "r1", GrafanaNow: to, Found: true, Health: "ok", + Abnormal: []Instance{{Labels: lbl("b"), State: StateFiring, ActiveAt: from}}, + }, + } + outcome, _, viols := classifyRule(def, polls, from, to, defaultBad, PreexistingFailUnlessRecovered) + require.Equal(t, OutcomePersistentlyBad, outcome, "the worse of {recovered, persistently_bad}") + require.Len(t, viols, 1) + require.Equal(t, OutcomePersistentlyBad, viols[0].Outcome) +} + +// --- decide(): skipped rules, unobservable, MinObserved, exit mapping --- + +func TestDecide_SkippedRuleNeverReachesProveCoverage(t *testing.T) { + from := time.Date(2026, 1, 1, 0, 0, 0, 0, time.UTC) + to := from.Add(10 * time.Minute) + def := Definition{UID: "r1", Title: "R1", IsPaused: true} + defs := []Definition{def} + rt := map[string]ruleTimings{"r1": newRuleTimings(30*time.Second, 60)} + gt := globalTimings{} + pol := Policy{From: from, To: to, AllowPaused: true} + + // No polls, no sentinel at all: a heartbeat_gap/no_sentinel misclassification + // here would mean proveCoverage ran for a skipped rule. + // The HEADER is what says paused — decide reads skipped from there, not + // from def.IsPaused, which is a post-window reading (Header.pausedAtStart). + res, err := decide(pausedHeader(from.Add(-time.Hour), "r1"), nil, nil, defs, rt, gt, pol) + require.NoError(t, err, "a rule paused before the window is skipped, not unobservable") + require.Len(t, res.Verdicts, 1) + require.Equal(t, OutcomeSkipped, res.Verdicts[0].Outcome) + _, ok := res.Coverage["r1"] + require.False(t, ok, "a skipped rule has no coverage to prove") +} + +func TestDecide_UnobservableRuleAlwaysReturnsAnError(t *testing.T) { + from := time.Date(2026, 1, 1, 0, 0, 0, 0, time.UTC) + to := from.Add(10 * time.Minute) + def := Definition{UID: "r1", Title: "R1"} + defs := []Definition{def} + rt := map[string]ruleTimings{"r1": newRuleTimings(30*time.Second, 60)} + gt := globalTimings{} + pol := Policy{From: from, To: to} + + // No sentinel at all: check 1 fails, so the rule is unobservable + // regardless of anything else. + res, err := decide(Header{StartedAt: from.Add(-time.Hour)}, nil, nil, defs, rt, gt, pol) + require.Error(t, err, "an unobservable rule must always fail the run") + require.Len(t, res.Verdicts, 1) + require.Equal(t, OutcomeUnobservable, res.Verdicts[0].Outcome) +} + +// Any unobservable rule means exit 2, with no exception — even alongside a +// real newly_bad. +func TestDecide_UnobservableWinsEvenAlongsideARealViolation(t *testing.T) { + from := time.Date(2026, 1, 1, 0, 0, 0, 0, time.UTC) + to := from.Add(10 * time.Minute) + onset := from.Add(5 * time.Minute) + + defBroken := Definition{UID: "broken", Title: "Broken"} + defBad := Definition{UID: "bad", Title: "Bad"} + defs := []Definition{defBroken, defBad} + rt := map[string]ruleTimings{ + "broken": newRuleTimings(30*time.Second, 60), + "bad": newRuleTimings(30*time.Second, 60), + } + gt := globalTimings{} + pol := Policy{From: from, To: to} + + var polls []Poll + for ts := from; !ts.After(to); ts = ts.Add(30 * time.Second) { + if ts.Equal(onset) || ts.After(onset) { + polls = append(polls, abnormalPoll("bad", ts, StateFiring, lbl("a"), onset)) + } else { + polls = append(polls, quietPoll("bad", ts)) + } + } + // "broken" gets no polls at all: no sentinel, no heartbeats -> unobservable. + sentinel := to + + res, err := decide(Header{StartedAt: from.Add(-time.Hour)}, polls, &sentinel, defs, rt, gt, pol) + require.Error(t, err, "one rule is unobservable") + var gotBroken, gotBad Outcome + for _, v := range res.Verdicts { + switch v.RuleUID { + case "broken": + gotBroken = v.Outcome + case "bad": + gotBad = v.Outcome + } + } + require.Equal(t, OutcomeUnobservable, gotBroken) + require.Equal(t, OutcomeNewlyBad, gotBad, + "classification still runs and is still visible in Verdicts") + require.NotEmpty(t, res.Violations, + "the newly_bad instance still reported even though the run fails on the unobservable rule") +} + +// A clean verdict with a coverage gap must never give exit 0, and recovered +// and skipped verdicts need proved coverage of the full window just as much. +// One genuinely unobservable rule ("broken", zero polls) alongside a rule with +// each of the three favorable outcomes — none of them may waive the run. +func TestDecide_UnobservableRuleWinsOverEveryFavorableOutcome(t *testing.T) { + from := time.Date(2026, 1, 1, 0, 0, 0, 0, time.UTC) + to := from.Add(10 * time.Minute) + + tests := []struct { + name string + goodPolls []Poll + pausedAtStart bool + wantOutcome Outcome + }{ + { + name: "clean", + goodPolls: func() []Poll { + var polls []Poll + for ts := from; !ts.After(to); ts = ts.Add(30 * time.Second) { + polls = append(polls, quietPoll("good", ts)) + } + return polls + }(), + wantOutcome: OutcomeClean, + }, + { + // Dense 30s-spaced polls throughout, so "good"'s own coverage + // proves clean on its own — a sparse abnormal/cleared/quiet + // triple (enough for classifyRule alone) would leave a + // heartbeat gap that muddies which rule made the run fail. + name: "recovered", + goodPolls: func() []Poll { + var polls []Poll + clearAt := from.Add(3 * time.Minute) + key := instanceKey(lbl("a")) + cleared := false + for ts := from; !ts.After(to); ts = ts.Add(30 * time.Second) { + switch { + case ts.Equal(clearAt): + polls = append(polls, clearedPoll("good", ts, key)) + cleared = true + case !cleared: + polls = append(polls, abnormalPoll("good", ts, StateFiring, lbl("a"), from.Add(-time.Hour))) + default: + polls = append(polls, quietPoll("good", ts)) + } + } + return polls + }(), + wantOutcome: OutcomeRecovered, + }, + { + name: "skipped", + pausedAtStart: true, + wantOutcome: OutcomeSkipped, + }, + } + + for _, tc := range tests { + t.Run(tc.name, func(t *testing.T) { + defs := []Definition{{UID: "good", Title: "Good"}, {UID: "broken", Title: "Broken"}} + rt := map[string]ruleTimings{ + "good": newRuleTimings(30*time.Second, 60), + "broken": newRuleTimings(30*time.Second, 60), + } + gt := globalTimings{} + pol := Policy{From: from, To: to} + + h := Header{StartedAt: from.Add(-time.Hour)} + if tc.pausedAtStart { + h.Rules = []LoggedRule{{UID: "good", IsPaused: true}} + } + + // "broken" gets no polls at all: no sentinel-worthy heartbeats, + // so it is unobservable regardless of "good". + sentinel := to + res, err := decide(h, tc.goodPolls, &sentinel, defs, rt, gt, pol) + require.Errorf(t, err, "'broken' is unobservable regardless of 'good' being %s", tc.name) + var gotGood, gotBroken Outcome + for _, v := range res.Verdicts { + switch v.RuleUID { + case "good": + gotGood = v.Outcome + case "broken": + gotBroken = v.Outcome + } + } + require.Equal(t, tc.wantOutcome, gotGood) + require.Equal(t, OutcomeUnobservable, gotBroken) + }) + } +} + +// The table above puts the coverage gap on a DIFFERENT rule from the one with +// the favorable outcome. This pins the tighter claim: a rule that +// itself recovers, but ALSO itself has a coverage gap, is still overridden to +// unobservable — the favorable classification of a rule is never a reason to +// skip that same rule's own coverage check. +func TestDecide_RecoveredOutcomeOverriddenByItsOwnCoverageGap(t *testing.T) { + from := time.Date(2026, 1, 1, 0, 0, 0, 0, time.UTC) + to := from.Add(10 * time.Minute) + def := Definition{UID: "r1", Title: "R1"} + defs := []Definition{def} + rt := map[string]ruleTimings{"r1": newRuleTimings(30*time.Second, 60)} // maxGap = 60s + gt := globalTimings{} + pol := Policy{From: from, To: to} + key := instanceKey(lbl("a")) + + polls := []Poll{ + abnormalPoll("r1", from, StateFiring, lbl("a"), from.Add(-time.Hour)), + clearedPoll("r1", from.Add(30*time.Second), key), + } + for ts := from.Add(time.Minute); !ts.After(to); ts = ts.Add(30 * time.Second) { + // A gap from from+1.5m to from+4m — well past the 60s maxGap — + // sitting entirely AFTER the clear, so classifyRule alone would + // still call this rule `recovered`. + if ts.After(from.Add(90*time.Second)) && ts.Before(from.Add(4*time.Minute)) { + continue + } + polls = append(polls, quietPoll("r1", ts)) + } + sentinel := to + + res, err := decide(Header{StartedAt: from.Add(-time.Hour)}, polls, &sentinel, defs, rt, gt, pol) + require.Error(t, err, "r1's own coverage gap must fail the run even though it recovered") + require.Len(t, res.Verdicts, 1) + require.Equal(t, OutcomeUnobservable, res.Verdicts[0].Outcome, "never recovered") + require.False(t, res.Coverage["r1"].Proved) +} + +func TestDecide_CleanWindowIsAPass(t *testing.T) { + from := time.Date(2026, 1, 1, 0, 0, 0, 0, time.UTC) + to := from.Add(10 * time.Minute) + def := Definition{UID: "r1", Title: "R1"} + defs := []Definition{def} + rt := map[string]ruleTimings{"r1": newRuleTimings(30*time.Second, 60)} + gt := globalTimings{} + pol := Policy{From: from, To: to} + + var polls []Poll + for ts := from; !ts.After(to); ts = ts.Add(30 * time.Second) { + polls = append(polls, quietPoll("r1", ts)) + } + sentinel := to + + res, err := decide(Header{StartedAt: from.Add(-time.Hour)}, polls, &sentinel, defs, rt, gt, pol) + require.NoError(t, err) + require.Empty(t, res.Violations, "a pass is exactly len(Violations)==0 && err==nil") + require.Equal(t, OutcomeClean, res.Verdicts[0].Outcome) +} + +// A pause and then an unpause inside the window, with an episode that would +// fire and resolve entirely inside the blind interval. A drain wait alone — +// "did the rule eventually evaluate through windowEnd?" — would see +// lastEvaluation catch up after the unpause and answer yes, a pass. decide() +// never runs a drain wait (that is check.go's I/O concern); this pins that +// proveCoverage's own per-poll checks already refuse the window without one, +// so a live drain wait is not what is saving this case. +func TestDecide_PauseThenUnpauseWithHiddenEpisodeGivesUnobservableNotClean(t *testing.T) { + from := time.Date(2026, 1, 1, 0, 0, 0, 0, time.UTC) + to := from.Add(20 * time.Minute) + def := Definition{UID: "r1", Title: "R1"} + defs := []Definition{def} + rt := map[string]ruleTimings{"r1": newRuleTimings(30*time.Second, 60)} + gt := globalTimings{} + pol := Policy{From: from, To: to} + + pauseStart := from.Add(5 * time.Minute) + pauseEnd := from.Add(10 * time.Minute) + + var polls []Poll + for ts := from; !ts.After(pauseStart.Add(-30 * time.Second)); ts = ts.Add(30 * time.Second) { + polls = append(polls, quietPoll("r1", ts)) + } + for ts := pauseStart; !ts.After(pauseEnd); ts = ts.Add(30 * time.Second) { + // No fire/resolve is ever observed here: the rule was not + // evaluating, so any real episode inside this stretch is invisible + // to every poll. + polls = append(polls, Poll{RuleUID: "r1", GrafanaNow: ts, Found: true, Health: "ok", IsPaused: true, LastEvaluation: pauseStart}) + } + for ts := pauseEnd.Add(30 * time.Second); !ts.After(to); ts = ts.Add(30 * time.Second) { + // Evaluations resume and catch straight up — a drain wait's final + // "did it reach windowEnd" question would answer yes. + polls = append(polls, quietPoll("r1", ts)) + } + sentinel := to + + res, err := decide(Header{StartedAt: from.Add(-time.Hour)}, polls, &sentinel, defs, rt, gt, pol) + require.Error(t, err, "the pause-then-unpause blind interval must fail closed") + require.Len(t, res.Verdicts, 1) + require.Equal(t, OutcomeUnobservable, res.Verdicts[0].Outcome, "never clean") + require.False(t, res.Coverage["r1"].Proved) +} + +// --- MinObserved shortfall --- + +func TestDecide_SkippedOnlyShortfallProducesAViolationWithoutAnError(t *testing.T) { + from := time.Date(2026, 1, 1, 0, 0, 0, 0, time.UTC) + to := from.Add(10 * time.Minute) + + watched := Definition{UID: "watched", Title: "Watched"} + paused := Definition{UID: "paused", Title: "Paused", IsPaused: true} + defs := []Definition{watched, paused} + rt := map[string]ruleTimings{ + "watched": newRuleTimings(30*time.Second, 60), + "paused": newRuleTimings(30*time.Second, 60), + } + gt := globalTimings{} + // MinObserved defaults to len(defs) = 2, but only "watched" is observable. + pol := Policy{From: from, To: to} + + var polls []Poll + for ts := from; !ts.After(to); ts = ts.Add(30 * time.Second) { + polls = append(polls, quietPoll("watched", ts)) + } + sentinel := to + + res, err := decide(pausedHeader(from.Add(-time.Hour), "paused"), polls, &sentinel, defs, rt, gt, pol) + require.NoError(t, err, "a shortfall caused only by a skipped rule is exit 1, not exit 2") + require.Len(t, res.Violations, 1) + v := res.Violations[0] + require.Equal(t, OutcomeSkipped, v.Outcome) + require.Equal(t, "paused", v.RuleUID) + require.Equal(t, "Paused", v.Alert) + require.NotEmpty(t, v.Note, + "the shortfall reason must not be smuggled into LastError") +} + +// An operator-supplied MinObserved that exceeds what could ever be resolved is +// still a shortfall, even with zero paused rules to blame it on — it must not +// silently read as a pass. +func TestDecide_ExplicitMinObservedShortfallWithNoPausedRuleStillProducesAViolation(t *testing.T) { + from := time.Date(2026, 1, 1, 0, 0, 0, 0, time.UTC) + to := from.Add(10 * time.Minute) + + def := Definition{UID: "r1", Title: "R1"} + defs := []Definition{def} + rt := map[string]ruleTimings{"r1": newRuleTimings(30*time.Second, 60)} + gt := globalTimings{} + pol := Policy{From: from, To: to, MinObserved: 3} // only one rule will ever be resolved + + var polls []Poll + for ts := from; !ts.After(to); ts = ts.Add(30 * time.Second) { + polls = append(polls, quietPoll("r1", ts)) + } + sentinel := to + + res, err := decide(Header{StartedAt: from.Add(-time.Hour)}, polls, &sentinel, defs, rt, gt, pol) + require.NoError(t, err, "an unmet MinObserved is exit 1, never exit 2") + require.Len(t, res.Violations, 2, "the shortfall (3-1=2) must surface directly rather than pass silently") + for _, v := range res.Violations { + require.Equal(t, OutcomeSkipped, v.Outcome) + } +} + +func TestDecide_AllowPausedSuppressesTheShortfall(t *testing.T) { + from := time.Date(2026, 1, 1, 0, 0, 0, 0, time.UTC) + to := from.Add(10 * time.Minute) + + watched := Definition{UID: "watched", Title: "Watched"} + paused := Definition{UID: "paused", Title: "Paused", IsPaused: true} + defs := []Definition{watched, paused} + rt := map[string]ruleTimings{ + "watched": newRuleTimings(30*time.Second, 60), + "paused": newRuleTimings(30*time.Second, 60), + } + gt := globalTimings{} + pol := Policy{From: from, To: to, AllowPaused: true} + + var polls []Poll + for ts := from; !ts.After(to); ts = ts.Add(30 * time.Second) { + polls = append(polls, quietPoll("watched", ts)) + } + sentinel := to + + res, err := decide(pausedHeader(from.Add(-time.Hour), "paused"), polls, &sentinel, defs, rt, gt, pol) + require.NoError(t, err) + require.Empty(t, res.Violations, "--allow-paused must suppress the shortfall entirely") +} + +// --- nodata escalation (decide's own Policy-driven check) --- + +func TestDecide_NodataIsUnobservableEscalatesASustainedRun(t *testing.T) { + from := time.Date(2026, 1, 1, 0, 0, 0, 0, time.UTC) + to := from.Add(10 * time.Minute) + def := Definition{UID: "r1", Title: "R1"} + defs := []Definition{def} + rt := map[string]ruleTimings{"r1": newRuleTimings(30*time.Second, 60)} // healthGrace = max(60s,60s) = 60s + gt := globalTimings{} + pol := Policy{From: from, To: to, NodataIsUnobservable: true} + + var polls []Poll + for ts := from; !ts.After(to); ts = ts.Add(30 * time.Second) { + polls = append(polls, Poll{RuleUID: "r1", GrafanaNow: ts, Found: true, Health: "nodata", LastEvaluation: ts}) + } + sentinel := to + + res, err := decide(Header{StartedAt: from.Add(-time.Hour)}, polls, &sentinel, defs, rt, gt, pol) + require.Error(t, err, "a sustained nodata run must be unobservable under --nodata-is-unobservable") + require.Equal(t, ReasonNodata, res.Coverage["r1"].Reason) +} + +func TestDecide_NodataIsANoteByDefault(t *testing.T) { + from := time.Date(2026, 1, 1, 0, 0, 0, 0, time.UTC) + to := from.Add(10 * time.Minute) + def := Definition{UID: "r1", Title: "R1"} + defs := []Definition{def} + rt := map[string]ruleTimings{"r1": newRuleTimings(30*time.Second, 60)} + gt := globalTimings{} + pol := Policy{From: from, To: to} // NodataIsUnobservable defaults to false + + var polls []Poll + for ts := from; !ts.After(to); ts = ts.Add(30 * time.Second) { + polls = append(polls, Poll{RuleUID: "r1", GrafanaNow: ts, Found: true, Health: "nodata", LastEvaluation: ts}) + } + sentinel := to + + res, err := decide(Header{StartedAt: from.Add(-time.Hour)}, polls, &sentinel, defs, rt, gt, pol) + require.NoError(t, err, "96%% of the fleet runs no_data_state:OK and must not fail by default") + require.False(t, res.Coverage["r1"].Unobservable) +} + +// --- preexisting is decided by ActiveAt, not poll timing --- + +// An instance whose true onset (ActiveAt) falls strictly inside the window — +// even though the first poll that happens to observe it already shows it bad — +// must never be treated as preexisting. If it then clears, that is newly_bad +// (exit 1), not recovered (exit 0). +func TestClassifyRule_OnsetBetweenFromAndFirstPollIsNewlyBadNotRecovered(t *testing.T) { + from := time.Date(2026, 1, 1, 0, 0, 0, 0, time.UTC) + to := from.Add(10 * time.Minute) + onset := from.Add(1 * time.Minute) // the true onset, strictly after `from` + firstPoll := from.Add(2 * time.Minute) // the first poll that happens to observe it + clearAt := from.Add(5 * time.Minute) + def := Definition{UID: "r1", Title: "R1"} + key := instanceKey(lbl("a")) + + polls := []Poll{ + abnormalPoll("r1", firstPoll, StateFiring, lbl("a"), onset), + clearedPoll("r1", clearAt, key), + quietPoll("r1", to), + } + outcome, badFor, viols := classifyRule(def, polls, from, to, defaultBad, PreexistingFailUnlessRecovered) + require.Equal(t, OutcomeNewlyBad, outcome, + "the onset is after `from`, so it is not preexisting even though the FIRST in-window poll already observes it bad") + require.Len(t, viols, 1) + require.Equal(t, OutcomeNewlyBad, viols[0].Outcome) + require.Equal(t, clearAt.Sub(onset), badFor, "BadFor must count from the true onset, not from `from`") +} + +// TestClassifyRule_OnsetJustBeforeFromIsPreexisting is the mirror check: an +// onset at or before `from` (even if the first poll is later) is genuinely +// preexisting and, if it clears, is `recovered`. +func TestClassifyRule_OnsetJustBeforeFromIsPreexisting(t *testing.T) { + from := time.Date(2026, 1, 1, 0, 0, 0, 0, time.UTC) + to := from.Add(10 * time.Minute) + onset := from.Add(-time.Minute) + firstPoll := from.Add(2 * time.Minute) + clearAt := from.Add(5 * time.Minute) + def := Definition{UID: "r1", Title: "R1"} + key := instanceKey(lbl("a")) + + polls := []Poll{ + abnormalPoll("r1", firstPoll, StateFiring, lbl("a"), onset), + clearedPoll("r1", clearAt, key), + quietPoll("r1", to), + } + outcome, badFor, viols := classifyRule(def, polls, from, to, defaultBad, PreexistingFailUnlessRecovered) + require.Equal(t, OutcomeRecovered, outcome, "the onset is at/before `from`, genuinely preexisting") + require.Empty(t, viols, "default policy passes a recovered preexisting instance") + require.Equal(t, clearAt.Sub(from), badFor, + "a preexisting episode's BadFor is clamped to window-open, not backdated past it") +} + +// A poll carrying a nonzero skew must have its ActiveAt (and GrafanaNow) +// translated to the runner domain before comparing against `from` — a raw, +// untranslated comparison would land on the wrong side of that boundary. +func TestClassifyRule_SkewTranslatesActiveAtAcrossTheWindowBoundary(t *testing.T) { + from := time.Date(2026, 1, 1, 0, 0, 0, 0, time.UTC) + to := from.Add(10 * time.Minute) + def := Definition{UID: "r1", Title: "R1"} + + // Grafana's clock reads 90s ahead of the runner's (skew = +90s). The + // poll's raw GrafanaNow/ActiveAt both sit 90s past `from` in Grafana's + // domain, but translate to exactly `from` in the runner domain — genuinely + // preexisting once translated, and wrongly "newly_bad" if the skew is + // ignored. + skew := 90 * time.Second + rawActiveAt := from.Add(skew) + poll := Poll{ + RuleUID: "r1", GrafanaNow: from.Add(skew), Found: true, Health: "ok", + LastEvaluation: from.Add(skew), SkewMS: skew.Milliseconds(), + Abnormal: []Instance{{Labels: lbl("a"), State: StateFiring, ActiveAt: rawActiveAt}}, + } + stillBad := poll + stillBad.GrafanaNow = to.Add(skew) + stillBad.LastEvaluation = to.Add(skew) + + outcome, badFor, _ := classifyRule(def, []Poll{poll, stillBad}, from, to, defaultBad, PreexistingFailUnlessRecovered) + require.Equal(t, OutcomePersistentlyBad, outcome, + "a +90s skew must translate ActiveAt back to exactly `from`") + require.Equal(t, to.Sub(from), badFor) +} + +// --- InstanceLabels must survive a timeline first created by a bare marker --- + +func TestClassifyRule_LabelsSurviveWhenTimelineStartsFromAClearedMarker(t *testing.T) { + from := time.Date(2026, 1, 1, 0, 0, 0, 0, time.UTC) + to := from.Add(10 * time.Minute) + def := Definition{UID: "r1", Title: "R1"} + key := instanceKey(lbl("a")) + newOnset := from.Add(5 * time.Minute) + + polls := []Poll{ + // The very first mention of this key is a bare Cleared marker (its + // prior bad episode, if any, started before the window) — no labels + // travel with a Cleared/Vanished event. + clearedPoll("r1", from.Add(1*time.Minute), key), + abnormalPoll("r1", newOnset, StateFiring, lbl("a"), newOnset), + quietPoll("r1", to), + } + _, _, viols := classifyRule(def, polls, from, to, defaultBad, PreexistingFailUnlessRecovered) + require.Len(t, viols, 1) + require.NotNil(t, viols[0].InstanceLabels) + require.Equal(t, "a", viols[0].InstanceLabels["instance"], + "labels must backfill even though the timeline was first created by a label-less Cleared marker") +} + +// FirstSeen/ClearedAt are pinned exactly, not just that a violation exists. +func TestClassifyRule_ViolationFieldsArePrecise(t *testing.T) { + from := time.Date(2026, 1, 1, 0, 0, 0, 0, time.UTC) + to := from.Add(10 * time.Minute) + onset := from.Add(2 * time.Minute) + clearAt := from.Add(3 * time.Minute) + def := Definition{UID: "r1", Title: "R1"} + + polls := []Poll{ + quietPoll("r1", from), + abnormalPoll("r1", onset, StateFiring, lbl("a"), onset), + clearedPoll("r1", clearAt, instanceKey(lbl("a"))), + quietPoll("r1", to), + } + _, _, viols := classifyRule(def, polls, from, to, defaultBad, PreexistingFailUnlessRecovered) + require.Len(t, viols, 1) + v := viols[0] + require.True(t, v.FirstSeen.Equal(onset)) + require.True(t, v.ClearedAt.Equal(clearAt)) + require.Equal(t, "a", v.InstanceLabels["instance"]) +} + +// The episode.end clamp: inWindowPolls admits a poll up to its own skew bound +// past windowEnd, so a genuine Cleared event on such a poll must not leave the +// episode extending beyond windowEnd. +func TestClassifyRule_ClearedEventPastWindowEndClampsToWindowEnd(t *testing.T) { + from := time.Date(2026, 1, 1, 0, 0, 0, 0, time.UTC) + to := from.Add(10 * time.Minute) + def := Definition{UID: "r1", Title: "R1"} + key := instanceKey(lbl("a")) + + bound := 30 * time.Second + clearedAt := to.Add(20 * time.Second) // past windowEnd, but within the skew bound + + polls := []Poll{ + abnormalPoll("r1", from, StateFiring, lbl("a"), from.Add(-time.Hour)), + { + RuleUID: "r1", GrafanaNow: clearedAt, Found: true, Health: "ok", + SkewBoundMS: bound.Milliseconds(), Cleared: []string{key}, + }, + } + outcome, badFor, _ := classifyRule(def, polls, from, to, defaultBad, PreexistingFailUnlessRecovered) + require.Equal(t, OutcomeRecovered, outcome) + require.Equal(t, to.Sub(from), badFor, + "the episode end must clamp to windowEnd, not extend to the late Cleared event's raw time") +} + +// TestClassifyRule_OnsetJustPastWindowEndIsNewlyBadNotClean pins the fail-closed +// reading of the upper boundary: an instance whose runner-domain onset lands +// only slightly past windowEnd (to + transitionGrace) is reachable at all only +// because inWindowPolls widens the boundary outward by the skew bound, so the +// gate cannot PROVE it belongs to the next window. It is charged as newly_bad — +// with BadFor truncated to zero — rather than silently forgiven as clean. +func TestClassifyRule_OnsetJustPastWindowEndIsNewlyBadNotClean(t *testing.T) { + from := time.Date(2026, 1, 1, 0, 0, 0, 0, time.UTC) + to := from.Add(10 * time.Minute) + grace := time.Minute + windowEnd := to.Add(grace) + def := Definition{UID: "r1", Title: "R1"} + + // A poll admitted only by its own skew bound: its GrafanaNow sits 20s past + // windowEnd, inside the 30s tolerance. It carries an instance whose onset + // is 10s past windowEnd — still "after the grace", but only by less than + // the measurement's own uncertainty. + bound := 30 * time.Second + poll := Poll{ + RuleUID: "r1", GrafanaNow: windowEnd.Add(20 * time.Second), Found: true, Health: "ok", + LastEvaluation: windowEnd.Add(20 * time.Second), SkewBoundMS: bound.Milliseconds(), + Abnormal: []Instance{{Labels: lbl("a"), State: StateFiring, ActiveAt: windowEnd.Add(10 * time.Second)}}, + } + + outcome, badFor, viols := classifyRule(def, []Poll{poll}, from, windowEnd, defaultBad, PreexistingFailUnlessRecovered) + require.Equal(t, OutcomeNewlyBad, outcome, + "an onset past windowEnd seen only via the skew bound must fail closed") + require.Zero(t, badFor, "the zero-length episode must truncate to the window end") + require.Len(t, viols, 1) +} + +// A clear after `to` gives persistently_bad. classifyRule filters +// its input to [from, windowEnd] itself (inWindowPolls), so a Cleared event +// GENUINELY past windowEnd — well beyond any skew bound, unlike the clamp +// case above — never reaches the timeline at all: the instance is still bad +// at windowEnd as far as this window is concerned. +func TestClassifyRule_ClearAfterWindowEndIsPersistentlyBad(t *testing.T) { + from := time.Date(2026, 1, 1, 0, 0, 0, 0, time.UTC) + to := from.Add(10 * time.Minute) + def := Definition{UID: "r1", Title: "R1"} + key := instanceKey(lbl("a")) + + polls := []Poll{ + abnormalPoll("r1", from, StateFiring, lbl("a"), from.Add(-time.Hour)), + clearedPoll("r1", to.Add(time.Hour), key), // far past `to`, not a boundary case + } + outcome, badFor, viols := classifyRule(def, polls, from, to, defaultBad, PreexistingFailUnlessRecovered) + require.Equal(t, OutcomePersistentlyBad, outcome, "a clear outside the window must not read as a recovery") + require.Equal(t, to.Sub(from), badFor) + require.Len(t, viols, 1) + require.Equal(t, OutcomePersistentlyBad, viols[0].Outcome) +} + +// TestClassifyRule_CloseBeforeOpenClampsToZeroNotNegative pins the +// end-before-start clamp: two polls with different measured skews can +// translate so that a closing poll's runner-domain time lands before the +// opening poll's, which — unclamped — would feed mergeDurations a negative +// span. +func TestClassifyRule_CloseBeforeOpenClampsToZeroNotNegative(t *testing.T) { + from := time.Date(2026, 1, 1, 0, 0, 0, 0, time.UTC) + to := from.Add(10 * time.Minute) + onsetPoll := from.Add(5 * time.Minute) + closePoll := from.Add(6 * time.Minute) + closeSkew := 2 * time.Minute // translates closePoll back to from+4min, before onsetPoll's from+5min + def := Definition{UID: "r1", Title: "R1"} + key := instanceKey(lbl("a")) + + polls := []Poll{ + quietPoll("r1", from), + abnormalPoll("r1", onsetPoll, StateFiring, lbl("a"), onsetPoll), // skew 0 + { + RuleUID: "r1", GrafanaNow: closePoll, Found: true, Health: "ok", + SkewMS: closeSkew.Milliseconds(), Cleared: []string{key}, + }, + quietPoll("r1", to), + } + outcome, badFor, viols := classifyRule(def, polls, from, to, defaultBad, PreexistingFailUnlessRecovered) + require.Equal(t, OutcomeNewlyBad, outcome) + require.GreaterOrEqual(t, badFor, time.Duration(0), + "a non-negative duration even though the closing poll's translated time landed before the opening poll's") + require.Zero(t, badFor, "the clamp collapses the inverted span to a zero-length episode") + require.Len(t, viols, 1) +} + +// --- mergeDurations --- + +func TestMergeDurations_OverlappingEpisodesCountOnce(t *testing.T) { + from := time.Date(2026, 1, 1, 0, 0, 0, 0, time.UTC) + eps := []episode{ + {start: from, end: from.Add(5 * time.Minute)}, + {start: from.Add(2 * time.Minute), end: from.Add(8 * time.Minute)}, // overlaps the first + {start: from.Add(20 * time.Minute), end: from.Add(21 * time.Minute)}, // disjoint + } + got := mergeDurations(eps) + want := 8*time.Minute + 1*time.Minute // [0,8) merged = 8m, plus the disjoint 1m + require.Equal(t, want, got, "two simultaneously-bad instances must not double-count their overlap") +} + +func TestMergeDurations_Empty(t *testing.T) { + require.Zero(t, mergeDurations(nil)) +} diff --git a/grafana-alertcheck/internal/gate/coverage.go b/grafana-alertcheck/internal/gate/coverage.go index 3d3704b90..603a7f7d8 100644 --- a/grafana-alertcheck/internal/gate/coverage.go +++ b/grafana-alertcheck/internal/gate/coverage.go @@ -2,33 +2,21 @@ package gate import ( "fmt" - "slices" "time" ) -// keepLastReason is the instance Reason that check 9 watches for (§10.2). +// keepLastReason is the instance Reason that check 9 watches for. const keepLastReason = "KeepLast" -// Obligations this phase leaves for later ones — carried forward the same -// way P6's own deviations list did, so a later review has something concrete -// to check against: -// -// - fromFutureTolerance (§5: 60s) has no constant and no hard-error check -// anywhere yet. Check 2 below implements only "from < StartedAt"; the -// second clause — from more than fromFutureTolerance ahead is a hard -// error — is once-per-run input validation, not a per-rule coverage -// check, and belongs to Check's construction in a later phase (P9). -// - decide (P8) must read a rule's skipped status from the definitions -// (LoggedRule.IsPaused / Definition.IsPaused), never from the polls, and -// must do so BEFORE calling proveCoverage for that rule: a rule paused -// before the window opened is never scheduled or polled (§4.3), so it -// reaches this function with zero polls and today reads as one large -// heartbeat_gap, not skipped (pinned by -// TestProveCoverage_SkippedRuleWithZeroPollsPinnedAsHeartbeatGap). +// Two things this file leaves to its callers: the "from too far ahead" bound is +// Config.validate's once-per-run input validation (check 2 owns only the +// "from < StartedAt" half), and a rule paused at the window open never reaches +// proveCoverage — decide reads `skipped` from Header.pausedAtStart first, so a +// paused rule's zero polls read as skipped, not as one large heartbeat gap. // UnobservableReason names why proveCoverage could not prove a rule's window. -// It is machine-readable — this reaches the action's JSON outputs, so it is a -// published vocabulary like Outcome (§19.0); prose belongs in Notes. +// It reaches the JSON output, so it is a published vocabulary like Outcome; +// prose belongs in Notes. type UnobservableReason string const ( @@ -41,16 +29,16 @@ const ( ReasonFutureEvaluation UnobservableReason = "future_evaluation" ReasonPausedInWindow UnobservableReason = "paused_in_window" ReasonRuleAbsent UnobservableReason = "rule_absent" - // ReasonDrainTimeout is set by check.go's drain wait (a later phase), - // never by proveCoverage: the wait is I/O and must not be added to this - // pure function — that would put HTTP inside the pure layer and destroy - // the seam §2's architecture depends on. + // ReasonDrainTimeout is set by check.go's drain wait, never by + // proveCoverage: the wait is I/O and must not be added to this pure + // function — that would put HTTP inside the pure layer and destroy the + // seam this design depends on. ReasonDrainTimeout UnobservableReason = "drain_timeout" ) // CoverageResult is proveCoverage's whole answer for one rule. No interval // list: proved-or-not plus the largest gap and where is everything a human -// reads on exit 2, and everything §20.2's table needs. +// reads on exit 2, and everything the rendered table needs. type CoverageResult struct { Proved bool LargestGap time.Duration @@ -63,36 +51,19 @@ type CoverageResult struct { BlindFor time.Duration } -// proveCoverage applies the nine coverage checks (§6, §10, §14) to one rule's -// polls and is PURE: no HTTP, no files, no clock reads — everything it needs -// arrives as an argument, which is what lets §22's tests build []Poll literals -// instead of a fixture server (§2). -// -// polls need not be pre-filtered to this rule: proveCoverage selects by -// def.UID itself, exactly as Reduce selects by UID rather than by title -// (§14.5) — a caller handing it a whole log's polls must not have to -// pre-filter to get a correct answer. -// -// Every check always runs, even once an earlier one has already set -// Unobservable: LargestGap and the notes are diagnostics an operator reads on -// exit 2 regardless of which check actually failed (§20.2). Reason names the -// FIRST check, in the order below, that failed; a later failure still adds -// its own Note. +// proveCoverage applies the nine coverage checks to one rule's polls. PURE: no +// HTTP, no files, no clock reads — everything arrives as an argument. polls need +// not be pre-filtered to this rule (selection is by def.UID). Every check runs +// even after Unobservable is set, so LargestGap and the notes are complete on +// exit 2; Reason names only the FIRST check that failed. func proveCoverage(h Header, polls []Poll, sentinel *time.Time, t ruleTimings, def Definition, from, to time.Time, grace time.Duration) CoverageResult { windowEnd := to.Add(grace) - var rulePolls []Poll - for _, p := range polls { - if p.RuleUID == def.UID { - rulePolls = append(rulePolls, p) - } - } - // Stable, not sort.Slice: two polls sharing a GrafanaNow (a coarse Date - // header, or a corrupted/replayed log) must not reorder nondeterministically - // in a function that promises to be pure. - slices.SortStableFunc(rulePolls, func(a, b Poll) int { return a.GrafanaNow.Compare(b.GrafanaNow) }) + // pollsForRule (classify.go) is the single filter+sort implementation; this + // and classifyRule must not carry two independent copies. + rulePolls := pollsForRule(polls, def.UID) var res CoverageResult fail := func(reason UnobservableReason, note string) { @@ -103,9 +74,9 @@ func proveCoverage(h Header, polls []Poll, sentinel *time.Time, t ruleTimings, d res.Notes = append(res.Notes, fmt.Sprintf("rule %q: %s", def.Title, note)) } - // Check 1 — sentinel (§4.5). Present and At >= to+grace -> coverage - // provable; absent, or short of it, is never a pass. A recorder that died - // early must look exactly like a coverage gap, because it is one. + // Check 1 — sentinel. Present and At >= to+grace -> coverage provable; + // absent, or short of it, is never a pass. A recorder that died early must + // look exactly like a coverage gap, because it is one. switch { case sentinel == nil: fail(ReasonNoSentinel, "no stopped sentinel: the recorder never reported finishing") @@ -114,36 +85,36 @@ func proveCoverage(h Header, polls []Poll, sentinel *time.Time, t ruleTimings, d sentinel.Format(time.RFC3339), windowEnd.Format(time.RFC3339))) } - // Check 2 — from bounds (§7), first sentence only: from < StartedAt makes - // coverage unprovable, no matter how healthy the polls that DO exist look. - // Both are runner-domain clock reads (the recorder's own Clock.Now()), so - // no cross-domain translation applies here. The second sentence — from - // more than fromFutureTolerance ahead is a hard error — is Check's input - // validation, once per run rather than per rule, and belongs to a later - // phase: this function has no error return, only a per-rule verdict. - if from.Before(h.StartedAt) { + // Check 2 — from bounds: from < StartedAt makes coverage unprovable, no + // matter how healthy the polls that DO exist look. Both are runner-domain + // clock reads (the recorder's own Clock.Now()), so no cross-domain + // translation applies here. The comparison is at whole-second granularity: + // `from` is supplied at second precision (--from RFC3339) while StartedAt + // carries the recorder's sub-second clock stamp, so an operator naming the + // exact second the recording opened must not be judged early for the + // sub-second sliver inside that same second. The other half of the bound — + // from too far ahead of the runner's clock — is Check's input validation, + // once per run rather than per rule. + if from.Truncate(time.Second).Before(h.StartedAt.Truncate(time.Second)) { fail(ReasonFromBeforeRecord, fmt.Sprintf( - "requested from %s is before recording started at %s", from.Format(time.RFC3339), h.StartedAt.Format(time.RFC3339))) + "requested from %s is before recording started at %s", from.Format(time.RFC3339Nano), h.StartedAt.Format(time.RFC3339Nano))) } - // Filtered once, here, and threaded through every remaining check — - // ruleHeartbeatGap included — rather than re-filtered per check: two - // independent filters over the same polls would only invite one of them - // drifting from the other's membership test. + // Filtered once and threaded through every remaining check. inWindow := inWindowPolls(rulePolls, from, windowEnd) - // Check 3 — heartbeat continuity (§6). Data at both ends with a hole in - // between is not enough (§22.4): this scans every gap inside the window, - // not just its edges. + // Check 3 — heartbeat continuity. Data at both ends with a hole in between + // is not enough: this scans every gap inside the window, not just its + // edges. res.LargestGap, res.LargestGapAt = ruleHeartbeatGap(inWindow, from, windowEnd) if res.LargestGap > t.maxGap { fail(ReasonHeartbeatGap, fmt.Sprintf( "gap of %s starting at %s exceeds maxGap %s", res.LargestGap, res.LargestGapAt.Format(time.RFC3339), t.maxGap)) } - // Check 4 — health=="error" (§10.1). A short blip is a note only (§22.1: - // one failed evaluation must not exit 2 over an otherwise clean window); - // only a run longer than healthGrace consumes coverage. + // Check 4 — health=="error". A short blip is a note only — one failed + // evaluation must not exit 2 over an otherwise clean window; only a run + // longer than healthGrace consumes coverage. if runLen, sawAny := longestHealthRun(inWindow, "error"); sawAny { res.Notes = append(res.Notes, fmt.Sprintf("rule %q: health=error observed (longest run %s)", def.Title, runLen)) if runLen > t.healthGrace { @@ -151,30 +122,19 @@ func proveCoverage(h Header, polls []Poll, sentinel *time.Time, t ruleTimings, d } } - // Check 5 — health=="nodata" (§10.1/§10.2). Never fatal here: 96% of the - // fleet runs no_data_state:OK, so treating this as fatal by default would - // block nearly every healthy deploy in an idle environment. Escalating it - // under Policy.NodataIsUnobservable is decide's job (a later phase), - // applied directly against the raw polls — this pure function has no - // Policy to consult and must not invent one. + // Check 5 — health=="nodata". Never fatal here (most of the fleet runs + // no_data_state:OK, so it would block healthy idle deploys). Escalating + // under Policy.NodataIsUnobservable is decide's job, since this pure + // function has no Policy to consult. if _, sawAny := longestHealthRun(inWindow, "nodata"); sawAny { res.Notes = append(res.Notes, fmt.Sprintf("rule %q: health=nodata observed (not fatal; see --nodata-is-unobservable)", def.Title)) } - // Check 6 — liveness (H3). Absolute only, per poll: GrafanaNow and - // LastEvaluation are both Grafana-domain reads off the SAME response, so - // this is a same-domain comparison and uses raw values — never a delta - // against a previous poll, which reports stale on ~half the polls of a - // perfectly healthy rule (polling runs at intervalSeconds/2). - // - // Skipped only for a poll whose own flags SAY there is nothing to check: - // IsPaused (a zero LastEvaluation is legal only while paused, §2.3; check - // 7 is its detector) or !Found (no rule, no evaluation; check 8 is its - // detector). Deliberately NOT skipped merely because LastEvaluation is - // zero: ReadLog does no field validation, so a corrupted or hand-edited - // log line can claim found:true, is_paused:false and still carry a zero - // LastEvaluation, and that combination must read as maximally stale - // rather than being silently waved through. + // Check 6 — liveness. Same-domain (GrafanaNow and LastEvaluation are from + // the SAME response), so raw values — never a delta against a previous + // poll, which reports stale ~half the polls of a healthy rule. Skipped only + // for IsPaused (check 7) or !Found (check 8); a zero LastEvaluation on a + // found, unpaused poll is treated as maximally stale, not waved through. var staleCount int var worstStale time.Duration var worstStaleAt time.Time @@ -186,7 +146,7 @@ func proveCoverage(h Header, polls []Poll, sentinel *time.Time, t ruleTimings, d // corrupted or hand-edited data (ReadLog does no field validation); // GrafanaNow-LastEvaluation would go negative and silently read as // fresh — fail-open. Treat it as unobservable instead. - if p.LastEvaluation.After(p.GrafanaNow) { + if p.LastEvaluation.Truncate(time.Second).After(p.GrafanaNow) { fail(ReasonFutureEvaluation, fmt.Sprintf( "lastEvaluation %s is after grafana_now %s (corrupted poll)", p.LastEvaluation.Format(time.RFC3339), p.GrafanaNow.Format(time.RFC3339))) @@ -206,11 +166,10 @@ func proveCoverage(h Header, polls []Poll, sentinel *time.Time, t ruleTimings, d staleCount, worstStale, t.evalStaleAfter, worstStaleAt.Format(time.RFC3339))) } - // Check 7 — isPaused in-window (§12.2, §14.8). The PRIMARY pause - // detector: liveness (check 6) is only the backup for what IsPaused - // cannot show (a deleted rule, a stopped scheduler, a blocked - // evaluation). This is what catches pause-then-unpause, which the drain - // wait alone passes (§14.7). + // Check 7 — isPaused in-window. The PRIMARY pause detector: liveness + // (check 6) is only the backup for what IsPaused cannot show (a deleted + // rule, a stopped scheduler, a blocked evaluation). This is what catches + // pause-then-unpause, which the drain wait alone passes. var pausedCount int var pausedAt time.Time for _, p := range inWindow { @@ -225,8 +184,8 @@ func proveCoverage(h Header, polls []Poll, sentinel *time.Time, t ruleTimings, d fail(ReasonPausedInWindow, fmt.Sprintf("observed paused on %d poll(s), first at %s", pausedCount, pausedAt.Format(time.RFC3339))) } - // Check 8 — rule absent (§14.5). Found==false is authoritative (P2 - // already retried every transport failure before a Poll record ever + // Check 8 — rule absent. Found==false is authoritative (the transport + // already retried every transient failure before a Poll record ever // exists): the rule resolved at resolve time but the state endpoint // stopped serving it. Never drop a watched rule from the verdict set // silently. @@ -244,10 +203,21 @@ func proveCoverage(h Header, polls []Poll, sentinel *time.Time, t ruleTimings, d fail(ReasonRuleAbsent, fmt.Sprintf("state endpoint returned no rule on %d poll(s), first at %s", absentCount, absentAt.Format(time.RFC3339))) } - // Check 9 — KeepLast (§10.2). A note, never fatal. It surfaces only as an - // instance Reason after P1.2a's parsing, and Reasons keys can be - // comma-joined composites, so membership (reasonsContain) is required — - // indexing "KeepLast" directly would miss "KeepLast, MissingSeries". + // Check 9 — KeepLast. Two non-fatal notes: DECLARED (the rule is configured + // with no_data_state/exec_err_state=KeepLast, read from def so it fires once), + // and OBSERVED (an instance reported KeepLast in-window; Reasons keys can be + // comma-joined, so membership via reasonsContain, never a literal index). + nds, ees := def.NoDataState, def.ExecErrState + for _, lr := range h.Rules { + if lr.UID == def.UID { + nds, ees = lr.NoDataState, lr.ExecErrState + break + } + } + if nds == keepLastReason || ees == keepLastReason { + res.Notes = append(res.Notes, fmt.Sprintf( + "rule %q: configured with no_data_state/exec_err_state=KeepLast — a stale state can continue past a real fault", def.Title)) + } for _, p := range inWindow { if reasonsContain(p.Reasons, keepLastReason) { res.Notes = append(res.Notes, fmt.Sprintf( @@ -260,22 +230,16 @@ func proveCoverage(h Header, polls []Poll, sentinel *time.Time, t ruleTimings, d return res } -// inWindowPolls filters polls to those inside [from, windowEnd] using the -// CROSS-DOMAIN membership test (§16): each poll's Grafana-domain GrafanaNow -// is translated to the runner domain by its OWN skew, and its own skew bound -// is the membership tolerance, so a poll that is genuinely inside the window -// is never excluded by ordinary clock imprecision. -// -// Everything downstream of this filter (health runs, liveness, pause, -// absence) reads the poll's raw fields: GrafanaNow paired with -// LastEvaluation on the SAME response, or one poll's GrafanaNow against the -// next's, are same-domain comparisons and need no translation (§16, "Clock -// domains" — only window membership and check 3's two boundary segments do). +// inWindowPolls filters to polls inside [from, windowEnd] via the cross-domain +// membership test: each GrafanaNow is translated to the runner domain by its +// own skew, widened by its skew bound, so clock imprecision never excludes a +// genuinely in-window poll. Everything downstream reads same-domain raw fields; +// only this filter and check 3's boundary segments cross domains. func inWindowPolls(polls []Poll, from, windowEnd time.Time) []Poll { var out []Poll for _, p := range polls { bound := p.SkewBound() - runner := p.GrafanaNow.Add(-p.Skew()) + runner := runnerTime(p, p.GrafanaNow) if runner.Before(from.Add(-bound)) || runner.After(windowEnd.Add(bound)) { continue } @@ -284,28 +248,21 @@ func inWindowPolls(polls []Poll, from, windowEnd time.Time) []Poll { return out } -// ruleHeartbeatGap finds the largest unobserved span inside [from, windowEnd] -// (§6), including the two boundary segments — which is why "data at both -// ends with a hole in the middle" still fails (§22.4): the segment between -// the polls just inside each edge is exactly what this measures. in must -// already be filtered to this window (inWindowPolls) and sorted by -// GrafanaNow — proveCoverage computes that filter once and threads it through -// every check, this one included, rather than each check re-filtering. +// ruleHeartbeatGap finds the largest unobserved span inside [from, windowEnd], +// including the two boundary segments — which is why data at both ends with a +// hole in the middle still fails. in must be filtered (inWindowPolls) and +// sorted by GrafanaNow. // -// The two boundary segments compare a Grafana-domain poll time against the -// runner-domain from/windowEnd, so each is translated by its own poll's skew -// AND widened by that same poll's skew bound (§16: "with that poll's bound as -// the tolerance") — on the side that makes the segment larger, never smaller, -// so an uncertain boundary reads as at least as big a gap as it might really -// be. Understating it by up to the bound would be fail-open. The spacing -// BETWEEN consecutive polls compares two Grafana-domain reads to each other — -// same domain — and uses the raw GrafanaNow difference, no bound needed. +// Boundary segments are cross-domain, so each translated poll time is widened +// by its skew bound on the side that makes the gap LARGER (never smaller — an +// uncertain boundary must read as at least as big a gap as it might be). +// Consecutive-poll spacing is same-domain and uses the raw GrafanaNow diff. func ruleHeartbeatGap(in []Poll, from, windowEnd time.Time) (largestGap time.Duration, largestGapAt time.Time) { if len(in) == 0 { return windowEnd.Sub(from), from } - runnerOf := func(p Poll) time.Time { return p.GrafanaNow.Add(-p.Skew()) } + runnerOf := func(p Poll) time.Time { return runnerTime(p, p.GrafanaNow) } first := in[0] if gap := runnerOf(first).Sub(from) + first.SkewBound(); gap > largestGap { @@ -323,16 +280,10 @@ func ruleHeartbeatGap(in []Poll, from, windowEnd time.Time) (largestGap time.Dur return largestGap, largestGapAt } -// longestHealthRun returns the longest contiguous wall-clock span (§10.1) -// during which polls — already sorted by GrafanaNow, same-domain spacing -// (§16) — read the given rule-level Health, and whether any poll matched it -// at all. -// -// It detects the span as it accumulates rather than waiting for the run to -// end, so an open-ended run that is still failing at the last poll in the -// window is measured correctly without needing data past the window: waiting -// for the run to "end" would have to assume the best case about what happens -// next, which is exactly what this gate must not do (§1). +// longestHealthRun returns the longest contiguous span of polls reading the +// given rule-level Health, measured incrementally so a run still failing at the +// last in-window poll is measured correctly without assuming anything past the +// window. func longestHealthRun(polls []Poll, health string) (longest time.Duration, sawAny bool) { var runStart time.Time for _, p := range polls { diff --git a/grafana-alertcheck/internal/gate/coverage_test.go b/grafana-alertcheck/internal/gate/coverage_test.go index 6419a8b2e..3507ee4fc 100644 --- a/grafana-alertcheck/internal/gate/coverage_test.go +++ b/grafana-alertcheck/internal/gate/coverage_test.go @@ -4,6 +4,8 @@ import ( "strings" "testing" "time" + + "github.com/stretchr/testify/require" ) func TestProveCoverage_CleanWindowIsProved(t *testing.T) { @@ -19,9 +21,9 @@ func TestProveCoverage_CleanWindowIsProved(t *testing.T) { sentinel := to res := proveCoverage(Header{StartedAt: from.Add(-time.Hour)}, polls, &sentinel, rt, def, from, to, 0) - if !res.Proved || res.Unobservable || res.Reason != "" { - t.Fatalf("res = %+v, want a clean proved window", res) - } + require.True(t, res.Proved) + require.False(t, res.Unobservable) + require.Empty(t, res.Reason) } func TestProveCoverage_FiltersPollsByUID(t *testing.T) { @@ -40,12 +42,10 @@ func TestProveCoverage_FiltersPollsByUID(t *testing.T) { sentinel := to res := proveCoverage(Header{StartedAt: from.Add(-time.Hour)}, polls, &sentinel, rt, def, from, to, 0) - if !res.Proved { - t.Fatalf("res = %+v, want proved: a different rule's broken polls must not affect this rule's verdict", res) - } + require.True(t, res.Proved, "a different rule's broken polls must not affect this rule's verdict") } -// --- Check 1: sentinel (§4.5) --- +// --- Check 1: sentinel --- func TestProveCoverage_NoSentinelIsUnobservable(t *testing.T) { from := time.Date(2026, 1, 1, 0, 0, 0, 0, time.UTC) @@ -54,9 +54,8 @@ func TestProveCoverage_NoSentinelIsUnobservable(t *testing.T) { def := Definition{UID: "r1", Title: "R1"} res := proveCoverage(Header{StartedAt: from.Add(-time.Hour)}, nil, nil, rt, def, from, to, 0) - if res.Proved || res.Reason != ReasonNoSentinel { - t.Fatalf("res = %+v, want unobservable/no_sentinel: an absent sentinel must never be a pass", res) - } + require.False(t, res.Proved) + require.Equal(t, ReasonNoSentinel, res.Reason, "an absent sentinel must never be a pass") } func TestProveCoverage_SentinelBeforeGraceIsUnobservable(t *testing.T) { @@ -68,9 +67,19 @@ func TestProveCoverage_SentinelBeforeGraceIsUnobservable(t *testing.T) { sentinel := to.Add(grace).Add(-time.Second) // one second short of to+grace res := proveCoverage(Header{StartedAt: from.Add(-time.Hour)}, nil, &sentinel, rt, def, from, to, grace) - if res.Reason != ReasonSentinelEarly { - t.Fatalf("Reason = %q, want sentinel_early", res.Reason) - } + require.Equal(t, ReasonSentinelEarly, res.Reason) + require.True(t, res.Unobservable) + require.False(t, res.Proved, "a reason string with no consequence is not a coverage failure") + + // The consequence: decide() must turn this into exit 2, never a pass. + defs := []Definition{def} + drt := map[string]ruleTimings{def.UID: rt} + gt := globalTimings{transitionGrace: grace} + pol := Policy{From: from, To: to} + dres, err := decide(Header{StartedAt: from.Add(-time.Hour)}, nil, &sentinel, defs, drt, gt, pol) + require.Error(t, err, "a sentinel short of to+grace must fail the run") + require.Len(t, dres.Verdicts, 1) + require.Equal(t, OutcomeUnobservable, dres.Verdicts[0].Outcome) } func TestProveCoverage_SentinelExactlyAtGraceIsFine(t *testing.T) { @@ -88,12 +97,10 @@ func TestProveCoverage_SentinelExactlyAtGraceIsFine(t *testing.T) { sentinel := windowEnd res := proveCoverage(Header{StartedAt: from.Add(-time.Hour)}, polls, &sentinel, rt, def, from, to, grace) - if !res.Proved { - t.Fatalf("Proved = false, want true: sentinel exactly at to+grace must satisfy check 1: %+v", res) - } + require.True(t, res.Proved, "sentinel exactly at to+grace must satisfy check 1") } -// --- Check 2: from bounds (§7) --- +// --- Check 2: from bounds --- func TestProveCoverage_FromBeforeRecordIsUnobservable(t *testing.T) { started := time.Date(2026, 1, 1, 1, 0, 0, 0, time.UTC) @@ -104,15 +111,82 @@ func TestProveCoverage_FromBeforeRecordIsUnobservable(t *testing.T) { sentinel := to res := proveCoverage(Header{StartedAt: started}, nil, &sentinel, rt, def, from, to, 0) - if res.Reason != ReasonFromBeforeRecord { - t.Fatalf("Reason = %q, want from_before_record", res.Reason) + require.Equal(t, ReasonFromBeforeRecord, res.Reason) + require.True(t, res.Unobservable) + require.False(t, res.Proved) + + // The consequence: decide() must turn this into exit 2, never a pass. + defs := []Definition{def} + drt := map[string]ruleTimings{def.UID: rt} + gt := globalTimings{} + pol := Policy{From: from, To: to} + dres, err := decide(Header{StartedAt: started}, nil, &sentinel, defs, drt, gt, pol) + require.Error(t, err, "`from` before the recording started must fail the run") + require.Len(t, dres.Verdicts, 1) + require.Equal(t, OutcomeUnobservable, dres.Verdicts[0].Outcome) +} + +// The from-bounds check compares at whole-second granularity: a whole-second +// `from` may precede the recorder's sub-second StartedAt INSIDE the same second +// without being judged early. That one sliver is the --from truncation, not a +// blind interval, so the window is still proved. +func TestProveCoverage_FromSameSecondAsStartedAtIsProved(t *testing.T) { + from := time.Date(2026, 1, 1, 0, 0, 0, 0, time.UTC) + to := from.Add(10 * time.Minute) + started := from.Add(500 * time.Millisecond) + rt := newRuleTimings(30*time.Second, 60) + def := Definition{UID: "r1", Title: "R1"} + + var polls []Poll + for ts := from; !ts.After(to); ts = ts.Add(30 * time.Second) { + polls = append(polls, Poll{RuleUID: "r1", GrafanaNow: ts, Found: true, Health: "ok", State: "inactive", LastEvaluation: ts}) } + sentinel := to + + res := proveCoverage(Header{StartedAt: started}, polls, &sentinel, rt, def, from, to, 0) + require.True(t, res.Proved) + require.False(t, res.Unobservable) + require.Empty(t, res.Reason) } -// --- Check 3: heartbeat continuity (§6) --- +// Exactly one whole second later is a different second: even at the boundary, +// the whole-second comparison reads it as before, however healthy the polls. +func TestProveCoverage_FromExactlyOneSecondBeforeStartedAtIsUnobservable(t *testing.T) { + from := time.Date(2026, 1, 1, 0, 0, 0, 0, time.UTC) + to := from.Add(10 * time.Minute) + started := from.Add(time.Second) + rt := newRuleTimings(30*time.Second, 60) + def := Definition{UID: "r1", Title: "R1"} + + sentinel := to + res := proveCoverage(Header{StartedAt: started}, nil, &sentinel, rt, def, from, to, 0) + require.Equal(t, ReasonFromBeforeRecord, res.Reason) + require.True(t, res.Unobservable) + require.False(t, res.Proved) +} + +// A sub-second sliver that straddles the second boundary is still "before": +// 900ms into one second vs 100ms into the next are distinct seconds, so the +// 200ms gap is a from_before_record, not rounding noise. +func TestProveCoverage_FromSubSecondEarlierAcrossSecondBoundaryIsUnobservable(t *testing.T) { + base := time.Date(2026, 1, 1, 0, 0, 0, 0, time.UTC) + from := base.Add(900 * time.Millisecond) + started := base.Add(time.Second + 100*time.Millisecond) + to := from.Add(10 * time.Minute) + rt := newRuleTimings(30*time.Second, 60) + def := Definition{UID: "r1", Title: "R1"} -// TestProveCoverage_HeartbeatGapBetweenBoundariesIsUnobservable is §22.4's -// core regression: data at both ends with a hole between is not enough. + sentinel := to + res := proveCoverage(Header{StartedAt: started}, nil, &sentinel, rt, def, from, to, 0) + require.Equal(t, ReasonFromBeforeRecord, res.Reason) + require.True(t, res.Unobservable) + require.False(t, res.Proved) +} + +// --- Check 3: heartbeat continuity --- + +// The core heartbeat regression: data at both ends with a hole between is not +// enough. func TestProveCoverage_HeartbeatGapBetweenBoundariesIsUnobservable(t *testing.T) { from := time.Date(2026, 1, 1, 0, 0, 0, 0, time.UTC) to := from.Add(10 * time.Minute) @@ -126,21 +200,14 @@ func TestProveCoverage_HeartbeatGapBetweenBoundariesIsUnobservable(t *testing.T) sentinel := to res := proveCoverage(Header{StartedAt: from.Add(-time.Hour)}, polls, &sentinel, rt, def, from, to, 0) - if res.Reason != ReasonHeartbeatGap { - t.Fatalf("Reason = %q, want heartbeat_gap: healthy edges with a hole in the middle must still fail (§22.4)", res.Reason) - } + require.Equal(t, ReasonHeartbeatGap, res.Reason, "healthy edges with a hole in the middle must still fail") // The gap is the SPACING between the two polls (598s), not either // boundary segment (1s each) — pin the actual values, not just the verdict. - if res.LargestGap != 598*time.Second { - t.Fatalf("LargestGap = %s, want 598s (the spacing between the two polls, not a boundary segment)", res.LargestGap) - } - wantAt := from.Add(time.Second) - if !res.LargestGapAt.Equal(wantAt) { - t.Fatalf("LargestGapAt = %s, want %s (where the gap starts, at the first poll)", res.LargestGapAt, wantAt) - } + require.Equal(t, 598*time.Second, res.LargestGap) + require.True(t, res.LargestGapAt.Equal(from.Add(time.Second))) } -// --- Check 4/5: health (§10.1/§10.2) --- +// --- Check 4/5: health --- func TestProveCoverage_HealthErrorShortBlipPassesWithNote(t *testing.T) { from := time.Date(2026, 1, 1, 0, 0, 0, 0, time.UTC) @@ -160,12 +227,8 @@ func TestProveCoverage_HealthErrorShortBlipPassesWithNote(t *testing.T) { sentinel := to res := proveCoverage(Header{StartedAt: from.Add(-time.Hour)}, polls, &sentinel, rt, def, from, to, 0) - if !res.Proved { - t.Fatalf("Proved = false, want true: one failed evaluation must not fail an otherwise clean window (§22.1): %+v", res) - } - if !anyContains(res.Notes, "health=error") { - t.Fatalf("Notes = %v, want a health=error note even though it did not fail the window", res.Notes) - } + require.True(t, res.Proved, "one failed evaluation must not fail an otherwise clean window") + require.True(t, anyContains(res.Notes, "health=error"), "want a health=error note even though it did not fail the window") } func TestProveCoverage_HealthErrorSustainedIsUnobservable(t *testing.T) { @@ -186,9 +249,7 @@ func TestProveCoverage_HealthErrorSustainedIsUnobservable(t *testing.T) { sentinel := to res := proveCoverage(Header{StartedAt: from.Add(-time.Hour)}, polls, &sentinel, rt, def, from, to, 0) - if res.Reason != ReasonHealthError { - t.Fatalf("Reason = %q, want health_error for a run that outlasts healthGrace", res.Reason) - } + require.Equal(t, ReasonHealthError, res.Reason, "a run that outlasts healthGrace") } func TestProveCoverage_HealthNodataNeverFatalHere(t *testing.T) { @@ -204,22 +265,16 @@ func TestProveCoverage_HealthNodataNeverFatalHere(t *testing.T) { sentinel := to res := proveCoverage(Header{StartedAt: from.Add(-time.Hour)}, polls, &sentinel, rt, def, from, to, 0) - if !res.Proved { - t.Fatalf("Proved = false, want true: health=nodata for the WHOLE window must still not be fatal by itself "+ - "(escalating it is Policy.NodataIsUnobservable's job, applied by decide in a later phase): %+v", res) - } - if !anyContains(res.Notes, "health=nodata") { - t.Fatalf("Notes = %v, want a health=nodata note", res.Notes) - } + require.True(t, res.Proved, "health=nodata for the WHOLE window must still not be fatal by itself") + require.True(t, anyContains(res.Notes, "health=nodata")) } -// --- Check 6: liveness / H3 --- +// --- Check 6: liveness --- -// TestProveCoverage_LivenessAbsoluteNeverFalseStale is §22.7's disproportionate -// test: a healthy rule polled at intervalSeconds/2, across the full window, -// must show zero staleness violations. lastEvaluation only advances once per -// full evaluation interval here — the realistic shape a delta check -// misreads as stale on roughly half of all polls (H3). +// A healthy rule polled at intervalSeconds/2, across the full window, must +// show zero staleness violations. lastEvaluation only advances once per full +// evaluation interval here — the realistic shape a delta check misreads as +// stale on roughly half of all polls. func TestProveCoverage_LivenessAbsoluteNeverFalseStale(t *testing.T) { from := time.Date(2026, 1, 1, 0, 0, 0, 0, time.UTC) pollEvery := 30 * time.Second @@ -239,13 +294,9 @@ func TestProveCoverage_LivenessAbsoluteNeverFalseStale(t *testing.T) { sentinel := windowEnd res := proveCoverage(Header{StartedAt: from.Add(-time.Hour)}, polls, &sentinel, rt, def, from, windowEnd, 0) - if res.Reason == ReasonStaleEvaluation || res.BlindFor != 0 { - t.Fatalf("proveCoverage flagged staleness on a healthy rule polled at intervalSeconds/2 — H3 must be absolute, "+ - "never a delta against a previous poll: %+v", res) - } - if !res.Proved { - t.Fatalf("Proved = false, want true: %+v (notes: %v)", res, res.Notes) - } + require.NotEqual(t, ReasonStaleEvaluation, res.Reason, "liveness must be absolute, never a delta against a previous poll") + require.Zero(t, res.BlindFor) + require.True(t, res.Proved) } func TestProveCoverage_StaleEvaluationIsUnobservable(t *testing.T) { @@ -266,12 +317,8 @@ func TestProveCoverage_StaleEvaluationIsUnobservable(t *testing.T) { sentinel := to res := proveCoverage(Header{StartedAt: from.Add(-time.Hour)}, polls, &sentinel, rt, def, from, to, 0) - if res.Reason != ReasonStaleEvaluation { - t.Fatalf("Reason = %q, want stale_evaluation", res.Reason) - } - if res.BlindFor != 3*time.Minute { - t.Fatalf("BlindFor = %s, want 3m", res.BlindFor) - } + require.Equal(t, ReasonStaleEvaluation, res.Reason) + require.Equal(t, 3*time.Minute, res.BlindFor) } func TestProveCoverage_ZeroLastEvaluationNeverFalseStale(t *testing.T) { @@ -280,21 +327,18 @@ func TestProveCoverage_ZeroLastEvaluationNeverFalseStale(t *testing.T) { rt := newRuleTimings(30*time.Second, 60) def := Definition{UID: "r1", Title: "R1"} - // A paused rule legitimately reports the zero time (§2.3); check 6 must - // not read that as an enormous staleness violation. Check 7 is its - // detector. + // A paused rule legitimately reports the zero time; check 6 must not read + // that as an enormous staleness violation. Check 7 is its detector. polls := []Poll{ {RuleUID: "r1", GrafanaNow: from.Add(time.Minute), Found: true, IsPaused: true}, } sentinel := to res := proveCoverage(Header{StartedAt: from.Add(-time.Hour)}, polls, &sentinel, rt, def, from, to, 0) - if res.Reason == ReasonStaleEvaluation { - t.Fatalf("a zero lastEvaluation on a paused poll must not trigger check 6: %+v", res) - } + require.NotEqual(t, ReasonStaleEvaluation, res.Reason, "a zero lastEvaluation on a paused poll must not trigger check 6") } -// --- Check 7: isPaused in-window (§12.2, §14.8) --- +// --- Check 7: isPaused in-window --- func TestProveCoverage_PausedInWindowIsUnobservable(t *testing.T) { from := time.Date(2026, 1, 1, 0, 0, 0, 0, time.UTC) @@ -309,15 +353,13 @@ func TestProveCoverage_PausedInWindowIsUnobservable(t *testing.T) { for i := range polls { if polls[i].GrafanaNow.Equal(pausedAt) { polls[i].IsPaused = true - polls[i].LastEvaluation = time.Time{} // legal only while paused, §2.3 + polls[i].LastEvaluation = time.Time{} // legal only while paused } } sentinel := to res := proveCoverage(Header{StartedAt: from.Add(-time.Hour)}, polls, &sentinel, rt, def, from, to, 0) - if res.Reason != ReasonPausedInWindow { - t.Fatalf("Reason = %q, want paused_in_window", res.Reason) - } + require.Equal(t, ReasonPausedInWindow, res.Reason) } // TestProveCoverage_PausedAfterWindowIsFine pins check 7's respect for the @@ -339,15 +381,11 @@ func TestProveCoverage_PausedAfterWindowIsFine(t *testing.T) { sentinel := to res := proveCoverage(Header{StartedAt: from.Add(-time.Hour)}, polls, &sentinel, rt, def, from, to, 0) - if res.Reason == ReasonPausedInWindow { - t.Fatalf("a paused poll after windowEnd tripped check 7: %+v", res.Notes) - } - if !res.Proved { - t.Fatalf("Proved = false, want a clean window: %+v", res.Notes) - } + require.NotEqual(t, ReasonPausedInWindow, res.Reason, "a paused poll after windowEnd tripped check 7") + require.True(t, res.Proved) } -// --- Check 8: rule absent (§14.5) --- +// --- Check 8: rule absent --- func TestProveCoverage_RuleAbsentIsUnobservable(t *testing.T) { from := time.Date(2026, 1, 1, 0, 0, 0, 0, time.UTC) @@ -369,9 +407,7 @@ func TestProveCoverage_RuleAbsentIsUnobservable(t *testing.T) { sentinel := to res := proveCoverage(Header{StartedAt: from.Add(-time.Hour)}, polls, &sentinel, rt, def, from, to, 0) - if res.Reason != ReasonRuleAbsent { - t.Fatalf("Reason = %q, want rule_absent", res.Reason) - } + require.Equal(t, ReasonRuleAbsent, res.Reason) } // denseHealthyPolls builds a clean poll sequence at a fixed cadence, with @@ -386,9 +422,9 @@ func denseHealthyPolls(uid string, from, to time.Time, every time.Duration) []Po return out } -// --- Check 9: KeepLast (§10.2) --- +// --- Check 9: KeepLast --- -func TestProveCoverage_KeepLastIsNoteOnly(t *testing.T) { +func TestProveCoverage_KeepLastObservedIsNoteOnly(t *testing.T) { from := time.Date(2026, 1, 1, 0, 0, 0, 0, time.UTC) to := from.Add(10 * time.Minute) rt := newRuleTimings(30*time.Second, 60) @@ -399,27 +435,53 @@ func TestProveCoverage_KeepLastIsNoteOnly(t *testing.T) { polls = append(polls, Poll{ RuleUID: "r1", GrafanaNow: ts, Found: true, Health: "ok", LastEvaluation: ts, // A comma-joined composite — reasonsContain must match by - // membership, never by an exact key, per P5's markers. + // membership, never by an exact key. Reasons: map[string]int{"KeepLast, MissingSeries": 1}, }) } sentinel := to res := proveCoverage(Header{StartedAt: from.Add(-time.Hour)}, polls, &sentinel, rt, def, from, to, 0) - if !res.Proved { - t.Fatalf("Proved = false, want true: KeepLast is a note, never fatal: %+v", res) - } - if !anyContains(res.Notes, "KeepLast") { - t.Fatalf("Notes = %v, want a KeepLast note (comma-joined membership, not a literal-key match)", res.Notes) + require.True(t, res.Proved, "KeepLast is a note, never fatal") + require.True(t, anyContains(res.Notes, "KeepLast"), "comma-joined membership, not a literal-key match") +} + +// KeepLast in the CONFIGURATION gives a note — a different claim from the +// observed-reason test above. A rule DECLARED with +// no_data_state or exec_err_state = KeepLast is a standing blind spot +// whether or not any poll ever actually reports the reason, so the note +// must fire off the definition alone, over an otherwise perfectly healthy +// window with zero KeepLast reasons anywhere in it. +func TestProveCoverage_KeepLastConfiguredIsNoteOnly(t *testing.T) { + from := time.Date(2026, 1, 1, 0, 0, 0, 0, time.UTC) + to := from.Add(10 * time.Minute) + rt := newRuleTimings(30*time.Second, 60) + + tests := []struct { + name string + def Definition + }{ + {"no_data_state", Definition{UID: "r1", Title: "R1", NoDataState: "KeepLast"}}, + {"exec_err_state", Definition{UID: "r1", Title: "R1", ExecErrState: "KeepLast"}}, + } + for _, tc := range tests { + t.Run(tc.name, func(t *testing.T) { + polls := denseHealthyPolls("r1", from, to, 30*time.Second) + sentinel := to + + res := proveCoverage(Header{StartedAt: from.Add(-time.Hour)}, polls, &sentinel, rt, tc.def, from, to, 0) + require.True(t, res.Proved, "a declared KeepLast is a note, never fatal") + require.True(t, anyContains(res.Notes, "KeepLast"), + "from the definition alone, with zero KeepLast reasons observed") + }) } } -// --- Clock domains (§16) --- +// --- Clock domains --- -// TestProveCoverage_SkewTranslationAtWindowBoundary pins §16's "Clock -// domains" rule: a constant clock skew on every poll must not itself read as -// a coverage gap or a from-before-record violation, because every -// cross-domain comparison translates by that poll's own skew first. +// A constant clock skew on every poll must not itself read as a coverage gap +// or a from-before-record violation, because every cross-domain comparison +// translates by that poll's own skew first. func TestProveCoverage_SkewTranslationAtWindowBoundary(t *testing.T) { from := time.Date(2026, 1, 1, 0, 0, 0, 0, time.UTC) to := from.Add(10 * time.Minute) @@ -440,16 +502,13 @@ func TestProveCoverage_SkewTranslationAtWindowBoundary(t *testing.T) { sentinel := to res := proveCoverage(Header{StartedAt: from.Add(-time.Hour)}, polls, &sentinel, rt, def, from, to, 0) - if !res.Proved { - t.Fatalf("res = %+v, want proved: a constant clock skew must not itself read as a coverage gap (§16)", res) - } + require.True(t, res.Proved, "a constant clock skew must not itself read as a coverage gap") } -// --- Override round-trip (P5's "two authorities") --- +// --- Override round-trip: one authority for the cadence --- -// TestProveCoverage_OverrideRoundTrip is P7's other disproportionate done-gate -// test: it exercises DeriveTimingsFromLog and proveCoverage together, exactly -// as check will, to prove maxGap tracks the RECORDED cadence, never a +// This exercises DeriveTimingsFromLog and proveCoverage together, exactly as +// check does, to prove maxGap tracks the RECORDED cadence and never a // re-derivation from the rule's own evaluation interval. func TestProveCoverage_OverrideRoundTrip(t *testing.T) { from := time.Date(2026, 1, 1, 0, 0, 0, 0, time.UTC) @@ -462,9 +521,7 @@ func TestProveCoverage_OverrideRoundTrip(t *testing.T) { } defs := []Definition{{UID: "r1", Title: "R1", IntervalSeconds: 60}} rt, _, err := DeriveTimingsFromLog(h, defs) - if err != nil { - t.Fatalf("DeriveTimingsFromLog: %v", err) - } + require.NoError(t, err) var polls []Poll for ts := from; !ts.After(windowEnd); ts = ts.Add(120 * time.Second) { @@ -473,9 +530,8 @@ func TestProveCoverage_OverrideRoundTrip(t *testing.T) { sentinel := windowEnd res := proveCoverage(h, polls, &sentinel, rt["r1"], defs[0], from, windowEnd, 0) - if !res.Proved { - t.Fatalf("Proved = false, want true (maxGap must come from the recorded 120s cadence, not the 30s default): %+v", res) - } + require.True(t, res.Proved, + "maxGap must come from the recorded 120s cadence, not the 30s default") }) t.Run("faster override still catches a real recorder gap", func(t *testing.T) { @@ -486,9 +542,7 @@ func TestProveCoverage_OverrideRoundTrip(t *testing.T) { } defs := []Definition{{UID: "r1", Title: "R1", IntervalSeconds: 300}} rt, _, err := DeriveTimingsFromLog(h, defs) - if err != nil { - t.Fatalf("DeriveTimingsFromLog: %v", err) - } + require.NoError(t, err) var polls []Poll ts := from @@ -509,10 +563,8 @@ func TestProveCoverage_OverrideRoundTrip(t *testing.T) { sentinel := windowEnd res := proveCoverage(h, polls, &sentinel, rt["r1"], defs[0], from, windowEnd, 0) - if res.Reason != ReasonHeartbeatGap { - t.Fatalf("Reason = %q, want heartbeat_gap: if maxGap had been re-derived from the 300s definition instead of "+ - "the recorded 5s cadence, this 250s gap would pass silently — the fail-open direction P5 warns about", res.Reason) - } + require.Equal(t, ReasonHeartbeatGap, res.Reason, + "if maxGap had been re-derived from the 300s definition, this 250s gap would pass silently") }) } @@ -532,7 +584,7 @@ func anyContains(notes []string, substr string) bool { // found:true, is_paused:false and still carry a zero LastEvaluation (a // corrupted write, a hand-edited fixture, a future log format bug). That // combination must read as maximally stale, not be waved through the way a -// legitimately paused poll's zero time is (§2.3) — the skip must key off +// legitimately paused poll's zero time is — the skip must key off // IsPaused/Found, never off LastEvaluation being zero. func TestProveCoverage_ZeroLastEvaluationWithoutPauseIsStale(t *testing.T) { from := time.Date(2026, 1, 1, 0, 0, 0, 0, time.UTC) @@ -550,10 +602,8 @@ func TestProveCoverage_ZeroLastEvaluationWithoutPauseIsStale(t *testing.T) { sentinel := to res := proveCoverage(Header{StartedAt: from.Add(-time.Hour)}, polls, &sentinel, rt, def, from, to, 0) - if res.Reason != ReasonStaleEvaluation { - t.Fatalf("Reason = %q, want stale_evaluation: a zero lastEvaluation on a found, non-paused poll must fail "+ - "closed, not be silently skipped as if it were a legitimately paused observation", res.Reason) - } + require.Equal(t, ReasonStaleEvaluation, res.Reason, + "a zero lastEvaluation on a found, non-paused poll must fail closed") } // A lastEvaluation in the future of grafana_now (corrupted log) must fail closed. @@ -573,18 +623,15 @@ func TestProveCoverage_FutureLastEvaluationIsUnobservable(t *testing.T) { sentinel := to res := proveCoverage(Header{StartedAt: from.Add(-time.Hour)}, polls, &sentinel, rt, def, from, to, 0) - if res.Reason != ReasonFutureEvaluation { - t.Fatalf("Reason = %q, want future_evaluation: a lastEvaluation in the future of grafana_now must fail "+ - "closed rather than read its negative staleness as fresh", res.Reason) - } + require.Equal(t, ReasonFutureEvaluation, res.Reason, + "a lastEvaluation in the future of grafana_now must fail closed rather than read its negative staleness as fresh") } // --- Check 3, tightened: the boundary segments must widen by the skew bound --- -// TestProveCoverage_BoundaryGapWidensBySkewBound pins §16's "with that -// poll's bound as the tolerance" for the two boundary segments specifically: -// a boundary gap that lands EXACTLY at maxGap must still fail once the -// poll's own skew bound is added, because the translation is only a best +// The two boundary segments take their own poll's bound as the tolerance: a +// boundary gap that lands EXACTLY at maxGap must still fail once the poll's +// own skew bound is added, because the translation is only a best // estimate and understating the gap by up to the bound would be fail-open. func TestProveCoverage_BoundaryGapWidensBySkewBound(t *testing.T) { from := time.Date(2026, 1, 1, 0, 0, 0, 0, time.UTC) @@ -602,19 +649,16 @@ func TestProveCoverage_BoundaryGapWidensBySkewBound(t *testing.T) { sentinel := to res := proveCoverage(Header{StartedAt: from.Add(-time.Hour)}, polls, &sentinel, rt, def, from, to, 0) - if res.Reason != ReasonHeartbeatGap { - t.Fatalf("Reason = %q, want heartbeat_gap: the leading boundary segment sits at EXACTLY maxGap (60s) before "+ - "widening; the poll's own %s skew bound must push it past the threshold (§16), not just the skew translation", res.Reason, bound) - } + require.Equal(t, ReasonHeartbeatGap, res.Reason, + "the poll's own %s skew bound must push it past the threshold", bound) } // --- Multi-failure contract --- -// TestProveCoverage_MultipleFailuresReasonIsFirstButAllNoted exercises two -// checks failing in the same rule: check 7 (paused in-window) precedes check -// 8 (rule absent) in the §5 order, so Reason must name the pause even though -// the rule also goes absent later — and the later failure must still add its -// own Note rather than being swallowed once Reason is set. +// Two checks failing in the same rule: check 7 (paused in-window) runs before +// check 8 (rule absent), so Reason must name the pause even though the rule +// also goes absent later — and the later failure must still add its own Note +// rather than being swallowed once Reason is set. func TestProveCoverage_MultipleFailuresReasonIsFirstButAllNoted(t *testing.T) { from := time.Date(2026, 1, 1, 0, 0, 0, 0, time.UTC) to := from.Add(10 * time.Minute) @@ -638,30 +682,22 @@ func TestProveCoverage_MultipleFailuresReasonIsFirstButAllNoted(t *testing.T) { sentinel := to res := proveCoverage(Header{StartedAt: from.Add(-time.Hour)}, polls, &sentinel, rt, def, from, to, 0) - if res.Reason != ReasonPausedInWindow { - t.Fatalf("Reason = %q, want paused_in_window (the FIRST check to fail, in §5's order)", res.Reason) - } - if !anyContains(res.Notes, "paused") { - t.Fatalf("Notes = %v, want a note about the pause", res.Notes) - } - if !anyContains(res.Notes, "no rule") { - t.Fatalf("Notes = %v, want a note about the absence too — a later failure must still be recorded, "+ - "not swallowed once Reason is already set", res.Notes) - } + require.Equal(t, ReasonPausedInWindow, res.Reason, "the FIRST check to fail names the reason") + require.True(t, anyContains(res.Notes, "paused")) + require.True(t, anyContains(res.Notes, "no rule"), + "a later failure must still be recorded, not swallowed once Reason is already set") } -// --- Skipped rules (P6/P8 obligation) --- +// --- Skipped rules --- -// TestProveCoverage_SkippedRuleWithZeroPollsPinnedAsHeartbeatGap pins a known -// gap in this function's contract, not a bug in it: a rule paused BEFORE the -// window opened is never scheduled or polled (watch.go, §4.3), so it reaches -// proveCoverage with zero polls at all. proveCoverage has no notion of +// A known limit of this function's contract, not a bug in it: a rule paused +// BEFORE the window opened is never scheduled or polled (watch.go), so it +// reaches proveCoverage with zero polls at all. proveCoverage has no notion of // "skipped" — that classification belongs to the definitions -// (LoggedRule.IsPaused / Definition.IsPaused), never to the polls — so today -// it reports the whole window as one big heartbeat_gap instead. decide (P8) -// MUST read skipped status from the definitions and either skip calling this -// function for that rule entirely, or override this result — this test pins -// today's behavior so that review has something concrete to check against. +// (LoggedRule.IsPaused / Definition.IsPaused), never to the polls — so it +// reports the whole window as one big heartbeat_gap instead. decide is what +// reads skipped status from the header and never calls this function for such +// a rule; this pins the behavior it relies on not reaching. func TestProveCoverage_SkippedRuleWithZeroPollsPinnedAsHeartbeatGap(t *testing.T) { from := time.Date(2026, 1, 1, 0, 0, 0, 0, time.UTC) to := from.Add(10 * time.Minute) @@ -670,9 +706,6 @@ func TestProveCoverage_SkippedRuleWithZeroPollsPinnedAsHeartbeatGap(t *testing.T sentinel := to res := proveCoverage(Header{StartedAt: from.Add(-time.Hour)}, nil, &sentinel, rt, def, from, to, 0) - if res.Reason != ReasonHeartbeatGap { - t.Fatalf("Reason = %q, want heartbeat_gap (pinned, not the desired end state): proveCoverage has no "+ - "'skipped' concept, so decide (P8) must handle a skipped rule's classification itself, before or "+ - "instead of calling this function", res.Reason) - } + require.Equal(t, ReasonHeartbeatGap, res.Reason, + "proveCoverage has no 'skipped' concept, so decide must handle a skipped rule's classification itself") } diff --git a/grafana-alertcheck/internal/gate/duration.go b/grafana-alertcheck/internal/gate/duration.go index 011c27f14..3690a2549 100644 --- a/grafana-alertcheck/internal/gate/duration.go +++ b/grafana-alertcheck/internal/gate/duration.go @@ -29,7 +29,7 @@ var promDurationUnits = []promDurationUnit{ } // ParsePromDuration parses a Grafana/Prometheus-style duration ("1h30m", "1d", "1w"). -// Unlike time.ParseDuration, it accepts "d" and "w" (§11.8). "" and "0" are 0. +// Unlike time.ParseDuration, it accepts "d" and "w". "" and "0" are 0. func ParsePromDuration(s string) (time.Duration, error) { if s == "" || s == "0" { return 0, nil diff --git a/grafana-alertcheck/internal/gate/duration_test.go b/grafana-alertcheck/internal/gate/duration_test.go index ba3a1629a..6262c713e 100644 --- a/grafana-alertcheck/internal/gate/duration_test.go +++ b/grafana-alertcheck/internal/gate/duration_test.go @@ -3,6 +3,8 @@ package gate import ( "testing" "time" + + "github.com/stretchr/testify/require" ) func TestParsePromDuration(t *testing.T) { @@ -35,13 +37,8 @@ func TestParsePromDuration(t *testing.T) { } for _, c := range cases { got, err := ParsePromDuration(c.in) - if err != nil { - t.Errorf("ParsePromDuration(%q): unexpected error: %v", c.in, err) - continue - } - if got != c.want { - t.Errorf("ParsePromDuration(%q) = %v, want %v", c.in, got, c.want) - } + require.NoErrorf(t, err, "ParsePromDuration(%q)", c.in) + require.Equalf(t, c.want, got, "ParsePromDuration(%q)", c.in) } } @@ -60,8 +57,7 @@ func TestParsePromDuration_Errors(t *testing.T) { "carrot", // completely invalid } for _, in := range cases { - if _, err := ParsePromDuration(in); err == nil { - t.Errorf("ParsePromDuration(%q): expected an error, got none", in) - } + _, err := ParsePromDuration(in) + require.Errorf(t, err, "ParsePromDuration(%q): expected an error, got none", in) } } diff --git a/grafana-alertcheck/internal/gate/flock.go b/grafana-alertcheck/internal/gate/flock.go index af4cf99fe..ae21f943b 100644 --- a/grafana-alertcheck/internal/gate/flock.go +++ b/grafana-alertcheck/internal/gate/flock.go @@ -8,7 +8,7 @@ import ( ) // lockExclusive takes a non-blocking exclusive lock on f. Non-blocking is the -// point (§8): a second writer must fail immediately with an error the operator +// point: a second writer must fail immediately with an error the operator // sees, not queue behind the first and start appending to a log somebody else // already finished. func lockExclusive(f *os.File) error { @@ -23,3 +23,23 @@ func lockExclusive(f *os.File) error { func isLockContention(err error) bool { return errors.Is(err, syscall.EWOULDBLOCK) || errors.Is(err, syscall.EAGAIN) } + +// tryLockExclusive is the same call read as a question rather than as a +// demand: held is false when another process holds the lock, and err is +// non-nil only for a failure that is not contention. +// +// check needs that distinction where NewWriter does not. NewWriter is entitled +// to treat any refusal as "another writer has it", because it wants the lock; +// check only wants to know whether a writer EXISTS. The lock answers +// that directly, where a pid can only infer it — the kernel releases a flock +// when the holder exits, crash included, and pids get reused. +func tryLockExclusive(f *os.File) (held bool, err error) { + switch err := syscall.Flock(int(f.Fd()), syscall.LOCK_EX|syscall.LOCK_NB); { + case err == nil: + return true, nil + case errors.Is(err, syscall.EWOULDBLOCK): + return false, nil + default: + return false, fmt.Errorf("flock %s: %w", f.Name(), err) + } +} diff --git a/grafana-alertcheck/internal/gate/flock_test.go b/grafana-alertcheck/internal/gate/flock_test.go index 3e9ef268c..3dfdd8b73 100644 --- a/grafana-alertcheck/internal/gate/flock_test.go +++ b/grafana-alertcheck/internal/gate/flock_test.go @@ -5,18 +5,16 @@ import ( "fmt" "syscall" "testing" + + "github.com/stretchr/testify/require" ) func TestIsLockContention(t *testing.T) { contended := []error{syscall.EWOULDBLOCK, syscall.EAGAIN} for _, e := range contended { - if !isLockContention(e) { - t.Errorf("isLockContention(%v) = false, want true", e) - } + require.Truef(t, isLockContention(e), "isLockContention(%v)", e) // lockExclusive wraps the raw error via fmt.Errorf("flock: %w", ...). - if !isLockContention(fmt.Errorf("flock: %w", e)) { - t.Errorf("isLockContention(wrapped %v) = false, want true", e) - } + require.Truef(t, isLockContention(fmt.Errorf("flock: %w", e)), "isLockContention(wrapped %v)", e) } notContended := []error{ @@ -28,11 +26,7 @@ func TestIsLockContention(t *testing.T) { errors.New("something else"), } for _, e := range notContended { - if isLockContention(e) { - t.Errorf("isLockContention(%v) = true, want false (not a contender)", e) - } - if isLockContention(fmt.Errorf("flock: %w", e)) { - t.Errorf("isLockContention(wrapped %v) = true, want false", e) - } + require.Falsef(t, isLockContention(e), "isLockContention(%v)", e) + require.Falsef(t, isLockContention(fmt.Errorf("flock: %w", e)), "isLockContention(wrapped %v)", e) } } diff --git a/grafana-alertcheck/internal/gate/jsonreq.go b/grafana-alertcheck/internal/gate/jsonreq.go index bfddca382..5b0799a55 100644 --- a/grafana-alertcheck/internal/gate/jsonreq.go +++ b/grafana-alertcheck/internal/gate/jsonreq.go @@ -8,7 +8,7 @@ import ( // req decodes m[key] into *dst. It returns an error when key is absent from m // or explicitly JSON null, so a caller can never mistake absence for a zero -// value (H1) — json.Unmarshal treats "null" as a documented no-op for +// value — json.Unmarshal treats "null" as a documented no-op for // non-pointer targets (string, bool, int, ...), so without this check a // required field sent as null would silently pass through as its zero value. func req[T any](m map[string]json.RawMessage, key string, dst *T) error { diff --git a/grafana-alertcheck/internal/gate/jsonreq_test.go b/grafana-alertcheck/internal/gate/jsonreq_test.go index bf3881e4b..72b446503 100644 --- a/grafana-alertcheck/internal/gate/jsonreq_test.go +++ b/grafana-alertcheck/internal/gate/jsonreq_test.go @@ -3,14 +3,14 @@ package gate import ( "encoding/json" "testing" + + "github.com/stretchr/testify/require" ) func rawMap(t *testing.T, jsonObj string) map[string]json.RawMessage { t.Helper() var m map[string]json.RawMessage - if err := json.Unmarshal([]byte(jsonObj), &m); err != nil { - t.Fatalf("rawMap: %v", err) - } + require.NoError(t, json.Unmarshal([]byte(jsonObj), &m)) return m } @@ -19,34 +19,24 @@ func TestReq(t *testing.T) { t.Run("present key decodes", func(t *testing.T) { var s string - if err := req(m, "present", &s); err != nil { - t.Fatalf("unexpected error: %v", err) - } - if s != "hello" { - t.Errorf("got %q, want hello", s) - } + require.NoError(t, req(m, "present", &s)) + require.Equal(t, "hello", s) }) t.Run("absent key errors", func(t *testing.T) { var s string - if err := req(m, "missing", &s); err == nil { - t.Fatalf("expected an error, got none") - } + require.Error(t, req(m, "missing", &s)) }) t.Run("wrong type errors", func(t *testing.T) { var s string - if err := req(m, "wrongtype", &s); err == nil { - t.Fatalf("expected an error, got none") - } + require.Error(t, req(m, "wrongtype", &s)) }) t.Run("explicit JSON null errors, never a zero value", func(t *testing.T) { var s string err := req(m, "nullval", &s) - if err == nil { - t.Fatalf("expected an error, got none (s=%q) — a null required field must not silently become a zero value", s) - } + require.Error(t, err, "a null required field must not silently become a zero value") }) } @@ -55,38 +45,24 @@ func TestOpt(t *testing.T) { t.Run("present key decodes", func(t *testing.T) { var s string - if err := opt(m, "present", &s); err != nil { - t.Fatalf("unexpected error: %v", err) - } - if s != "hello" { - t.Errorf("got %q, want hello", s) - } + require.NoError(t, opt(m, "present", &s)) + require.Equal(t, "hello", s) }) t.Run("absent key leaves dst untouched", func(t *testing.T) { s := "unchanged" - if err := opt(m, "missing", &s); err != nil { - t.Fatalf("unexpected error: %v", err) - } - if s != "unchanged" { - t.Errorf("got %q, want unchanged", s) - } + require.NoError(t, opt(m, "missing", &s)) + require.Equal(t, "unchanged", s) }) t.Run("wrong type errors", func(t *testing.T) { var s string - if err := opt(m, "wrongtype", &s); err == nil { - t.Fatalf("expected an error, got none") - } + require.Error(t, opt(m, "wrongtype", &s)) }) t.Run("explicit JSON null leaves dst at its zero value", func(t *testing.T) { var s string - if err := opt(m, "nullval", &s); err != nil { - t.Fatalf("unexpected error: %v", err) - } - if s != "" { - t.Errorf("got %q, want empty string", s) - } + require.NoError(t, opt(m, "nullval", &s)) + require.Equal(t, "", s) }) } diff --git a/grafana-alertcheck/internal/gate/log.go b/grafana-alertcheck/internal/gate/log.go index a66b508e1..f8f52d01f 100644 --- a/grafana-alertcheck/internal/gate/log.go +++ b/grafana-alertcheck/internal/gate/log.go @@ -1,6 +1,7 @@ package gate import ( + "bufio" "encoding/json" "fmt" "os" @@ -12,11 +13,11 @@ import ( // LogSchemaVersion is the version stamped into every log header. A log with // any other value is a read error, never a best-effort read: the log is the -// gate's only evidence, and misreading a stale shape is a fail-open (§5). +// gate's only evidence, and misreading a stale shape is a fail-open. const LogSchemaVersion = 1 // RecordType tags each JSONL line. There are exactly three, and a poll record -// IS the heartbeat — there is deliberately no separate heartbeat type (§4.6). +// IS the heartbeat — there is deliberately no separate heartbeat type. type RecordType string const ( @@ -27,53 +28,78 @@ const ( // missingSeriesReason is the reason Grafana parks a disappearing series at // ("Normal (MissingSeries)") for a couple of evaluations before deleting the -// instance. Reading that as a recovery is H2's named bug, so the markers below -// route it to Vanished (P1.2a). +// instance. Reading that as a recovery would turn a disappearing series into a +// fake recovery, so the markers below route it to Vanished. const missingSeriesReason = "MissingSeries" // LoggedRule is the per-rule identity written into the header. Together with -// the header URL it IS the log's identity, which check validates (§19.1 step -// 3), and it supplies the alert set in check mode. +// the header URL it IS the log's identity, which check validates, and it +// supplies the alert set in check mode. type LoggedRule struct { UID string `json:"uid"` Title string `json:"title"` Folder string `json:"folder"` Group string `json:"group"` - // ForSeconds, IntervalSeconds, IsPaused, NoDataState and ExecErrState are - // purely forensic: a resolve-time snapshot that makes the uploaded - // artifact self-describing to a human reading it after the runner is gone - // (§21.3). check never converts them back into a Definition — it always - // re-resolves definitions from the ruler API (§19.1 step 2). + // ForSeconds, IntervalSeconds, NoDataState and ExecErrState are purely + // forensic: a resolve-time snapshot that makes the uploaded artifact + // self-describing to a human reading it after the runner is gone. check + // never converts them back into a Definition — it always re-resolves + // definitions from the ruler API. ForSeconds float64 `json:"for_seconds"` IntervalSeconds int `json:"interval_seconds"` - IsPaused bool `json:"is_paused"` - NoDataState string `json:"no_data_state"` - ExecErrState string `json:"exec_err_state"` - // PollEverySeconds is the cadence this recording ACTUALLY used, after any - // --poll-interval override. Load-bearing, not forensic: check derives - // maxGap from it and never re-derives it from the definitions. Getting - // that wrong is fail-open in the faster-override direction — a real - // recorder gap would pass silently (see "Two authorities", P5). + // IsPaused is load-bearing (beside PollEverySeconds): the pause state at + // record start, the only moment `skipped` can honestly mean. decide reads + // it via Header.pausedAtStart, never a ruler read taken after the window. + IsPaused bool `json:"is_paused"` + NoDataState string `json:"no_data_state"` + ExecErrState string `json:"exec_err_state"` + // PollEverySeconds is the cadence this recording ACTUALLY used. Load-bearing: + // check derives maxGap from it, never from the definitions — getting that + // wrong is fail-open in the faster-override direction. PollEverySeconds float64 `json:"poll_every_seconds"` } // Header is the log's first line: what was recorded, from where, and when the // recording started. It carries no States field — recording is deliberately // unfiltered, so the same log can be re-classified under different --states -// without re-recording (P6). +// without re-recording. type Header struct { SchemaVersion int `json:"schema_version"` - URL string `json:"url"` // the log's identity (§19.1 step 3) + URL string `json:"url"` // the log's identity GrafanaVersion string `json:"grafana_version"` - StartedAt time.Time `json:"started_at"` // the record start (§7 validation) - Rules []LoggedRule `json:"rules"` // THE alert set (§19.1 step 3) + StartedAt time.Time `json:"started_at"` // the record start + Rules []LoggedRule `json:"rules"` // THE alert set +} + +// pausedAtStart reports, per rule UID, whether the rule was paused when the +// recording opened. That instant — and no other — is what `skipped` means: a +// rule nobody was watching on purpose. +// +// It is the authority for `skipped` in BOTH modes, and the reason is that no +// other source knows the right moment. `check` re-resolves the definitions +// AFTER the window closed, so Definition.IsPaused there describes the present, +// not the window: a rule that fired and was then paused would read as skipped, +// its firing would never be classified, and under --allow-paused the run would +// pass. The header cannot drift that way, because watch stamps it before the +// deploy step runs and single-step check stamps it from definitions resolved at +// the start of its own step. +// +// A UID the header does not name is reported NOT paused, which is the safe +// direction: it then reaches proveCoverage with no polls and fails closed as +// a heartbeat gap, rather than being waved through as legitimately unwatched. +func (h Header) pausedAtStart() map[string]bool { + paused := make(map[string]bool, len(h.Rules)) + for _, lr := range h.Rules { + paused[lr.UID] = lr.IsPaused + } + return paused } // Poll is one reduced observation of one rule — the log's heartbeat and the // only input the pure coverage and classification layers ever see. type Poll struct { RuleUID string `json:"rule_uid"` - GrafanaNow time.Time `json:"grafana_now"` // the Date header — H4 + GrafanaNow time.Time `json:"grafana_now"` // the response's Date header // SkewMS, SkewBoundMS and LatencyMS are milliseconds for JSONL // compactness ONLY. The pure layer never touches raw ms: it reads // Skew(), SkewBound() and Latency() below, which convert at the @@ -81,59 +107,55 @@ type Poll struct { SkewMS int64 `json:"skew_ms"` SkewBoundMS int64 `json:"skew_bound_ms"` LatencyMS int64 `json:"latency_ms"` - // Found false means an authoritative 2xx in which this rule was absent - // (§14.5) — never a transport failure, which P2 retried and never turns - // into a Poll. P7 check 8 turns it into unobservable. + // Found false means an authoritative 2xx in which this rule was absent — + // never a transport failure, which the transport retries and never turns + // into a Poll. The coverage proof turns it into unobservable. Found bool `json:"found"` // State, Health and LastError are the raw rule-level strings, reporting - // only and never classified (P1.2a). + // only and never classified. State string `json:"state,omitempty"` Health string `json:"health,omitempty"` LastError string `json:"last_error,omitempty"` - // omitzero, not omitempty: a not-found poll (and a paused rule, §2.3) has - // no evaluation time, and writing "0001-01-01T00:00:00Z" into an artifact - // humans and jq read (§21.3) invites reading it as a real timestamp. + // omitzero, not omitempty: a not-found poll (and a paused rule) has no + // evaluation time, and writing "0001-01-01T00:00:00Z" into an artifact + // humans and jq read invites reading it as a real timestamp. LastEvaluation time.Time `json:"last_evaluation,omitzero"` IsPaused bool `json:"is_paused"` - Histogram map[string]int `json:"histogram,omitempty"` // §4.9 — written, never analysed + Histogram map[string]int `json:"histogram,omitempty"` // written, never analysed // Reasons counts this poll's non-empty instance reasons, e.g. - // {"NoData":1091,"Error":14}; nil when none. Reporting-only, and the ONLY - // place composite states stay visible: they are canonical normal (so they - // are dropped from Abnormal) and `totals` never carries composite keys. - // - // The KEYS are raw reason strings and can be comma-joined composites - // ("KeepLast, MissingSeries") — newer Grafana versions join several - // reasons into one. So any consumer, P7 check 9's KeepLast note included, - // must test membership across the keys with reasonNames and must NEVER - // index a literal key: reasons["KeepLast"] misses every composite. + // {"NoData":1091,"Error":14}; nil when none. Reporting-only, and the only + // place composite states stay visible (they are canonical normal, dropped + // from Abnormal). Keys are raw reason strings and can be comma-joined + // composites ("KeepLast, MissingSeries"), so consumers must test membership + // via reasonNames and never index a literal key. Reasons map[string]int `json:"reasons,omitempty"` - // Abnormal holds the instances whose CANONICAL state is not normal - // (§4.6). "Normal (NoData)" and "Normal (Error)" are canonical normal and - // are deliberately not retained here (P1.2a). + // Abnormal holds the instances whose CANONICAL state is not normal. + // "Normal (NoData)" and "Normal (Error)" are canonical normal and are + // deliberately not retained here. Abnormal []Instance `json:"abnormal,omitempty"` - // Cleared and Vanished are instance keys (§4.7): keys that left the - // abnormal set, resolved against the SAME response — a clear and a - // discontinuity are not the same fact (H2). + // Cleared and Vanished are instance keys that left the abnormal set, + // resolved against the SAME response — a clear and a discontinuity are not + // the same fact. Cleared []string `json:"cleared,omitempty"` Vanished []string `json:"vanished,omitempty"` } -// Skew is the signed clock skew of this poll (§16). +// Skew is the signed clock skew of this poll. func (p Poll) Skew() time.Duration { return time.Duration(p.SkewMS) * time.Millisecond } // SkewBound is the uncertainty on Skew — the tolerance every cross-domain -// comparison in P7 applies alongside it. +// comparison applies alongside it. func (p Poll) SkewBound() time.Duration { return time.Duration(p.SkewBoundMS) * time.Millisecond } -// Latency is the wall time this poll's request took, feeding §5.2's budget check. +// Latency is the wall time this poll's request took, feeding the budget check. func (p Poll) Latency() time.Duration { return time.Duration(p.LatencyMS) * time.Millisecond } // Reducer turns each Observation into the single Poll record that goes into // the log. It holds the previous poll's abnormal instance keys per rule, which -// is all the state the transition markers need (§4.7). +// is all the state the transition markers need. // // A Reducer is safe for concurrent use: watch polls a fleet of rules -// concurrently (P6) and every one of those goroutines reduces through the same +// concurrently and every one of those goroutines reduces through the same // instance, because the per-rule marker state has to live in one place. The // lock is per-Reducer rather than per-rule — Reduce only touches maps and // slices, so it never blocks on I/O while holding it. @@ -148,12 +170,12 @@ func NewReducer() *Reducer { // Reduce selects the rule identified by uid out of obs and reduces it to a // Poll. Selection is BY UID, never by title: a filtered response can carry -// several rules sharing one title (the known 2-way collision, §14.5), and -// picking the first would silently watch the wrong rule. +// several rules sharing one title, and picking the first would silently watch +// the wrong rule. // -// The reduction (§4.6) keeps the rule-level fields, the raw totals histogram, -// the reason counts, and only the instances whose canonical state is not -// normal. That makes per-poll size independent of NORMAL cardinality — not of +// The reduction keeps the rule-level fields, the raw totals histogram, the +// reason counts, and only the instances whose canonical state is not normal. +// That makes per-poll size independent of NORMAL cardinality — not of // cardinality outright: a rule with 449 firing instances still stores all 449. func (r *Reducer) Reduce(uid string, obs Observation) Poll { r.mu.Lock() @@ -167,13 +189,7 @@ func (r *Reducer) Reduce(uid string, obs Observation) Poll { LatencyMS: obs.Latency.Milliseconds(), } - var rule *StateRule - for i := range obs.Rules { - if obs.Rules[i].UID == uid { - rule = &obs.Rules[i] - break - } - } + rule := stateRuleByUID(obs.Rules, uid) if rule == nil { // An authoritative "the rule is absent". No markers are computed and // the previous abnormal set is kept untouched: if the rule comes back @@ -191,8 +207,8 @@ func (r *Reducer) Reduce(uid string, obs Observation) Poll { p.Histogram = rule.Totals // present indexes every instance in THIS response, normal ones included — - // the markers below must resolve a departed key against the same response - // (H2), which is impossible from the abnormal subset alone. + // the markers below must resolve a departed key against the same response, + // which is impossible from the abnormal subset alone. present := make(map[string]Instance, len(rule.Instances)) curAbnormal := make(map[string]struct{}) for _, inst := range rule.Instances { @@ -221,7 +237,7 @@ func (r *Reducer) Reduce(uid string, obs Observation) Poll { p.Vanished = append(p.Vanished, key) case reasonNames(inst.Reason, missingSeriesReason): // The vanish in disguise, caught one poll earlier than the fully - // absent case — H2's named bug. + // absent case. p.Vanished = append(p.Vanished, key) default: // Present as canonical normal without a MissingSeries reason. @@ -243,10 +259,10 @@ func (r *Reducer) Reduce(uid string, obs Observation) Poll { // // It exists for the one place a recording changes hands: watch's parent takes // the first observation of every rule and its detached child continues from -// there (P6). Without the seed, an instance that is abnormal in the parent's +// there. Without the seed, an instance that is abnormal in the parent's // observation and gone by the child's first poll produces no marker at all — -// it leaves the record as though it had never been bad, which is H2's -// fail-open reached through the handoff rather than through a reason string. +// it leaves the record as though it had never been bad, the same fail-open a +// misread MissingSeries causes, reached through the handoff instead. // // Not-found polls are skipped, mirroring Reduce: an absent rule leaves the // previous abnormal set untouched rather than emptying it. @@ -265,9 +281,22 @@ func (r *Reducer) seedFrom(polls []Poll) { } } +// stateRuleByUID picks one rule out of a state response BY UID (nil = the +// authoritative "rule absent"). Never by title: the ?rule_name= filter is a +// title filter and can return several rules sharing a title. The single +// selection for the package — Reduce and the drain wait both use it. +func stateRuleByUID(rules []StateRule, uid string) *StateRule { + for i := range rules { + if rules[i].UID == uid { + return &rules[i] + } + } + return nil +} + // reasonNames reports whether reason names want. Newer Grafana versions // comma-join several reasons into one string, so this tests membership rather -// than equality (P7 check 9 needs the same test for KeepLast). +// than equality. func reasonNames(reason, want string) bool { for part := range strings.SplitSeq(reason, ",") { if strings.TrimSpace(part) == want { @@ -277,19 +306,12 @@ func reasonNames(reason, want string) bool { return false } -// VerifyNormalInstancesVisible checks §3.2's assumption on a first -// observation: that the state endpoint really does return normal instances, -// not only the abnormal ones. If it ever stops doing so, the reduction's -// "keep the non-normal instances" becomes "keep everything the API happened to -// send" and the transition markers lose their ground truth — a silent -// fail-open. So this is verified at start, never assumed. -// -// The counts are summed over every totals key whose LOWERCASED name is -// "normal" or "inactive". Never index one literal key: the captured -// vocabulary is mixed across rules ({"alerting":445,"normal":2004} on one, -// {"firing":2,"inactive":363} on another) and its case has already drifted -// from the original recon. Composite states never appear in totals — Grafana -// counts a "Normal (NoData)" instance under normal. +// VerifyNormalInstancesVisible checks, on a first observation, that the state +// endpoint really returns normal instances: if it ever stops, the reduction's +// "keep the non-normal" becomes "keep everything the API sent", a silent +// fail-open in the transition markers. Counts are summed over every totals key +// whose lowercased name is "normal" or "inactive" — never a literal key, since +// the vocabulary is mixed and its case has drifted. func VerifyNormalInstancesVisible(rules []StateRule) error { for _, r := range rules { var claimed int @@ -307,7 +329,7 @@ func VerifyNormalInstancesVisible(rules []StateRule) error { } return fmt.Errorf( "rule %q (%s): totals claim %d normal instances but the response returned none — "+ - "the state endpoint no longer returns normal instances, which the §3.2 reduction depends on", + "the state endpoint no longer returns normal instances, which the reduction depends on", r.Title, r.UID, claimed) } return nil @@ -340,10 +362,10 @@ type stoppedRecord struct { At time.Time `json:"at"` } -// Writer appends records to the JSONL log. It is append-only by construction -// (§8): O_APPEND|O_CREATE|O_WRONLY, never O_TRUNC, so no writer can ever -// destroy evidence a previous one recorded. An exclusive non-blocking flock -// makes a second writer fail immediately rather than interleave. +// Writer appends records to the JSONL log. It is append-only by construction — +// O_APPEND|O_CREATE|O_WRONLY, never O_TRUNC — so no writer can ever destroy +// evidence a previous one recorded. An exclusive non-blocking flock makes a +// second writer fail immediately rather than interleave. type Writer struct { mu sync.Mutex f *os.File @@ -374,8 +396,8 @@ func NewWriter(path string, clock Clock) (*Writer, error) { // WriteHeader writes line 1 and stamps the current schema version, so no // caller can leave it at zero. It refuses a non-empty file: the log already // has a header, and a second one would make ReadLog's "header is line 1" -// contract a lie. In the P6 handoff the parent writes the header and the child -// only appends polls. +// contract a lie. In watch's handoff the parent writes the header and the +// detached child only appends polls. func (w *Writer) WriteHeader(h Header) error { w.mu.Lock() defer w.mu.Unlock() @@ -409,19 +431,12 @@ func (w *Writer) WritePoll(p Poll) error { return nil } -// Stop finishes recording in the fixed §4.4 order, which must not be -// reordered: let the in-flight write finish (the mutex), append the stopped -// sentinel, fsync, then release. Any other order can leave a log whose last -// durable byte is a sentinel that was never actually preceded by the polls it -// vouches for. -// -// Stop writes the sentinel with the recorder's OWN stop time and makes no -// comparison against `to` — watch never knows `to` or the transition grace. -// check does that comparison, after this writer has exited (§4.5). -// -// Calling Stop twice is a no-op: watch reaches it from both a signal handler -// and a defer, and a second sentinel would be indistinguishable from a second -// writer. +// Stop finishes recording in a fixed order that must not be rearranged: let the +// in-flight write finish, append the sentinel, fsync, release — any other order +// can leave a sentinel that was never preceded by the polls it vouches for. +// The sentinel uses the writer's OWN stop time; check does the `to` comparison +// after this has exited. Calling Stop twice is a no-op (watch reaches it from a +// signal handler and a defer). func (w *Writer) Stop() error { w.mu.Lock() defer w.mu.Unlock() @@ -447,9 +462,8 @@ func (w *Writer) Stop() error { // Close releases the file and the lock WITHOUT writing a sentinel. It exists // for exactly one caller: watch's parent, which writes the header and then -// hands the log to the detached child that will finish it (P6). A sentinel -// here would tell check the recording ended before the child had even -// started. +// hands the log to the detached child that will finish it. A sentinel here +// would tell check the recording ended before the child had even started. func (w *Writer) Close() error { w.mu.Lock() defer w.mu.Unlock() @@ -463,21 +477,52 @@ func (w *Writer) Close() error { return nil } +// ReadLogHeader reads ONLY line 1 — the one read safe while a writer may still +// hold the log. The header is written once by watch's parent before any child +// appends a byte, so line 1 is immutable. It lets check fail closed EARLY on a +// wrong URL or unresolvable rule; it is advisory only, and the authoritative +// identity read is still ReadLog after the writer exits. +func ReadLogHeader(path string) (Header, error) { + f, err := os.Open(path) + if err != nil { + return Header{}, fmt.Errorf("read log header %s: %w", path, err) + } + defer f.Close() + + line, err := bufio.NewReader(f).ReadString('\n') + if err != nil { + // io.EOF included: a log whose first line has no terminating newline is + // a log whose header was never fully written, which is not a header. + return Header{}, fmt.Errorf("log %s: no complete header on line 1: %w", path, err) + } + + var rec headerRecord + if err := json.Unmarshal([]byte(line), &rec); err != nil { + return Header{}, fmt.Errorf("log %s line 1: unparseable header: %w", path, err) + } + if rec.Type != RecordHeader { + return Header{}, fmt.Errorf("log %s line 1: got record type %q; the header must be line 1", path, rec.Type) + } + if rec.SchemaVersion != LogSchemaVersion { + return Header{}, fmt.Errorf( + "log %s: schema version %d is not %d — this log was written by a different version of the gate", + path, rec.SchemaVersion, LogSchemaVersion) + } + return rec.Header, nil +} + // ReadLog reads the whole log once and returns its header, its polls in // recorded order, and the sentinel time when one is present (nil when the // recording never finished — check turns that into unobservable, never a // pass). // -// Call this only after the writer has exited (§4.4 step 4). Reading a log a -// writer can still append to can only produce a shorter window than the one -// that was recorded. +// Call this only after the writer has exited. Reading a log a writer can still +// append to can only produce a shorter window than the one that was recorded. // -// The parse rules are deliberately the crudest possible (§24.2): the header -// must be line 1 with a matching schema version, and ANY unparseable line — -// including the last one, and including a last line that follows a sentinel — -// is an error, full stop. No heuristics, no discarding an untidy tail: a -// truncated log is evidence that something killed the recorder, which is -// exactly what must not pass. +// The parse rules are deliberately the crudest possible: the header must be +// line 1 with a matching schema version, and ANY unparseable line — including +// the last, or one after a sentinel — is an error, full stop. A truncated log +// is evidence something killed the recorder, which must not pass. func ReadLog(path string) (Header, []Poll, *time.Time, error) { b, err := os.ReadFile(path) if err != nil { diff --git a/grafana-alertcheck/internal/gate/log_test.go b/grafana-alertcheck/internal/gate/log_test.go index 494c9535c..8ef3863e1 100644 --- a/grafana-alertcheck/internal/gate/log_test.go +++ b/grafana-alertcheck/internal/gate/log_test.go @@ -5,17 +5,18 @@ import ( "fmt" "os" "path/filepath" - "reflect" "strings" "sync" "testing" "time" + + "github.com/stretchr/testify/require" ) // Every time literal in this file is UTC and built with time.Date, so it // carries no monotonic reading and survives a JSON round trip byte-identical — -// which is what lets the round-trip tests below use reflect.DeepEqual on whole -// Poll values instead of comparing field by field. +// which is what lets the round-trip tests below compare whole Poll values +// instead of comparing field by field. var testNow = time.Date(2026, 8, 31, 9, 0, 0, 0, time.UTC) func testInstance(state State, reason, instanceLabel string) Instance { @@ -48,7 +49,7 @@ func TestLogReduceKeepsOnlyAbnormalInstances(t *testing.T) { Instances: []Instance{ testInstance(StateNormal, "", "a"), testInstance(StateFiring, "", "b"), - // Both composites are canonical normal (P1.2a): they must NOT be + // Both composites are canonical normal: they must NOT be // retained as abnormal, and their reasons must still be counted. testInstance(StateNormal, "NoData", "c"), testInstance(StateNormal, "Error", "d"), @@ -57,34 +58,24 @@ func TestLogReduceKeepsOnlyAbnormalInstances(t *testing.T) { p := NewReducer().Reduce("rule1", observation(testNow, rule)) - if !p.Found { - t.Fatalf("Found = false, want true") - } - if len(p.Abnormal) != 1 || p.Abnormal[0].Labels["instance"] != "b" { - t.Errorf("Abnormal = %+v, want only the firing instance b", p.Abnormal) - } - if want := map[string]int{"NoData": 1, "Error": 1}; !reflect.DeepEqual(p.Reasons, want) { - t.Errorf("Reasons = %v, want %v", p.Reasons, want) - } + require.True(t, p.Found) + require.Len(t, p.Abnormal, 1) + require.Equal(t, "b", p.Abnormal[0].Labels["instance"]) + require.Equal(t, map[string]int{"NoData": 1, "Error": 1}, p.Reasons) // The histogram is a verbatim copy of the response totals — raw keys, no - // normalization (§4.9). - if want := map[string]int{"alerting": 1, "normal": 2}; !reflect.DeepEqual(p.Histogram, want) { - t.Errorf("Histogram = %v, want %v", p.Histogram, want) - } - // Rule-level state and health stay raw and unnormalized (P1.2a). - if p.State != "firing" || p.Health != "ok" { - t.Errorf("State/Health = %q/%q, want firing/ok", p.State, p.Health) - } - if p.Skew() != 1500*time.Millisecond || p.SkewBound() != 40*time.Millisecond || p.Latency() != 1800*time.Millisecond { - t.Errorf("durations = %s/%s/%s, want 1.5s/40ms/1.8s", p.Skew(), p.SkewBound(), p.Latency()) - } - if p.Reasons["MissingSeries"] != 0 { - t.Errorf("unexpected MissingSeries count") - } + // normalization. + require.Equal(t, map[string]int{"alerting": 1, "normal": 2}, p.Histogram) + // Rule-level state and health stay raw and unnormalized. + require.Equal(t, "firing", p.State) + require.Equal(t, "ok", p.Health) + require.Equal(t, 1500*time.Millisecond, p.Skew()) + require.Equal(t, 40*time.Millisecond, p.SkewBound()) + require.Equal(t, 1800*time.Millisecond, p.Latency()) + require.Zero(t, p.Reasons["MissingSeries"]) } -// A filtered response can hold several rules sharing one title (the known -// 2-way collision, §14.5), so the reducer must select by UID. +// A filtered response can hold several rules sharing one title, so the reducer +// must select by UID. func TestLogReduceSelectsRuleByUID(t *testing.T) { first := StateRule{UID: "ruleA", Title: "Same Title", Health: "ok", State: "inactive", LastEvaluation: testNow} second := StateRule{ @@ -94,9 +85,8 @@ func TestLogReduceSelectsRuleByUID(t *testing.T) { p := NewReducer().Reduce("ruleB", observation(testNow, first, second)) - if p.Health != "error" || len(p.Abnormal) != 1 { - t.Errorf("reduced the wrong rule: %+v", p) - } + require.Equal(t, "error", p.Health) + require.Len(t, p.Abnormal, 1) } func TestLogReduceRuleAbsentIsAuthoritative(t *testing.T) { @@ -104,23 +94,17 @@ func TestLogReduceRuleAbsentIsAuthoritative(t *testing.T) { p := NewReducer().Reduce("rule1", observation(testNow, other)) - if p.Found { - t.Errorf("Found = true, want false for a rule absent from an authoritative 2xx") - } - if p.RuleUID != "rule1" { - t.Errorf("RuleUID = %q, want rule1 — an absent rule is still attributed", p.RuleUID) - } + require.False(t, p.Found, "a rule absent from an authoritative 2xx") + require.Equal(t, "rule1", p.RuleUID, "an absent rule is still attributed") // The heartbeat still exists: a not-found poll is evidence that Grafana // answered at this time, which the coverage proof reads. - if !p.GrafanaNow.Equal(testNow) || p.Latency() == 0 { - t.Errorf("absent-rule poll lost its timing evidence: %+v", p) - } - if p.Health != "" || p.Abnormal != nil { - t.Errorf("absent-rule poll carries rule fields: %+v", p) - } + require.True(t, p.GrafanaNow.Equal(testNow)) + require.NotZero(t, p.Latency()) + require.Empty(t, p.Health) + require.Nil(t, p.Abnormal) } -// H2: an instance that leaves the abnormal set is resolved against the SAME +// An instance that leaves the abnormal set is resolved against the SAME // response, and MissingSeries is a vanish, never a recovery. func TestTransitionMarkersClearedVersusVanished(t *testing.T) { badKey := instanceKey(testInstance(StateFiring, "", "b").Labels) @@ -175,19 +159,14 @@ func TestTransitionMarkersClearedVersusVanished(t *testing.T) { Instances: []Instance{testInstance(StateFiring, "", "b")}, } first := r.Reduce("rule1", observation(testNow, firing)) - if first.Cleared != nil || first.Vanished != nil { - t.Fatalf("first poll produced markers with no previous poll: %+v", first) - } + require.Nil(t, first.Cleared, "first poll produced cleared markers with no previous poll") + require.Nil(t, first.Vanished, "first poll produced vanished markers with no previous poll") next := StateRule{UID: "rule1", Health: "ok", State: "inactive", LastEvaluation: testNow, Instances: c.second} p := r.Reduce("rule1", observation(testNow.Add(30*time.Second), next)) - if !reflect.DeepEqual(p.Cleared, c.wantCleared) { - t.Errorf("Cleared = %q, want %q", p.Cleared, c.wantCleared) - } - if !reflect.DeepEqual(p.Vanished, c.wantVanished) { - t.Errorf("Vanished = %q, want %q", p.Vanished, c.wantVanished) - } + require.Equal(t, c.wantCleared, p.Cleared) + require.Equal(t, c.wantVanished, p.Vanished) }) } } @@ -204,15 +183,12 @@ func TestTransitionMarkersSurviveAnAbsentPoll(t *testing.T) { r.Reduce("rule1", observation(testNow, firing)) absent := r.Reduce("rule1", observation(testNow.Add(30*time.Second))) - if absent.Vanished != nil || absent.Cleared != nil { - t.Fatalf("an absent rule produced markers: %+v", absent) - } + require.Nil(t, absent.Vanished, "an absent rule produced vanished markers") + require.Nil(t, absent.Cleared, "an absent rule produced cleared markers") back := StateRule{UID: "rule1", Health: "ok", State: "inactive", LastEvaluation: testNow} p := r.Reduce("rule1", observation(testNow.Add(60*time.Second), back)) - if len(p.Vanished) != 1 { - t.Errorf("Vanished = %q, want the instance that disappeared across the absent poll", p.Vanished) - } + require.Len(t, p.Vanished, 1, "want the instance that disappeared across the absent poll") } func TestTransitionMarkersAreSortedAndPerRule(t *testing.T) { @@ -234,23 +210,18 @@ func TestTransitionMarkersAreSortedAndPerRule(t *testing.T) { clearedOne := StateRule{UID: "rule1", Health: "ok", State: "inactive", LastEvaluation: testNow} p := r.Reduce("rule1", observation(testNow.Add(time.Minute), clearedOne, ruleTwo)) - if len(p.Vanished) != 3 { - t.Fatalf("Vanished = %q, want 3 keys", p.Vanished) - } + require.Len(t, p.Vanished, 3) for i := 1; i < len(p.Vanished); i++ { - if p.Vanished[i-1] > p.Vanished[i] { - t.Errorf("Vanished is not sorted: %q", p.Vanished) - } + require.Less(t, p.Vanished[i-1], p.Vanished[i], "Vanished is not sorted: %q", p.Vanished) } // rule2's own abnormal set is untouched by rule1's transitions. q := r.Reduce("rule2", observation(testNow.Add(time.Minute), clearedOne, ruleTwo)) - if q.Cleared != nil || q.Vanished != nil { - t.Errorf("rule2 picked up rule1's transitions: %+v", q) - } + require.Nil(t, q.Cleared, "rule2 picked up rule1's transitions") + require.Nil(t, q.Vanished, "rule2 picked up rule1's transitions") } -// §3.2: the reduction depends on the state endpoint returning normal instances. -// If it ever stops, that must fail loudly at start, never be assumed. +// The reduction depends on the state endpoint returning normal instances. If it +// ever stops, that must fail loudly at start, never be assumed. func TestLogVerifyNormalInstancesVisible(t *testing.T) { cases := []struct { fixture string @@ -267,22 +238,14 @@ func TestLogVerifyNormalInstancesVisible(t *testing.T) { for _, c := range cases { t.Run(c.fixture, func(t *testing.T) { rules, err := ParseState(readFixture(t, c.fixture)) - if err != nil { - t.Fatalf("ParseState: %v", err) - } + require.NoError(t, err) err = VerifyNormalInstancesVisible(rules) if c.wantError { - if err == nil { - t.Fatalf("VerifyNormalInstancesVisible: want an error, got nil") - } - if !strings.Contains(err.Error(), "§3.2") { - t.Errorf("error does not name §3.2: %v", err) - } + require.Error(t, err) + require.Contains(t, err.Error(), "no longer returns normal instances") return } - if err != nil { - t.Fatalf("VerifyNormalInstancesVisible: unexpected error: %v", err) - } + require.NoError(t, err) }) } } @@ -310,9 +273,7 @@ func TestLogVerifyNormalInstancesVisibleVocabularies(t *testing.T) { Instances: []Instance{testInstance(StateFiring, "", "b")}, }} err := VerifyNormalInstancesVisible(rules) - if (err != nil) != c.wantError { - t.Errorf("VerifyNormalInstancesVisible: error = %v, want error = %v", err, c.wantError) - } + require.Equal(t, c.wantError, err != nil) }) } } @@ -328,37 +289,25 @@ func TestLogModeCadenceComesFromTheHeader(t *testing.T) { h.Rules[0].PollEverySeconds = 5 // an operator override far tighter than the default 150s rt, _, err := DeriveTimingsFromLog(h, defs) - if err != nil { - t.Fatalf("DeriveTimingsFromLog: %v", err) - } + require.NoError(t, err) got := rt["rule1"] - if got.pollEvery != 5*time.Second { - t.Errorf("pollEvery = %s, want the header's 5s, not the default 150s", got.pollEvery) - } + require.Equal(t, 5*time.Second, got.pollEvery, "the header's 5s, not the default 150s") // maxGap and healthGrace follow the recorded cadence; without this a 250s // hole in a log recorded at 5s would pass silently. - if got.maxGap != 10*time.Second { - t.Errorf("maxGap = %s, want 10s (2 x the recorded cadence)", got.maxGap) - } - if got.healthGrace != 300*time.Second { - t.Errorf("healthGrace = %s, want 300s (max(maxGap, interval))", got.healthGrace) - } + require.Equal(t, 10*time.Second, got.maxGap) + require.Equal(t, 300*time.Second, got.healthGrace) // evalStaleAfter is a rule fact, so it stays 2 x intervalSeconds from the // definitions regardless of how often the gate polled. - if got.evalStaleAfter != 600*time.Second { - t.Errorf("evalStaleAfter = %s, want 600s from the definition's interval", got.evalStaleAfter) - } + require.Equal(t, 600*time.Second, got.evalStaleAfter) // A log that cannot say how often it was written cannot have its coverage // proved, and neither can one naming a rule that no longer resolves. missingCadence := testHeader() missingCadence.Rules[0].PollEverySeconds = 0 - if _, _, err := DeriveTimingsFromLog(missingCadence, defs); err == nil { - t.Errorf("a header with no recorded cadence was accepted") - } - if _, _, err := DeriveTimingsFromLog(testHeader(), nil); err == nil { - t.Errorf("a header naming an unresolvable rule was accepted") - } + require.Error(t, func() error { _, _, err := DeriveTimingsFromLog(missingCadence, defs); return err }(), + "a header with no recorded cadence was accepted") + require.Error(t, func() error { _, _, err := DeriveTimingsFromLog(testHeader(), nil); return err }(), + "a header naming an unresolvable rule was accepted") // A duplicated UID must not resolve last-one-wins: the slower duplicate // would widen maxGap, which is fail-open through log corruption alone. @@ -366,12 +315,11 @@ func TestLogModeCadenceComesFromTheHeader(t *testing.T) { slower := duplicated.Rules[0] slower.PollEverySeconds = 600 duplicated.Rules = append(duplicated.Rules, slower) - if _, _, err := DeriveTimingsFromLog(duplicated, defs); err == nil { - t.Errorf("a header naming one rule twice was accepted") - } + require.Error(t, func() error { _, _, err := DeriveTimingsFromLog(duplicated, defs); return err }(), + "a header naming one rule twice was accepted") } -// watch polls a fleet concurrently through one Reducer (P6), so the marker +// watch polls a fleet concurrently through one Reducer, so the marker // state it holds per rule must be safe under -race — a latent data race here // surfaces as a wrong transition, which is the one thing markers exist to get // right. @@ -398,41 +346,28 @@ func TestLogReduceIsSafeForConcurrentUse(t *testing.T) { // transition — the concurrency must not corrupt the per-rule state either. for _, rule := range rules { p := r.Reduce(rule.UID, obs) - if p.Cleared != nil || p.Vanished != nil { - t.Errorf("rule %s: markers after concurrent reduction: %+v", rule.UID, p) - } + require.Nilf(t, p.Cleared, "rule %s: cleared markers after concurrent reduction", rule.UID) + require.Nilf(t, p.Vanished, "rule %s: vanished markers after concurrent reduction", rule.UID) } } // A not-found poll has no evaluation time, and the artifact is read by humans -// and jq (§21.3) — the zero time must not appear as though it were real. +// and jq — the zero time must not appear as though it were real. func TestLogPollOmitsTheZeroEvaluationTime(t *testing.T) { absent := NewReducer().Reduce("rule1", observation(testNow)) b, err := json.Marshal(pollRecord{Type: RecordPoll, Poll: absent}) - if err != nil { - t.Fatalf("marshal: %v", err) - } - if strings.Contains(string(b), "0001-01-01") { - t.Errorf("a not-found poll wrote the zero time: %s", b) - } - if strings.Contains(string(b), "last_evaluation") { - t.Errorf("a not-found poll wrote last_evaluation at all: %s", b) - } + require.NoError(t, err) + require.NotContains(t, string(b), "0001-01-01") + require.NotContains(t, string(b), "last_evaluation") // A real evaluation time still round-trips. found := StateRule{UID: "rule1", Health: "ok", State: "inactive", LastEvaluation: testNow} p := NewReducer().Reduce("rule1", observation(testNow, found)) b, err = json.Marshal(pollRecord{Type: RecordPoll, Poll: p}) - if err != nil { - t.Fatalf("marshal: %v", err) - } + require.NoError(t, err) var back pollRecord - if err := json.Unmarshal(b, &back); err != nil { - t.Fatalf("unmarshal: %v", err) - } - if !back.LastEvaluation.Equal(testNow) { - t.Errorf("last_evaluation = %s, want %s", back.LastEvaluation, testNow) - } + require.NoError(t, json.Unmarshal(b, &back)) + require.True(t, back.LastEvaluation.Equal(testNow)) } func testHeader() Header { @@ -452,9 +387,7 @@ func newTestWriter(t *testing.T, path string) (*Writer, *fakeClock) { t.Helper() clock := newFakeClock(testNow) w, err := NewWriter(path, clock) - if err != nil { - t.Fatalf("NewWriter: %v", err) - } + require.NoError(t, err) return w, clock } @@ -463,9 +396,7 @@ func TestWriterReadLogRoundTrip(t *testing.T) { w, clock := newTestWriter(t, path) h := testHeader() - if err := w.WriteHeader(h); err != nil { - t.Fatalf("WriteHeader: %v", err) - } + require.NoError(t, w.WriteHeader(h)) r := NewReducer() firing := StateRule{ @@ -479,80 +410,47 @@ func TestWriterReadLogRoundTrip(t *testing.T) { r.Reduce("rule1", observation(testNow.Add(time.Minute), cleared)), } for _, p := range want { - if err := w.WritePoll(p); err != nil { - t.Fatalf("WritePoll: %v", err) - } + require.NoError(t, w.WritePoll(p)) } clock.Advance(2 * time.Minute) - if err := w.Stop(); err != nil { - t.Fatalf("Stop: %v", err) - } + require.NoError(t, w.Stop()) gotHeader, gotPolls, sentinel, err := ReadLog(path) - if err != nil { - t.Fatalf("ReadLog: %v", err) - } + require.NoError(t, err) h.SchemaVersion = LogSchemaVersion // WriteHeader stamps it - if !reflect.DeepEqual(gotHeader, h) { - t.Errorf("header round trip:\n got %+v\nwant %+v", gotHeader, h) - } - if !reflect.DeepEqual(gotPolls, want) { - t.Errorf("poll round trip:\n got %+v\nwant %+v", gotPolls, want) - } - if sentinel == nil { - t.Fatalf("sentinel is nil after Stop") - } + require.Equal(t, h, gotHeader, "header round trip") + require.Equal(t, want, gotPolls, "poll round trip") + require.NotNil(t, sentinel, "sentinel is nil after Stop") // Stop stamps the recorder's own stop time and makes no comparison - // against `to` — watch never knows it (§4.5). - if !sentinel.Equal(testNow.Add(2 * time.Minute)) { - t.Errorf("sentinel = %s, want the writer's stop time %s", sentinel, testNow.Add(2*time.Minute)) - } + // against `to` — watch never knows it. + require.True(t, sentinel.Equal(testNow.Add(2*time.Minute))) } -// §8: the log is append-only. A second run against the same path must never +// The log is append-only. A second run against the same path must never // destroy the evidence the first one recorded. func TestWriterAppendsAndNeverTruncates(t *testing.T) { path := filepath.Join(t.TempDir(), "log.jsonl") w, _ := newTestWriter(t, path) - if err := w.WriteHeader(testHeader()); err != nil { - t.Fatalf("WriteHeader: %v", err) - } - if err := w.WritePoll(Poll{RuleUID: "rule1", Found: true, GrafanaNow: testNow}); err != nil { - t.Fatalf("WritePoll: %v", err) - } - if err := w.Close(); err != nil { - t.Fatalf("Close: %v", err) - } + require.NoError(t, w.WriteHeader(testHeader())) + require.NoError(t, w.WritePoll(Poll{RuleUID: "rule1", Found: true, GrafanaNow: testNow})) + require.NoError(t, w.Close()) before, err := os.ReadFile(path) - if err != nil { - t.Fatalf("read: %v", err) - } + require.NoError(t, err) - // The P6 handoff: the parent wrote the header and closed; the child + // The handoff: the parent wrote the header and closed; the child // reopens the same path and appends without a second header. child, _ := newTestWriter(t, path) - if err := child.WritePoll(Poll{RuleUID: "rule1", Found: true, GrafanaNow: testNow.Add(time.Minute)}); err != nil { - t.Fatalf("child WritePoll: %v", err) - } - if err := child.Stop(); err != nil { - t.Fatalf("child Stop: %v", err) - } + require.NoError(t, child.WritePoll(Poll{RuleUID: "rule1", Found: true, GrafanaNow: testNow.Add(time.Minute)})) + require.NoError(t, child.Stop()) after, err := os.ReadFile(path) - if err != nil { - t.Fatalf("read: %v", err) - } - if !strings.HasPrefix(string(after), string(before)) { - t.Fatalf("reopening the log rewrote earlier records:\n%s", after) - } + require.NoError(t, err) + require.True(t, strings.HasPrefix(string(after), string(before)), "reopening the log rewrote earlier records") _, polls, sentinel, err := ReadLog(path) - if err != nil { - t.Fatalf("ReadLog: %v", err) - } - if len(polls) != 2 || sentinel == nil { - t.Errorf("got %d polls, sentinel %v; want 2 polls and a sentinel", len(polls), sentinel) - } + require.NoError(t, err) + require.Len(t, polls, 2) + require.NotNil(t, sentinel) } // Two recorders on one log means one of them is recording a window nobody @@ -570,119 +468,75 @@ func TestWriterSecondWriterFails(t *testing.T) { select { case err := <-done: - if err == nil { - t.Fatalf("a second writer took the lock") - } - if !strings.Contains(err.Error(), "another writer") { - t.Errorf("error does not name the conflict: %v", err) - } + require.Error(t, err, "a second writer took the lock") + require.Contains(t, err.Error(), "another writer") case <-time.After(5 * time.Second): - t.Fatalf("the second NewWriter blocked instead of failing immediately") + require.Fail(t, "the second NewWriter blocked instead of failing immediately") } } func TestWriterHeaderRefusesANonEmptyLog(t *testing.T) { path := filepath.Join(t.TempDir(), "log.jsonl") w, _ := newTestWriter(t, path) - if err := w.WriteHeader(testHeader()); err != nil { - t.Fatalf("WriteHeader: %v", err) - } - if err := w.WriteHeader(testHeader()); err == nil { - t.Fatalf("a second header was accepted") - } - if err := w.Close(); err != nil { - t.Fatalf("Close: %v", err) - } + require.NoError(t, w.WriteHeader(testHeader())) + require.Error(t, w.WriteHeader(testHeader()), "a second header was accepted") + require.NoError(t, w.Close()) reopened, _ := newTestWriter(t, path) defer reopened.Close() - if err := reopened.WriteHeader(testHeader()); err == nil { - t.Fatalf("a header was accepted on a non-empty log") - } + require.Error(t, reopened.WriteHeader(testHeader()), "a header was accepted on a non-empty log") } func TestSentinelStopIsIdempotentAndLast(t *testing.T) { path := filepath.Join(t.TempDir(), "log.jsonl") w, _ := newTestWriter(t, path) - if err := w.WriteHeader(testHeader()); err != nil { - t.Fatalf("WriteHeader: %v", err) - } - if err := w.Stop(); err != nil { - t.Fatalf("Stop: %v", err) - } + require.NoError(t, w.WriteHeader(testHeader())) + require.NoError(t, w.Stop()) // watch reaches Stop from both a signal handler and a defer; a second // sentinel would be indistinguishable from a second writer. - if err := w.Stop(); err != nil { - t.Errorf("second Stop: %v", err) - } + require.NoError(t, w.Stop()) // Nothing may be appended after the sentinel — not even by the same writer. - if err := w.WritePoll(Poll{RuleUID: "rule1"}); err == nil { - t.Errorf("WritePoll after Stop was accepted") - } + require.Error(t, w.WritePoll(Poll{RuleUID: "rule1"}), "WritePoll after Stop was accepted") b, err := os.ReadFile(path) - if err != nil { - t.Fatalf("read: %v", err) - } + require.NoError(t, err) lines := strings.Split(strings.TrimSuffix(string(b), "\n"), "\n") - if len(lines) != 2 { - t.Fatalf("got %d lines, want header + one sentinel:\n%s", len(lines), b) - } - if !strings.Contains(lines[1], `"type":"stopped"`) { - t.Errorf("last line is not the sentinel: %s", lines[1]) - } + require.Len(t, lines, 2, "header + one sentinel") + require.Contains(t, lines[1], `"type":"stopped"`) } -// Close is the parent's handoff path in P6: a sentinel there would tell check -// the recording ended before the child had even started. +// Close is the parent's handoff path: a sentinel there would tell check the +// recording ended before the child had even started. func TestSentinelCloseWritesNone(t *testing.T) { path := filepath.Join(t.TempDir(), "log.jsonl") w, _ := newTestWriter(t, path) - if err := w.WriteHeader(testHeader()); err != nil { - t.Fatalf("WriteHeader: %v", err) - } - if err := w.Close(); err != nil { - t.Fatalf("Close: %v", err) - } + require.NoError(t, w.WriteHeader(testHeader())) + require.NoError(t, w.Close()) _, polls, sentinel, err := ReadLog(path) - if err != nil { - t.Fatalf("ReadLog: %v", err) - } - if sentinel != nil { - t.Errorf("Close wrote a sentinel: %s", sentinel) - } - if polls != nil { - t.Errorf("polls = %+v, want none", polls) - } + require.NoError(t, err) + require.Nil(t, sentinel, "Close wrote a sentinel") + require.Nil(t, polls) } // An unfinished recording reads cleanly with a nil sentinel — ReadLog reports -// the absence and P7 turns it into unobservable. It is never ReadLog's job to -// call that a failure, and never anyone's job to call it a pass. +// the absence and the coverage proof turns it into unobservable. It is never +// ReadLog's job to call that a failure, and never anyone's job to call it a +// pass. func TestReadLogWithoutASentinel(t *testing.T) { path := filepath.Join(t.TempDir(), "log.jsonl") w, _ := newTestWriter(t, path) - if err := w.WriteHeader(testHeader()); err != nil { - t.Fatalf("WriteHeader: %v", err) - } - if err := w.WritePoll(Poll{RuleUID: "rule1", Found: true, GrafanaNow: testNow}); err != nil { - t.Fatalf("WritePoll: %v", err) - } - if err := w.Close(); err != nil { - t.Fatalf("Close: %v", err) - } + require.NoError(t, w.WriteHeader(testHeader())) + require.NoError(t, w.WritePoll(Poll{RuleUID: "rule1", Found: true, GrafanaNow: testNow})) + require.NoError(t, w.Close()) _, polls, sentinel, err := ReadLog(path) - if err != nil { - t.Fatalf("ReadLog: %v", err) - } - if sentinel != nil || len(polls) != 1 { - t.Errorf("got %d polls, sentinel %v; want 1 poll and no sentinel", len(polls), sentinel) - } + require.NoError(t, err) + require.Nil(t, sentinel) + require.Len(t, polls, 1) } -// The read rules are deliberately the crudest possible (§24.2): any unparseable +// The read rules are deliberately the crudest possible: any unparseable // line is an error, full stop — including the last one, and including a last // line that follows a sentinel. func TestReadLogRejectsBadLogs(t *testing.T) { @@ -690,23 +544,17 @@ func TestReadLogRejectsBadLogs(t *testing.T) { h := testHeader() h.SchemaVersion = version b, err := json.Marshal(headerRecord{Type: RecordHeader, Header: h}) - if err != nil { - t.Fatalf("marshal header: %v", err) - } + require.NoError(t, err, "marshal header") return string(b) } poll := func() string { b, err := json.Marshal(pollRecord{Type: RecordPoll, Poll: Poll{RuleUID: "rule1", Found: true, GrafanaNow: testNow}}) - if err != nil { - t.Fatalf("marshal poll: %v", err) - } + require.NoError(t, err, "marshal poll") return string(b) } sentinel := func() string { b, err := json.Marshal(stoppedRecord{Type: RecordStopped, At: testNow}) - if err != nil { - t.Fatalf("marshal sentinel: %v", err) - } + require.NoError(t, err, "marshal sentinel") return string(b) } @@ -757,138 +605,90 @@ func TestReadLogRejectsBadLogs(t *testing.T) { for _, c := range cases { t.Run(c.name, func(t *testing.T) { path := filepath.Join(t.TempDir(), "log.jsonl") - if err := os.WriteFile(path, []byte(c.content), 0o600); err != nil { - t.Fatalf("write: %v", err) - } + require.NoError(t, os.WriteFile(path, []byte(c.content), 0o600)) _, _, _, err := ReadLog(path) - if err == nil { - t.Fatalf("ReadLog: want an error, got nil") - } - if !strings.Contains(err.Error(), c.wantIn) { - t.Errorf("error %q does not contain %q", err, c.wantIn) - } + require.Error(t, err) + require.Contains(t, err.Error(), c.wantIn) }) } } func TestReadLogMissingFile(t *testing.T) { _, _, _, err := ReadLog(filepath.Join(t.TempDir(), "absent.jsonl")) - if err == nil { - t.Fatalf("ReadLog on a missing log: want an error, got nil") - } + require.Error(t, err) } -// §22.3: per-poll log size must not grow across polls on a high-cardinality +// Per-poll log size must not grow across polls on a high-cardinality // rule, and the one firing instance among 2446 must still be attributed by its // labels. The reduction makes size independent of NORMAL cardinality — the // firing instances are still stored, which is why a clear shrinks the record. func TestLogSizeIsFlatAcrossPollsOnAHighCardinalityRule(t *testing.T) { body := synthesizeHighCardinalityState(t, 1, 2445) rules, err := ParseState(body) - if err != nil { - t.Fatalf("ParseState: %v", err) - } - if len(rules[0].Instances) != 2446 { - t.Fatalf("got %d instances, want 2446", len(rules[0].Instances)) - } + require.NoError(t, err) + require.Len(t, rules[0].Instances, 2446) uid := rules[0].UID path := filepath.Join(t.TempDir(), "log.jsonl") w, _ := newTestWriter(t, path) - if err := w.WriteHeader(testHeader()); err != nil { - t.Fatalf("WriteHeader: %v", err) - } + require.NoError(t, w.WriteHeader(testHeader())) r := NewReducer() var sizes []int64 // Measure from the end of the header line, so sizes[0] is the first poll // record alone rather than the header plus it. info, err := os.Stat(path) - if err != nil { - t.Fatalf("stat: %v", err) - } + require.NoError(t, err) previous := info.Size() for i := range 5 { p := r.Reduce(uid, observation(testNow.Add(time.Duration(i)*30*time.Second), rules[0])) - if len(p.Abnormal) != 1 { - t.Fatalf("poll %d: Abnormal = %d instances, want the single firing one", i, len(p.Abnormal)) - } - if got := p.Abnormal[0].Labels["instance"]; got != "alerting-0" { - t.Fatalf("poll %d: the firing instance lost its identity: %q", i, got) - } - if err := w.WritePoll(p); err != nil { - t.Fatalf("WritePoll: %v", err) - } + require.Lenf(t, p.Abnormal, 1, "poll %d", i) + require.Equalf(t, "alerting-0", p.Abnormal[0].Labels["instance"], "poll %d: the firing instance lost its identity", i) + require.NoError(t, w.WritePoll(p)) info, err := os.Stat(path) - if err != nil { - t.Fatalf("stat: %v", err) - } + require.NoError(t, err) sizes = append(sizes, info.Size()-previous) previous = info.Size() } for i := 1; i < len(sizes); i++ { - if sizes[i] != sizes[0] { - t.Errorf("per-poll size grew across polls: %v", sizes) - } + require.Equal(t, sizes[0], sizes[i], "per-poll size grew across polls: %v", sizes) } // One firing instance among 2446 costs a few hundred bytes, against the // ~600 KB the unreduced response carries. - if sizes[0] > 2048 { - t.Errorf("per-poll size %d bytes is not a reduction of a %d-byte response", sizes[0], len(body)) - } + require.LessOrEqual(t, sizes[0], int64(2048)) // When the firing instance clears, the record collapses further and the // transition is still attributed. rules[0].Instances[0].State = StateNormal p := r.Reduce(uid, observation(testNow.Add(5*30*time.Second), rules[0])) - if len(p.Cleared) != 1 || len(p.Abnormal) != 0 { - t.Errorf("cleared poll = %+v, want exactly one cleared key and no abnormal instances", p) - } - if err := w.Stop(); err != nil { - t.Fatalf("Stop: %v", err) - } + require.Len(t, p.Cleared, 1) + require.Empty(t, p.Abnormal) + require.NoError(t, w.Stop()) } // The log must stay readable by anything that reads JSONL, one flat object per -// line with its type tag — an uploaded artifact (§21.3) is read by humans and -// by jq, not only by ReadLog. +// line with its type tag — an uploaded artifact is read by humans and by jq, +// not only by ReadLog. func TestLogRecordsAreFlatOneLineObjects(t *testing.T) { path := filepath.Join(t.TempDir(), "log.jsonl") w, _ := newTestWriter(t, path) - if err := w.WriteHeader(testHeader()); err != nil { - t.Fatalf("WriteHeader: %v", err) - } - if err := w.WritePoll(Poll{RuleUID: "rule1", Found: true, GrafanaNow: testNow}); err != nil { - t.Fatalf("WritePoll: %v", err) - } - if err := w.Stop(); err != nil { - t.Fatalf("Stop: %v", err) - } + require.NoError(t, w.WriteHeader(testHeader())) + require.NoError(t, w.WritePoll(Poll{RuleUID: "rule1", Found: true, GrafanaNow: testNow})) + require.NoError(t, w.Stop()) b, err := os.ReadFile(path) - if err != nil { - t.Fatalf("read: %v", err) - } + require.NoError(t, err) lines := strings.Split(strings.TrimSuffix(string(b), "\n"), "\n") wantTypes := []RecordType{RecordHeader, RecordPoll, RecordStopped} - if len(lines) != len(wantTypes) { - t.Fatalf("got %d lines, want %d:\n%s", len(lines), len(wantTypes), b) - } + require.Len(t, lines, len(wantTypes)) for i, line := range lines { var m map[string]json.RawMessage - if err := json.Unmarshal([]byte(line), &m); err != nil { - t.Fatalf("line %d is not one JSON object: %v", i+1, err) - } + require.NoErrorf(t, json.Unmarshal([]byte(line), &m), "line %d is not one JSON object", i+1) var gotType RecordType - if err := json.Unmarshal(m["type"], &gotType); err != nil { - t.Fatalf("line %d has no type tag: %v", i+1, err) - } - if gotType != wantTypes[i] { - t.Errorf("line %d type = %q, want %q", i+1, gotType, wantTypes[i]) - } - if _, nested := m["header"]; nested { - t.Errorf("line %d wraps its payload instead of being flat: %s", i+1, line) - } + require.NoErrorf(t, json.Unmarshal(m["type"], &gotType), "line %d has no type tag", i+1) + require.Equalf(t, wantTypes[i], gotType, "line %d type", i+1) + _, nested := m["header"] + require.Falsef(t, nested, "line %d wraps its payload instead of being flat", i+1) } } diff --git a/grafana-alertcheck/internal/gate/parse_ruler.go b/grafana-alertcheck/internal/gate/parse_ruler.go index c27e5b9c4..5517fcce2 100644 --- a/grafana-alertcheck/internal/gate/parse_ruler.go +++ b/grafana-alertcheck/internal/gate/parse_ruler.go @@ -7,9 +7,9 @@ import ( "time" ) -// RuleKind classifies a ruler-endpoint rule by shape, not by name (P1.3). -// P3 rejects KindDatasourceManaged and KindRecording, but only for rules a -// user actually named — ParseDefinitions itself never rejects. +// RuleKind classifies a ruler-endpoint rule by shape, not by name. Resolve +// rejects KindDatasourceManaged and KindRecording, but only for rules a user +// actually named — ParseDefinitions itself never rejects. type RuleKind int const ( @@ -22,8 +22,8 @@ const ( // (/api/ruler/grafana/api/v1/rules). IntervalSeconds, NoDataState and // ExecErrState live inside the grafana_alert block and are only populated for // KindGrafanaManaged — a datasource-managed rule has no such block by -// definition (§11.6 drops relativeTimeRange/keep_firing_for entirely; neither -// is parsed here). +// definition. relativeTimeRange and keep_firing_for are deliberately not +// parsed: nothing in the gate reads them. type Definition struct { UID, Title, Folder, FolderUID, Group string For time.Duration @@ -43,8 +43,8 @@ func ParseDefinitions(body []byte) ([]Definition, error) { } // Map iteration order is nondeterministic; sort namespace names so - // ParseDefinitions' output order is stable across calls (P3's candidate - // listings and any golden test depend on that). + // ParseDefinitions' output order is stable across calls — Resolve's + // candidate listings and the golden tests depend on that. names := make([]string, 0, len(namespaces)) for name := range namespaces { names = append(names, name) @@ -84,7 +84,7 @@ func ParseDefinitions(body []byte) ([]Definition, error) { func parseDefinition(raw json.RawMessage, folder, group string) (Definition, error) { var m map[string]json.RawMessage if err := json.Unmarshal(raw, &m); err != nil { - return Definition{}, fmt.Errorf("%w", err) + return Definition{}, err } var forStr string @@ -137,11 +137,11 @@ func parseDefinition(raw json.RawMessage, folder, group string) (Definition, err // Classify by the presence of "record" before requiring anything else. // no_data_state/exec_err_state/is_paused/intervalSeconds are alerting-only // concepts a recording rule may not carry at all — its real shape is - // unverified (none exist in the fleet capture) — and P3 refuses this - // Kind categorically before any of this would gate a release. Strict- - // parsing a recording rule into a hard error over fields it was never - // going to use would brick `list` and every resolve for rules nobody - // named (§11.6, "do not reject here"). + // unverified, none exist in the fleet capture — and Resolve refuses this + // Kind categorically before any of this would gate a release. + // Strict-parsing a recording rule into a hard error over fields it was + // never going to use would brick `list` and every resolve for rules nobody + // named. var record json.RawMessage if err := opt(ga, "record", &record); err != nil { return Definition{}, fmt.Errorf("rule %q: grafana_alert: %w", uid, err) diff --git a/grafana-alertcheck/internal/gate/parse_ruler_test.go b/grafana-alertcheck/internal/gate/parse_ruler_test.go index 3cf2e3261..280dc9910 100644 --- a/grafana-alertcheck/internal/gate/parse_ruler_test.go +++ b/grafana-alertcheck/internal/gate/parse_ruler_test.go @@ -3,122 +3,88 @@ package gate import ( "testing" "time" + + "github.com/stretchr/testify/require" ) func TestParseDefinitions_RulerRules(t *testing.T) { defs, err := ParseDefinitions(readFixture(t, "ruler_rules.json")) - if err != nil { - t.Fatalf("ParseDefinitions: unexpected error: %v", err) - } + require.NoError(t, err) byUID := map[string]Definition{} for _, d := range defs { - if d.Kind != KindGrafanaManaged { - t.Errorf("rule %q: Kind = %v, want KindGrafanaManaged", d.UID, d.Kind) - } + require.Equalf(t, KindGrafanaManaged, d.Kind, "rule %q: Kind", d.UID) byUID[d.UID] = d } // The real 2-way duplicate title: same folder, same group, same title, - // distinct UIDs (§17, §22.2). + // distinct UIDs — only uid: can tell them apart. a, ok := byUID["rule0000006a"] - if !ok { - t.Fatalf("missing rule0000006a") - } + require.True(t, ok, "missing rule0000006a") b, ok := byUID["rule0000006b"] - if !ok { - t.Fatalf("missing rule0000006b") - } - if a.Title != b.Title || a.Folder != b.Folder || a.Group != b.Group { - t.Errorf("duplicate-title pair should share Title/Folder/Group: a=%+v b=%+v", a, b) - } - if a.UID == b.UID { - t.Errorf("duplicate-title pair should have distinct UIDs") - } + require.True(t, ok, "missing rule0000006b") + require.Equal(t, a.Title, b.Title, "duplicate-title pair should share Title") + require.Equal(t, a.Folder, b.Folder, "duplicate-title pair should share Folder") + require.Equal(t, a.Group, b.Group, "duplicate-title pair should share Group") + require.NotEqual(t, a.UID, b.UID, "duplicate-title pair should have distinct UIDs") // The 3 real paused rules. pausedUIDs := []string{"rule0000002", "rule0000007", "rule0000008"} for _, uid := range pausedUIDs { d, ok := byUID[uid] - if !ok { - t.Fatalf("missing paused rule %q", uid) - } - if !d.IsPaused { - t.Errorf("rule %q: IsPaused = false, want true", uid) - } + require.True(t, ok, "missing paused rule %q", uid) + require.Truef(t, d.IsPaused, "rule %q: IsPaused = false, want true", uid) } // for:1d and the derived for:1w rule. dayRule, ok := byUID["rule0000009"] - if !ok || dayRule.For != 24*time.Hour { - t.Fatalf("rule0000009: For = %v, want 24h (ok=%v)", dayRule.For, ok) - } + require.Truef(t, ok, "missing rule0000009") + require.Equal(t, 24*time.Hour, dayRule.For) weekRule, ok := byUID["rule0000010"] - if !ok || weekRule.For != 7*24*time.Hour { - t.Fatalf("rule0000010: For = %v, want 168h (ok=%v)", weekRule.For, ok) - } + require.Truef(t, ok, "missing rule0000010") + require.Equal(t, 7*24*time.Hour, weekRule.For) // Identity shared with testdata/state_paused.json. shared := byUID["rule0000002"] - if shared.FolderUID != "folder0000002" { - t.Errorf("rule0000002: FolderUID = %q, want folder0000002", shared.FolderUID) - } + require.Equal(t, "folder0000002", shared.FolderUID) } func TestParseDefinitions_DatasourceManaged(t *testing.T) { defs, err := ParseDefinitions(readFixture(t, "ruler_datasource_managed.json")) - if err != nil { - t.Fatalf("ParseDefinitions: unexpected error: %v", err) - } - if len(defs) != 1 { - t.Fatalf("got %d definitions, want 1", len(defs)) - } - if defs[0].Kind != KindDatasourceManaged { - t.Errorf("Kind = %v, want KindDatasourceManaged", defs[0].Kind) - } - if defs[0].For != 5*time.Minute { - t.Errorf("For = %v, want 5m", defs[0].For) - } + require.NoError(t, err) + require.Len(t, defs, 1) + require.Equal(t, KindDatasourceManaged, defs[0].Kind) + require.Equal(t, 5*time.Minute, defs[0].For) // A datasource-managed rule has no uid in this shape; its only identity // is the Prometheus "alert" name — a synthetic UID would be invented - // shape, and an empty Title would make P3's refusal-by-name unreachable. - if defs[0].Title != "ExampleTargetDown" { - t.Errorf("Title = %q, want ExampleTargetDown", defs[0].Title) - } - if defs[0].UID != "" { - t.Errorf("UID = %q, want empty (this shape has no uid)", defs[0].UID) - } + // shape, and an empty Title would make Resolve's refusal-by-name + // unreachable. + require.Equal(t, "ExampleTargetDown", defs[0].Title) + require.Empty(t, defs[0].UID, "this shape has no uid") } func TestParseDefinitions_Recording(t *testing.T) { defs, err := ParseDefinitions(readFixture(t, "ruler_recording.json")) - if err != nil { - t.Fatalf("ParseDefinitions: unexpected error: %v", err) - } - if len(defs) != 1 { - t.Fatalf("got %d definitions, want 1", len(defs)) - } + require.NoError(t, err) + require.Len(t, defs, 1) d := defs[0] - if d.Kind != KindRecording { - t.Errorf("Kind = %v, want KindRecording", d.Kind) - } - if d.UID != "rule0000011" { - t.Errorf("UID = %q, want rule0000011", d.UID) - } + require.Equal(t, KindRecording, d.Kind) + require.Equal(t, "rule0000011", d.UID) // The fixture deliberately omits no_data_state/exec_err_state/is_paused/ // intervalSeconds/namespace_uid — alerting-only concepts a recording // rule may not carry. Requiring them would brick ParseDefinitions for // every named rule in the same response over one recording rule // elsewhere in the fleet; they must come back as zero values, not errors. - if d.NoDataState != "" || d.ExecErrState != "" || d.IsPaused || d.IntervalSeconds != 0 || d.FolderUID != "" { - t.Errorf("expected zero-valued alert-only fields for a recording rule, got %+v", d) - } + require.Empty(t, d.NoDataState) + require.Empty(t, d.ExecErrState) + require.False(t, d.IsPaused) + require.Zero(t, d.IntervalSeconds) + require.Empty(t, d.FolderUID) } // A datasource-managed rule with no alert/record name must fail parsing. func TestParseDefinitions_DatasourceManagedNoName(t *testing.T) { body := []byte(`{"ExampleMetrics":[{"name":"g","rules":[{"expr":"up == 0","for":"5m"}]}]}`) - if _, err := ParseDefinitions(body); err == nil { - t.Fatalf("ParseDefinitions: expected error for datasource-managed rule with no alert/record, got nil") - } + _, err := ParseDefinitions(body) + require.Error(t, err, "a datasource-managed rule with no alert/record must fail") } diff --git a/grafana-alertcheck/internal/gate/parse_state.go b/grafana-alertcheck/internal/gate/parse_state.go index 7ffff1cfc..7c35c7e7d 100644 --- a/grafana-alertcheck/internal/gate/parse_state.go +++ b/grafana-alertcheck/internal/gate/parse_state.go @@ -7,7 +7,7 @@ import ( "time" ) -// State is the canonical instance state (P1.2a). It is distinct from the raw, +// State is the canonical instance state. It is distinct from the raw, // unnormalized vocabularies the API uses at the rule level and at the instance // level — see normalizeInstanceState. type State string @@ -22,12 +22,12 @@ const ( // Instance is one entry of a rule's alerts[]. State is always canonical; Reason // is the opaque suffix of a "State (Reason)" composite ("" when the API gave a -// bare state). Reason is reporting-only except for the H2 MissingSeries routing +// bare state). Reason is reporting-only except for the MissingSeries routing // done downstream in the log markers. // -// The json tags are for the JSONL log's abnormal-instance list (P5) only — -// parsing an API response never goes through them, because parseInstance -// decodes field by field through req/opt to keep H1's presence checks explicit. +// The json tags are for the JSONL log's abnormal-instance list only — parsing +// an API response never goes through them, because parseInstance decodes field +// by field through req/opt to keep the presence checks explicit. type Instance struct { Labels map[string]string `json:"labels"` State State `json:"state"` @@ -37,12 +37,12 @@ type Instance struct { } // StateRule is one rule from the state endpoint -// (/api/prometheus/grafana/api/v1/rules), fully and strictly parsed (H1). +// (/api/prometheus/grafana/api/v1/rules), fully and strictly parsed. type StateRule struct { UID, Title, Folder, Group string Interval time.Duration // State and Health are raw, lowercase, and reporting-only — never - // classified (P1.2a). State in particular is never normalized. + // classified. State in particular is never normalized. State, Health string LastError string LastEvaluation time.Time @@ -53,7 +53,7 @@ type StateRule struct { // ParseState strictly parses a state-endpoint response body into its rules. // A missing or unparseable required field (health, state, lastEvaluation on -// each rule; interval on each group) is an error, never a zero value (H1). +// each rule; interval on each group) is an error, never a zero value. func ParseState(body []byte) ([]StateRule, error) { var top map[string]json.RawMessage if err := json.Unmarshal(body, &top); err != nil { @@ -133,11 +133,10 @@ func parseStateRule(raw json.RawMessage, folder, group string, interval time.Dur if err := req(m, "health", &r.Health); err != nil { return StateRule{}, fmt.Errorf("rule %q: %w", uid, err) } - // isPaused is not one of H1's four named required fields, but this parser - // extends that contract to it: the zero-time rule below can't tell a - // paused rule from a broken one without it, and it's the primary - // in-window pause detector (H2/§12.2) — a silent false default would be - // exactly the fail-open bug H1 exists to kill. + // isPaused is required rather than optional: the zero-time rule below + // can't tell a paused rule from a broken one without it, and it's the + // primary in-window pause detector — a silent false default would be + // exactly the fail-open this parser's strictness exists to kill. if err := req(m, "isPaused", &r.IsPaused); err != nil { return StateRule{}, fmt.Errorf("rule %q: %w", uid, err) } @@ -150,7 +149,7 @@ func parseStateRule(raw json.RawMessage, folder, group string, interval time.Dur if err != nil { return StateRule{}, fmt.Errorf("rule %q: lastEvaluation: %w", uid, err) } - // The zero-time rule (§2.3): only a paused rule may report the zero time. + // Only a paused rule may report the zero time. if lastEval.IsZero() && !r.IsPaused { return StateRule{}, fmt.Errorf("rule %q: lastEvaluation is the zero time but isPaused is false", uid) } @@ -198,10 +197,10 @@ func parseInstance(raw json.RawMessage) (Instance, error) { return Instance{}, err } - // activeAt is also not in H1's named list, extended here for the same - // reason as StateRule.IsPaused: it's the onset time BadFor (P8) measures - // from, so a silently zeroed one would misclassify how long an instance - // has been bad rather than failing loudly. + // activeAt is required for the same reason as StateRule.IsPaused: it's the + // onset time BadFor measures from, so a silently zeroed one would + // misclassify how long an instance has been bad rather than failing + // loudly. var activeAtStr string if err := req(m, "activeAt", &activeAtStr); err != nil { return Instance{}, err @@ -224,8 +223,8 @@ func parseInstance(raw json.RawMessage) (Instance, error) { } // baseInstanceStates is the strict 5-value allowlist for the base of an -// instance state (P1.2a). Anything else — including an unrecognized base -// inside a "Base (Reason)" composite — is a parse error (H1, §2.7 control 3). +// instance state. Anything else — including an unrecognized base inside a +// "Base (Reason)" composite — is a parse error. var baseInstanceStates = map[string]State{ "Normal": StateNormal, "Alerting": StateFiring, diff --git a/grafana-alertcheck/internal/gate/parse_state_test.go b/grafana-alertcheck/internal/gate/parse_state_test.go index 8fca6e7ab..f55c10185 100644 --- a/grafana-alertcheck/internal/gate/parse_state_test.go +++ b/grafana-alertcheck/internal/gate/parse_state_test.go @@ -1,23 +1,20 @@ package gate import ( - "bytes" "encoding/json" "fmt" - "maps" "os" "path/filepath" - "strings" "testing" "time" + + "github.com/stretchr/testify/require" ) func readFixture(t *testing.T, name string) []byte { t.Helper() b, err := os.ReadFile(filepath.Join("testdata", name)) - if err != nil { - t.Fatalf("reading fixture %s: %v", name, err) - } + require.NoErrorf(t, err, "reading fixture %s", name) return b } @@ -33,24 +30,15 @@ func TestParseState_HappyPaths(t *testing.T) { wantRules: 1, wantInstances: 1, checkFirst: func(t *testing.T, r StateRule) { - if r.UID != "rule0000001" { - t.Errorf("UID = %q, want rule0000001", r.UID) - } - if r.Folder != "ExampleTeam" || r.Group != "Example Service - Prod" { - t.Errorf("Folder/Group = %q/%q, want ExampleTeam/Example Service - Prod", r.Folder, r.Group) - } - if r.Health != "ok" || r.State != "inactive" { - t.Errorf("Health/State = %q/%q, want ok/inactive", r.Health, r.State) - } - if r.Interval.Seconds() != 60 { - t.Errorf("Interval = %v, want 60s", r.Interval) - } - if r.IsPaused { - t.Errorf("IsPaused = true, want false") - } - if len(r.Instances) != 1 || r.Instances[0].State != StateNormal { - t.Fatalf("Instances = %+v, want one normal instance", r.Instances) - } + require.Equal(t, "rule0000001", r.UID) + require.Equal(t, "ExampleTeam", r.Folder) + require.Equal(t, "Example Service - Prod", r.Group) + require.Equal(t, "ok", r.Health) + require.Equal(t, "inactive", r.State) + require.Equal(t, float64(60), r.Interval.Seconds()) + require.False(t, r.IsPaused) + require.Len(t, r.Instances, 1) + require.Equal(t, StateNormal, r.Instances[0].State) inst := r.Instances[0] wantLabels := map[string]string{ @@ -62,19 +50,11 @@ func TestParseState_HappyPaths(t *testing.T) { "severity": "critical", "team": "example-team", } - if !maps.Equal(inst.Labels, wantLabels) { - t.Errorf("Labels = %+v, want %+v", inst.Labels, wantLabels) - } + require.Equal(t, wantLabels, inst.Labels) wantActiveAt, err := time.Parse(time.RFC3339, "2026-08-31T08:02:50Z") - if err != nil { - t.Fatalf("test setup: %v", err) - } - if !inst.ActiveAt.Equal(wantActiveAt) { - t.Errorf("ActiveAt = %v, want %v", inst.ActiveAt, wantActiveAt) - } - if inst.Value != "" { - t.Errorf("Value = %q, want empty string", inst.Value) - } + require.NoError(t, err, "test setup") + require.True(t, inst.ActiveAt.Equal(wantActiveAt)) + require.Empty(t, inst.Value) }, }, { @@ -82,15 +62,10 @@ func TestParseState_HappyPaths(t *testing.T) { wantRules: 1, wantInstances: 0, checkFirst: func(t *testing.T, r StateRule) { - if !r.IsPaused { - t.Errorf("IsPaused = false, want true") - } - if !r.LastEvaluation.IsZero() { - t.Errorf("LastEvaluation = %v, want zero time", r.LastEvaluation) - } - if r.Health != "ok" || r.State != "inactive" { - t.Errorf("Health/State = %q/%q, want ok/inactive", r.Health, r.State) - } + require.True(t, r.IsPaused) + require.True(t, r.LastEvaluation.IsZero()) + require.Equal(t, "ok", r.Health) + require.Equal(t, "inactive", r.State) }, }, { @@ -98,15 +73,10 @@ func TestParseState_HappyPaths(t *testing.T) { wantRules: 1, wantInstances: 1, checkFirst: func(t *testing.T, r StateRule) { - if r.Health != "error" { - t.Errorf("Health = %q, want error", r.Health) - } - if r.LastError == "" { - t.Errorf("LastError is empty, want a message") - } - if len(r.Instances) != 1 || r.Instances[0].State != StateError { - t.Fatalf("Instances = %+v, want one error instance", r.Instances) - } + require.Equal(t, "error", r.Health) + require.NotEmpty(t, r.LastError) + require.Len(t, r.Instances, 1) + require.Equal(t, StateError, r.Instances[0].State) }, }, { @@ -114,12 +84,9 @@ func TestParseState_HappyPaths(t *testing.T) { wantRules: 1, wantInstances: 1, checkFirst: func(t *testing.T, r StateRule) { - if r.Health != "nodata" { - t.Errorf("Health = %q, want nodata", r.Health) - } - if len(r.Instances) != 1 || r.Instances[0].State != StateNodata { - t.Fatalf("Instances = %+v, want one nodata instance", r.Instances) - } + require.Equal(t, "nodata", r.Health) + require.Len(t, r.Instances, 1) + require.Equal(t, StateNodata, r.Instances[0].State) }, }, { @@ -132,17 +99,14 @@ func TestParseState_HappyPaths(t *testing.T) { byReason[inst.Reason] = inst } errInst, ok := byReason["Error"] - if !ok || errInst.State != StateNormal { - t.Errorf(`want an instance with State=normal Reason="Error", got %+v`, byReason["Error"]) - } + require.True(t, ok, `want an instance with Reason="Error"`) + require.Equal(t, StateNormal, errInst.State) nodataInst, ok := byReason["NoData"] - if !ok || nodataInst.State != StateNormal { - t.Errorf(`want an instance with State=normal Reason="NoData", got %+v`, byReason["NoData"]) - } + require.True(t, ok, `want an instance with Reason="NoData"`) + require.Equal(t, StateNormal, nodataInst.State) plain, ok := byReason[""] - if !ok || plain.State != StateNormal { - t.Errorf(`want a plain State=normal Reason="" instance, got %+v`, byReason[""]) - } + require.True(t, ok, `want a plain Reason="" instance`) + require.Equal(t, StateNormal, plain.State) }, }, { @@ -150,12 +114,8 @@ func TestParseState_HappyPaths(t *testing.T) { wantRules: 1, wantInstances: 0, checkFirst: func(t *testing.T, r StateRule) { - if r.Instances != nil { - t.Errorf("Instances = %+v, want nil", r.Instances) - } - if r.Totals != nil { - t.Errorf("Totals = %+v, want nil", r.Totals) - } + require.Nil(t, r.Instances) + require.Nil(t, r.Totals) }, }, { @@ -163,12 +123,10 @@ func TestParseState_HappyPaths(t *testing.T) { wantRules: 1, wantInstances: 1, checkFirst: func(t *testing.T, r StateRule) { - if len(r.Instances) != 1 || r.Instances[0].State != StateFiring { - t.Fatalf("Instances = %+v, want one firing instance", r.Instances) - } - if r.Totals["normal"] == 0 { - t.Errorf(`Totals["normal"] = 0, want >0 (this is the §3.2 mismatch the fixture exists to capture)`) - } + require.Len(t, r.Instances, 1) + require.Equal(t, StateFiring, r.Instances[0].State) + require.NotZero(t, r.Totals["normal"], + "the totals/instances mismatch this fixture exists to capture") }, }, } @@ -176,15 +134,9 @@ func TestParseState_HappyPaths(t *testing.T) { for _, c := range cases { t.Run(c.fixture, func(t *testing.T) { rules, err := ParseState(readFixture(t, c.fixture)) - if err != nil { - t.Fatalf("ParseState(%s): unexpected error: %v", c.fixture, err) - } - if len(rules) != c.wantRules { - t.Fatalf("ParseState(%s): got %d rules, want %d", c.fixture, len(rules), c.wantRules) - } - if got := len(rules[0].Instances); got != c.wantInstances { - t.Fatalf("ParseState(%s): got %d instances, want %d", c.fixture, got, c.wantInstances) - } + require.NoErrorf(t, err, "ParseState(%s)", c.fixture) + require.Lenf(t, rules, c.wantRules, "ParseState(%s)", c.fixture) + require.Lenf(t, rules[0].Instances, c.wantInstances, "ParseState(%s)", c.fixture) if c.checkFirst != nil { c.checkFirst(t, rules[0]) } @@ -192,11 +144,11 @@ func TestParseState_HappyPaths(t *testing.T) { } } -// TestParseState_MustError is the H1 regression suite: it doesn't just check -// err != nil (a stray comma in a fixture would keep that green forever while -// the actual check regressed) — it asserts the error names the specific -// offending field or value, so a real H1 check going missing fails loudly -// here instead of surviving unnoticed. +// The strict-parsing regression suite: it doesn't just check err != nil (a +// stray comma in a fixture would keep that green forever while the actual +// check regressed) — it asserts the error names the specific offending field +// or value, so a check going missing fails loudly here instead of surviving +// unnoticed. func TestParseState_MustError(t *testing.T) { cases := []struct { fixture string @@ -214,13 +166,9 @@ func TestParseState_MustError(t *testing.T) { for _, c := range cases { t.Run(c.fixture, func(t *testing.T) { _, err := ParseState(readFixture(t, c.fixture)) - if err == nil { - t.Fatalf("ParseState(%s): expected an error, got none", c.fixture) - } + require.Errorf(t, err, "ParseState(%s): expected an error, got none", c.fixture) for _, want := range c.wantContains { - if !strings.Contains(err.Error(), want) { - t.Errorf("ParseState(%s): error %q does not mention %q", c.fixture, err.Error(), want) - } + require.Containsf(t, err.Error(), want, "ParseState(%s): error", c.fixture) } }) } @@ -248,81 +196,97 @@ func TestParseNormalizeInstanceState(t *testing.T) { for _, c := range cases { state, reason, err := normalizeInstanceState(c.in) if c.wantErr { - if err == nil { - t.Errorf("normalizeInstanceState(%q): expected an error, got none", c.in) - } + require.Errorf(t, err, "normalizeInstanceState(%q)", c.in) continue } - if err != nil { - t.Errorf("normalizeInstanceState(%q): unexpected error: %v", c.in, err) - continue - } - if state != c.wantState || reason != c.wantReason { - t.Errorf("normalizeInstanceState(%q) = (%q, %q), want (%q, %q)", c.in, state, reason, c.wantState, c.wantReason) - } + require.NoErrorf(t, err, "normalizeInstanceState(%q)", c.in) + require.Equalf(t, c.wantState, state, "normalizeInstanceState(%q)", c.in) + require.Equalf(t, c.wantReason, reason, "normalizeInstanceState(%q)", c.in) } } func TestInstanceKey(t *testing.T) { a := instanceKey(map[string]string{"b": "2", "a": "1"}) b := instanceKey(map[string]string{"a": "1", "b": "2"}) - if a != b { - t.Errorf("instanceKey order-independence: %q != %q", a, b) - } - if a != `{"a":"1","b":"2"}` { - t.Errorf("instanceKey = %q, want %q", a, `{"a":"1","b":"2"}`) - } + require.Equal(t, b, a, "instanceKey order-independence") + require.Equal(t, `{"a":"1","b":"2"}`, a) diff := instanceKey(map[string]string{"a": "1", "b": "3"}) - if a == diff { - t.Errorf("instanceKey should differ when a label value differs") - } + require.NotEqual(t, a, diff, "instanceKey should differ when a label value differs") - if instanceKey(nil) != "null" { - t.Errorf("instanceKey(nil) = %q, want \"null\"", instanceKey(nil)) - } + require.Equal(t, "null", instanceKey(nil), "instanceKey(nil) should be \"null\"") } -// Label values may contain "\n" or "="; the JSON encoding must keep them distinct. func TestInstanceKey_NoCollision(t *testing.T) { - if instanceKey(map[string]string{"a": "1\nb=2"}) == instanceKey(map[string]string{"a": "1", "b": "2"}) { - t.Errorf("instanceKey collided for sets {a:1\\nb=2} and {a:1,b:2}") + require.NotEqual(t, instanceKey(map[string]string{"a": "1\nb=2"}), instanceKey(map[string]string{"a": "1", "b": "2"}), "instanceKey should not collide for sets {a:1\\nb=2} and {a:1,b:2}") + + require.NotEqual(t, instanceKey(map[string]string{"a": "1=b"}), instanceKey(map[string]string{"a": "1", "b": ""}), "instanceKey should not collide for sets {a:1=b} and {a:1,b:}") +} + +// minimalStateBody is the smallest legal state response: one group, one +// rule, no optional keys at all, plus whatever extra is spliced in verbatim +// before the rule's closing brace — for isolating one optional key at a time +// rather than relying on a fixture that removes several together. +func minimalStateBody(extraRuleJSON string) []byte { + return fmt.Appendf(nil, + `{"status":"success","data":{"groups":[{"file":"F","name":"G","interval":60,`+ + `"rules":[{"uid":"r1","name":"R1","state":"inactive","health":"ok","isPaused":false,`+ + `"lastEvaluation":"2026-01-01T00:00:00Z"%s}]}]}}`, extraRuleJSON) +} + +// keepFiringFor is optional alongside alerts/totals/labels, but +// state_missing_optional.json removes it together with everything else — never +// in isolation, so a regression that made it required specifically would not +// be caught by that fixture alone. +func TestParseState_KeepFiringForIsOptional(t *testing.T) { + tests := []struct { + name string + extra string + }{ + {"present", `,"keepFiringFor":300`}, + {"absent", ""}, } - if instanceKey(map[string]string{"a": "1=b"}) == instanceKey(map[string]string{"a": "1", "b": ""}) { - t.Errorf("instanceKey collided for a value containing '='") + for _, tc := range tests { + t.Run(tc.name, func(t *testing.T) { + rules, err := ParseState(minimalStateBody(tc.extra)) + require.NoError(t, err) + require.Len(t, rules, 1) + }) } } +// labels is optional at the INSTANCE level (opt(m, "labels", ...) in +// parseInstance), distinct from the rule-level labels state_missing_optional.json +// already covers — an instance can exist with no labels of its own. +func TestParseState_InstanceWithoutLabelsParses(t *testing.T) { + body := minimalStateBody(`,"alerts":[{"state":"Normal","activeAt":"2026-01-01T00:00:00Z"}]`) + rules, err := ParseState(body) + require.NoError(t, err) + require.Len(t, rules, 1) + require.Len(t, rules[0].Instances, 1) + require.Empty(t, rules[0].Instances[0].Labels) +} + // synthesizeHighCardinalityState builds a state response with a single rule // holding `alerting` Alerting instances and `normal` Normal instances, by // cloning the one real instance in state_one_instance.json. It is never -// committed (§3.2, §22.3, §22.6) — the 2446-instance rule this stands in for -// is ~600 KB and exists only to prove the parser and (in later phases) the -// reducer don't choke on real fleet cardinality. +// committed — the 2446-instance rule this stands in for is ~600 KB and exists +// only to prove the parser and the reducer don't choke on real fleet +// cardinality. func synthesizeHighCardinalityState(t *testing.T, alerting, normal int) []byte { t.Helper() base := readFixture(t, "state_one_instance.json") var top map[string]json.RawMessage - if err := json.Unmarshal(base, &top); err != nil { - t.Fatalf("synthesize: %v", err) - } + require.NoError(t, json.Unmarshal(base, &top)) var data map[string]json.RawMessage - if err := json.Unmarshal(top["data"], &data); err != nil { - t.Fatalf("synthesize: %v", err) - } + require.NoError(t, json.Unmarshal(top["data"], &data)) var groups []map[string]json.RawMessage - if err := json.Unmarshal(data["groups"], &groups); err != nil { - t.Fatalf("synthesize: %v", err) - } + require.NoError(t, json.Unmarshal(data["groups"], &groups)) var rules []map[string]json.RawMessage - if err := json.Unmarshal(groups[0]["rules"], &rules); err != nil { - t.Fatalf("synthesize: %v", err) - } + require.NoError(t, json.Unmarshal(groups[0]["rules"], &rules)) var alerts []map[string]json.RawMessage - if err := json.Unmarshal(rules[0]["alerts"], &alerts); err != nil { - t.Fatalf("synthesize: %v", err) - } + require.NoError(t, json.Unmarshal(rules[0]["alerts"], &alerts)) template := alerts[0] newAlerts := make([]map[string]json.RawMessage, 0, alerting+normal) @@ -347,9 +311,7 @@ func synthesizeHighCardinalityState(t *testing.T, alerting, normal int) []byte { top["data"] = mustRaw(t, data) out, err := json.Marshal(top) - if err != nil { - t.Fatalf("synthesize: %v", err) - } + require.NoError(t, err) return out } @@ -366,29 +328,19 @@ func cloneRawMap(m map[string]json.RawMessage) map[string]json.RawMessage { func mustRaw(t *testing.T, v any) json.RawMessage { t.Helper() b, err := json.Marshal(v) - if err != nil { - t.Fatalf("marshal: %v", err) - } + require.NoError(t, err) return json.RawMessage(b) } func TestParseState_HighCardinality(t *testing.T) { body := synthesizeHighCardinalityState(t, 445, 2004) - if !bytes.Contains(body, []byte("alerting-0")) { - t.Fatalf("synthesized body missing expected content") - } + require.Contains(t, string(body), "alerting-0") rules, err := ParseState(body) - if err != nil { - t.Fatalf("ParseState: unexpected error: %v", err) - } - if len(rules) != 1 { - t.Fatalf("got %d rules, want 1", len(rules)) - } + require.NoError(t, err) + require.Len(t, rules, 1) r := rules[0] - if len(r.Instances) != 445+2004 { - t.Fatalf("got %d instances, want %d", len(r.Instances), 445+2004) - } + require.Len(t, r.Instances, 445+2004) var firing, normal int for _, inst := range r.Instances { @@ -398,28 +350,21 @@ func TestParseState_HighCardinality(t *testing.T) { case StateNormal: normal++ default: - t.Fatalf("unexpected instance state %q", inst.State) + require.Fail(t, fmt.Sprintf("unexpected instance state %q", inst.State)) } } - if firing != 445 || normal != 2004 { - t.Fatalf("got firing=%d normal=%d, want firing=445 normal=2004", firing, normal) - } + require.Equal(t, 445, firing) + require.Equal(t, 2004, normal) // Each synthesized instance carries a distinct "instance" label; confirm // Labels actually made it through parsing (not just State) by checking // instanceKey produces one unique key per instance, with no collisions. seen := make(map[string]bool, len(r.Instances)) for _, inst := range r.Instances { - if inst.Labels == nil { - t.Fatalf("instance has nil Labels") - } + require.NotNil(t, inst.Labels) k := instanceKey(inst.Labels) - if seen[k] { - t.Fatalf("duplicate instance key %q", k) - } + require.Falsef(t, seen[k], "duplicate instance key %q", k) seen[k] = true } - if len(seen) != 445+2004 { - t.Fatalf("got %d unique instance keys, want %d", len(seen), 445+2004) - } + require.Len(t, seen, 445+2004) } diff --git a/grafana-alertcheck/internal/gate/resolve.go b/grafana-alertcheck/internal/gate/resolve.go index 3c3710038..329644f95 100644 --- a/grafana-alertcheck/internal/gate/resolve.go +++ b/grafana-alertcheck/internal/gate/resolve.go @@ -7,8 +7,8 @@ import ( "strings" ) -// Resolve turns the operator-supplied alert names into resolved Definitions -// (§17). Order is load-bearing (§17.3): +// Resolve turns the operator-supplied alert names into resolved Definitions. +// Order is load-bearing: // // 1. Trim each name. // 2. Discard empty lines. @@ -18,9 +18,9 @@ import ( // user less than a failure). // // The caller-visible consequence: len(resolved) is the count *after* the -// collapse. A later phase's MinObserved must default from that length, never -// from len(names) — using the input line count would make one rule named -// twice turn an achievable default into an unsatisfiable one (§17.3). +// collapse, and MinObserved must default from that length, never from +// len(names) — using the input line count would make one rule named twice turn +// an achievable default into an unsatisfiable one. func Resolve(defs []Definition, names []string, folder string) (resolved []Definition, notes []string, err error) { seenUID := map[string]string{} // uid -> the first input name that resolved to it for _, raw := range names { @@ -45,15 +45,15 @@ func Resolve(defs []Definition, names []string, folder string) (resolved []Defin return resolved, notes, nil } -// resolveOne resolves a single trimmed, non-empty name against defs (§17.1): -// one match wins outright, zero is an error with suggestions, two or more is -// an error listing every candidate. folder scopes a bare title (no "/" in the -// name) to one folder; it is ignored for the "Folder/Title" and -// "Folder/Group/Title" forms, which already name their own folder. +// resolveOne resolves a single trimmed, non-empty name against defs: one match +// wins outright, zero is an error with suggestions, two or more is an error +// listing every candidate. folder scopes a bare title (no "/" in the name) to +// one folder; it is ignored for the "Folder/Title" and "Folder/Group/Title" +// forms, which already name their own folder. // -// Policy on unsupported kinds (datasource-managed, recording) — decided here -// because §17.1 only says to refuse them, not how they interact with the -// no-match/ambiguous surfaces: a name can still match an unsupported rule (so +// Unsupported kinds (datasource-managed, recording) are refused, and how that +// interacts with the no-match/ambiguous surfaces is decided here: a name can +// still match an unsupported rule (so // naming one by title still gets the specific, named refusal, not a bare "no // match"), but only *supported* candidates count for ambiguity — an // unsupported rule sharing a title with a supported one is resolved silently @@ -73,7 +73,7 @@ func resolveOne(defs []Definition, name, folder string) (Definition, error) { } // uid == "" falls through to the same message as "not found": several // Definition kinds legitimately carry UID == "" (datasource-managed - // rules have no uid at all, P1.3), so matching on an empty suffix + // rules have no uid at all), so matching on an empty suffix // would silently hit one of those and report a misleading // kind-specific refusal for what is really an empty/typo'd uid. This // deliberately does not go through noMatchError: that function's @@ -118,9 +118,9 @@ func resolveOne(defs []Definition, name, folder string) (Definition, error) { } } -// supportedDefs filters out the two kinds §17.1 refuses. Only these -// participate in name-based matching, the no-match rule count, and substring -// suggestions (see the policy note on resolveOne). +// supportedDefs filters out the two refused kinds. Only these participate in +// name-based matching, the no-match rule count, and substring suggestions (see +// the policy note on resolveOne). func supportedDefs(defs []Definition) []Definition { out := make([]Definition, 0, len(defs)) for _, d := range defs { @@ -132,7 +132,7 @@ func supportedDefs(defs []Definition) []Definition { } // classifyForm splits name into the Title | Folder/Title | Folder/Group/Title -// forms (§17). A bare title is scoped by folder when the caller supplied one; +// forms. A bare title is scoped by folder when the caller supplied one; // the two- and three-segment forms already carry their own folder and ignore // it. // @@ -158,8 +158,8 @@ func classifyForm(name, folder string) (wantFolder, wantGroup, wantTitle string, } } -// refuseUnsupportedKind rejects the two kinds §17.1 names explicitly with a -// clear, specific error — distinct from "no match" and from "ambiguous" — so +// refuseUnsupportedKind rejects the two unsupported kinds with a clear, +// specific error — distinct from "no match" and from "ambiguous" — so // an operator who names a recording or datasource-managed rule learns why, // not just that nothing matched. func refuseUnsupportedKind(name string, d Definition) (Definition, error) { @@ -174,8 +174,7 @@ func refuseUnsupportedKind(name string, d Definition) (Definition, error) { } // noMatchError reports a no-match with the count of rules the gate could see -// and, per Context decision 4, case-insensitive substring matches in place of -// the source plan's cut Levenshtein suggestions (§17.2). +// and case-insensitive substring matches as suggestions. func noMatchError(defs []Definition, name, wantTitle string) error { msg := fmt.Sprintf("no rule matched %q (%d rules available; run 'grafana-alertcheck list' to see titles)", name, len(defs)) @@ -194,8 +193,8 @@ func noMatchError(defs []Definition, name, wantTitle string) error { } // ambiguousError lists every candidate with its folder, its group, and the -// full copyable Folder/Group/Title (§17.1) — including the uid: form, which -// resolves unambiguously on the next attempt. +// full copyable Folder/Group/Title — including the uid: form, which resolves +// unambiguously on the next attempt. func ambiguousError(name string, candidates []Definition) error { sorted := append([]Definition(nil), candidates...) sort.Slice(sorted, func(i, j int) bool { return sorted[i].UID < sorted[j].UID }) diff --git a/grafana-alertcheck/internal/gate/resolve_test.go b/grafana-alertcheck/internal/gate/resolve_test.go index 77fd3bb5d..33214516e 100644 --- a/grafana-alertcheck/internal/gate/resolve_test.go +++ b/grafana-alertcheck/internal/gate/resolve_test.go @@ -2,128 +2,89 @@ package gate import ( "fmt" - "strings" "testing" + + "github.com/stretchr/testify/require" ) func rulerDefs(t *testing.T) []Definition { t.Helper() defs, err := ParseDefinitions(readFixture(t, "ruler_rules.json")) - if err != nil { - t.Fatalf("ParseDefinitions: unexpected error: %v", err) - } + require.NoError(t, err) return defs } func TestResolve_SingleMatch(t *testing.T) { defs := rulerDefs(t) resolved, notes, err := Resolve(defs, []string{"example_workflow_paused_rule"}, "") - if err != nil { - t.Fatalf("Resolve: unexpected error: %v", err) - } - if len(notes) != 0 { - t.Errorf("notes = %v, want none", notes) - } - if len(resolved) != 1 || resolved[0].UID != "rule0000007" { - t.Fatalf("resolved = %+v, want [rule0000007]", resolved) - } + require.NoError(t, err) + require.Empty(t, notes) + require.Len(t, resolved, 1) + require.Equal(t, "rule0000007", resolved[0].UID) } func TestResolve_UIDForm(t *testing.T) { defs := rulerDefs(t) resolved, _, err := Resolve(defs, []string{"uid:rule0000006a"}, "") - if err != nil { - t.Fatalf("Resolve: unexpected error: %v", err) - } - if len(resolved) != 1 || resolved[0].UID != "rule0000006a" { - t.Fatalf("resolved = %+v, want [rule0000006a]", resolved) - } + require.NoError(t, err) + require.Len(t, resolved, 1) + require.Equal(t, "rule0000006a", resolved[0].UID) } func TestResolve_FolderTitleForm(t *testing.T) { defs := rulerDefs(t) resolved, _, err := Resolve(defs, []string{"ExampleFeeds/TEMP - Example depeg alert"}, "") - if err != nil { - t.Fatalf("Resolve: unexpected error: %v", err) - } - if len(resolved) != 1 || resolved[0].UID != "rule0000008" { - t.Fatalf("resolved = %+v, want [rule0000008]", resolved) - } + require.NoError(t, err) + require.Len(t, resolved, 1) + require.Equal(t, "rule0000008", resolved[0].UID) } func TestResolve_FolderGroupTitleForm(t *testing.T) { defs := rulerDefs(t) resolved, _, err := Resolve(defs, []string{"Example-Zone-A/Gateway/Example No Gateways Available"}, "") - if err == nil { - t.Fatalf("Resolve: want ambiguous error (real 2-way collision), got resolved=%+v", resolved) - } - if !strings.Contains(err.Error(), "matches 2 rules") { - t.Fatalf("Resolve: error = %q, want it to report 2 matches", err) - } - if !strings.Contains(err.Error(), "uid:rule0000006a") || !strings.Contains(err.Error(), "uid:rule0000006b") { - t.Fatalf("Resolve: error = %q, want both candidate uids listed", err) - } + require.Error(t, err, "want ambiguous error (real 2-way collision), got resolved=%+v", resolved) + require.Contains(t, err.Error(), "matches 2 rules") + require.Contains(t, err.Error(), "uid:rule0000006a") + require.Contains(t, err.Error(), "uid:rule0000006b") } func TestResolve_TrueCollisionResolvesByUID(t *testing.T) { defs := rulerDefs(t) resolved, _, err := Resolve(defs, []string{"uid:rule0000006a", "uid:rule0000006b"}, "") - if err != nil { - t.Fatalf("Resolve: unexpected error: %v", err) - } - if len(resolved) != 2 { - t.Fatalf("resolved = %+v, want 2 distinct rules", resolved) - } + require.NoError(t, err) + require.Len(t, resolved, 2) } func TestResolve_NoMatch(t *testing.T) { defs := rulerDefs(t) _, _, err := Resolve(defs, []string{"Does Not Exist"}, "") - if err == nil { - t.Fatal("Resolve: want error for unknown name") - } - if !strings.Contains(err.Error(), "no rule matched") || !strings.Contains(err.Error(), "list") { - t.Errorf("Resolve: error = %q, want it to name 'no rule matched' and point at 'list'", err) - } + require.Error(t, err) + require.Contains(t, err.Error(), "no rule matched") + require.Contains(t, err.Error(), "list") } func TestResolve_NoMatchSubstringSuggestion(t *testing.T) { defs := rulerDefs(t) _, _, err := Resolve(defs, []string{"paused rule"}, "") - if err == nil { - t.Fatal("Resolve: want error for unknown name") - } - if !strings.Contains(err.Error(), "did you mean") || !strings.Contains(err.Error(), "Example Paused Rule") { - t.Errorf("Resolve: error = %q, want a case-insensitive substring suggestion", err) - } + require.Error(t, err) + require.Contains(t, err.Error(), "did you mean") + require.Contains(t, err.Error(), "Example Paused Rule") } func TestResolve_RefusesDatasourceManaged(t *testing.T) { defs, err := ParseDefinitions(readFixture(t, "ruler_datasource_managed.json")) - if err != nil { - t.Fatalf("ParseDefinitions: unexpected error: %v", err) - } + require.NoError(t, err) _, _, err = Resolve(defs, []string{"ExampleTargetDown"}, "") - if err == nil { - t.Fatal("Resolve: want refusal for a datasource-managed rule") - } - if !strings.Contains(err.Error(), "datasource-managed") { - t.Errorf("Resolve: error = %q, want it to name the datasource-managed kind", err) - } + require.Error(t, err) + require.Contains(t, err.Error(), "datasource-managed") } func TestResolve_RefusesRecording(t *testing.T) { defs, err := ParseDefinitions(readFixture(t, "ruler_recording.json")) - if err != nil { - t.Fatalf("ParseDefinitions: unexpected error: %v", err) - } + require.NoError(t, err) _, _, err = Resolve(defs, []string{"uid:rule0000011"}, "") - if err == nil { - t.Fatal("Resolve: want refusal for a recording rule") - } - if !strings.Contains(err.Error(), "recording rule") { - t.Errorf("Resolve: error = %q, want it to name the recording kind", err) - } + require.Error(t, err) + require.Contains(t, err.Error(), "recording rule") } func TestResolve_RejectsEmptySegments(t *testing.T) { @@ -132,67 +93,41 @@ func TestResolve_RejectsEmptySegments(t *testing.T) { for _, name := range cases { t.Run(name, func(t *testing.T) { _, _, err := Resolve(defs, []string{name}, "") - if err == nil { - t.Fatalf("Resolve(%q): want error for an empty /-separated segment", name) - } - if !strings.Contains(err.Error(), "empty") { - t.Errorf("Resolve(%q): error = %q, want it to name the empty segment", name, err) - } + require.Error(t, err) + require.Contains(t, err.Error(), "empty") }) } } func TestResolve_UIDEmptySuffix(t *testing.T) { - // ruler_datasource_managed.json's only rule has UID == "" (P1.3: this - // shape has no uid at all). "uid:" with an empty suffix must not match it + // ruler_datasource_managed.json's only rule has UID == "" — that shape has + // no uid at all. "uid:" with an empty suffix must not match it // — that would report the misleading "datasource-managed rule, not // supported" for what is really a typo'd/empty uid. defs, err := ParseDefinitions(readFixture(t, "ruler_datasource_managed.json")) - if err != nil { - t.Fatalf("ParseDefinitions: unexpected error: %v", err) - } + require.NoError(t, err) _, _, err = Resolve(defs, []string{"uid:"}, "") - if err == nil { - t.Fatal("Resolve: want error for an empty uid: suffix") - } - if !strings.Contains(err.Error(), "no rule has this uid") { - t.Errorf("Resolve: error = %q, want it to say no rule has this uid", err) - } - if strings.Contains(err.Error(), "datasource-managed") { - t.Errorf("Resolve: error = %q, must not misreport this as a datasource-managed refusal", err) - } + require.Error(t, err) + require.Contains(t, err.Error(), "no rule has this uid") + require.NotContains(t, err.Error(), "datasource-managed") } func TestResolve_UnsupportedKindsExcludedFromNoMatchSurfaces(t *testing.T) { dsDefs, err := ParseDefinitions(readFixture(t, "ruler_datasource_managed.json")) - if err != nil { - t.Fatalf("ParseDefinitions(datasource_managed): unexpected error: %v", err) - } + require.NoError(t, err) recDefs, err := ParseDefinitions(readFixture(t, "ruler_recording.json")) - if err != nil { - t.Fatalf("ParseDefinitions(recording): unexpected error: %v", err) - } + require.NoError(t, err) supported := rulerDefs(t) combined := append(append(append([]Definition{}, supported...), dsDefs...), recDefs...) _, _, err = Resolve(combined, []string{"Example"}, "") - if err == nil { - t.Fatal("Resolve: want a no-match error for a name matching no title exactly") - } + require.Error(t, err, "want a no-match error for a name matching no title exactly") wantCount := fmt.Sprintf("(%d rules available", len(supported)) - if !strings.Contains(err.Error(), wantCount) { - t.Errorf("Resolve: error = %q, want the available count scoped to the %d supported rules, not the %d combined", err, len(supported), len(combined)) - } - if strings.Contains(err.Error(), "ExampleTargetDown") { - t.Errorf("Resolve: error = %q, must not suggest the datasource-managed rule", err) - } - if strings.Contains(err.Error(), "example:recorded_metric:rate5m") { - t.Errorf("Resolve: error = %q, must not suggest the recording rule", err) - } - if !strings.Contains(err.Error(), "Example Paused Rule") { - t.Errorf("Resolve: error = %q, want it to still suggest a matching supported rule", err) - } + require.Contains(t, err.Error(), wantCount) + require.NotContains(t, err.Error(), "ExampleTargetDown") + require.NotContains(t, err.Error(), "example:recorded_metric:rate5m") + require.Contains(t, err.Error(), "Example Paused Rule") } func TestResolve_UnsupportedHomonymResolvesSupportedSilently(t *testing.T) { @@ -205,32 +140,37 @@ func TestResolve_UnsupportedHomonymResolvesSupportedSilently(t *testing.T) { {UID: "", Folder: "F", Group: "G", Title: "Shared Title", Kind: KindDatasourceManaged}, } resolved, _, err := Resolve(defs, []string{"F/G/Shared Title"}, "") - if err != nil { - t.Fatalf("Resolve: unexpected error: %v", err) - } - if len(resolved) != 1 || resolved[0].UID != "supported-1" { - t.Fatalf("resolved = %+v, want the supported rule alone, no ambiguity", resolved) - } + require.NoError(t, err) + require.Len(t, resolved, 1) + require.Equal(t, "supported-1", resolved[0].UID) } func TestResolve_CollapseByUIDGivesNoteNotError(t *testing.T) { defs := rulerDefs(t) // The bare title and its Folder/Group/Title spelling both name the same - // rule (rule0000007) — a duplicate-name copy mistake, not an error - // (§17.3). + // rule (rule0000007) — a duplicate-name copy mistake, not an error. resolved, notes, err := Resolve(defs, []string{ "example_workflow_paused_rule", "ExampleObservability/Example Auth Production/example_workflow_paused_rule", }, "") - if err != nil { - t.Fatalf("Resolve: unexpected error: %v", err) - } - if len(resolved) != 1 || resolved[0].UID != "rule0000007" { - t.Fatalf("resolved = %+v, want exactly one rule0000007", resolved) - } - if len(notes) != 1 { - t.Fatalf("notes = %v, want exactly one collapse note", notes) - } + require.NoError(t, err) + require.Len(t, resolved, 1) + require.Equal(t, "rule0000007", resolved[0].UID) + require.Len(t, notes, 1) +} + +// The same rule named twice with the identical string must collapse to one +// rule — distinct from the different-spellings case above. +func TestResolve_IdenticalDuplicateNameCollapsesWithNote(t *testing.T) { + defs := rulerDefs(t) + resolved, notes, err := Resolve(defs, []string{ + "example_workflow_paused_rule", + "example_workflow_paused_rule", + }, "") + require.NoError(t, err) + require.Len(t, resolved, 1) + require.Equal(t, "rule0000007", resolved[0].UID) + require.Len(t, notes, 1) } func TestResolve_MinObservedCountIsPostCollapse(t *testing.T) { @@ -241,44 +181,30 @@ func TestResolve_MinObservedCountIsPostCollapse(t *testing.T) { "Example Paused Rule", } resolved, notes, err := Resolve(defs, names, "") - if err != nil { - t.Fatalf("Resolve: unexpected error: %v", err) - } - // §17.3: the default MinObserved must come from len(resolved) (2 distinct + require.NoError(t, err) + // The default MinObserved must come from len(resolved) (2 distinct // rules) — never len(names) (3 input lines), which would be unsatisfiable. - if len(resolved) != 2 { - t.Fatalf("resolved = %+v, want 2 distinct rules after collapse", resolved) - } - if len(notes) != 1 { - t.Fatalf("notes = %v, want exactly one collapse note", notes) - } + require.Len(t, resolved, 2) + require.Len(t, notes, 1) } func TestResolve_EmptyAndBlankLinesDiscarded(t *testing.T) { defs := rulerDefs(t) resolved, _, err := Resolve(defs, []string{"", " ", "example_workflow_paused_rule", " \t "}, "") - if err != nil { - t.Fatalf("Resolve: unexpected error: %v", err) - } - if len(resolved) != 1 || resolved[0].UID != "rule0000007" { - t.Fatalf("resolved = %+v, want [rule0000007]", resolved) - } + require.NoError(t, err) + require.Len(t, resolved, 1) + require.Equal(t, "rule0000007", resolved[0].UID) } func TestResolve_FolderScopesBareTitle(t *testing.T) { defs := rulerDefs(t) // Bare title, scoped to the wrong folder — must not match. _, _, err := Resolve(defs, []string{"example_workflow_paused_rule"}, "Example-Zone-A") - if err == nil { - t.Fatal("Resolve: want no-match when folder scope excludes the only candidate") - } + require.Error(t, err, "want no-match when folder scope excludes the only candidate") // Scoped to the right folder — must match. resolved, _, err := Resolve(defs, []string{"example_workflow_paused_rule"}, "ExampleObservability") - if err != nil { - t.Fatalf("Resolve: unexpected error: %v", err) - } - if len(resolved) != 1 || resolved[0].UID != "rule0000007" { - t.Fatalf("resolved = %+v, want [rule0000007]", resolved) - } + require.NoError(t, err) + require.Len(t, resolved, 1) + require.Equal(t, "rule0000007", resolved[0].UID) } diff --git a/grafana-alertcheck/internal/gate/schedule.go b/grafana-alertcheck/internal/gate/schedule.go index ec89dcb13..c2fe6e1c5 100644 --- a/grafana-alertcheck/internal/gate/schedule.go +++ b/grafana-alertcheck/internal/gate/schedule.go @@ -8,23 +8,26 @@ import ( "time" ) -// skewHardLimit is one of §5's filled-in values (basis: §16; §22.11 asserts -// 120s errors, 30s does not). Defined here, in schedule.go's named-constants -// block, per §5's instruction — it moved out of source.go now that P4 exists; -// P2 needed it before this file did, so it started there. -const skewHardLimit = 60 * time.Second +// SkewHardLimit is the largest runner↔Grafana clock skew a run tolerates. +// Exported so the CLI reports it verbatim next to a measured skew. +const SkewHardLimit = 60 * time.Second -// minDrainTimeout is §5's floor on drainTimeout: max(2 x max(intervalSeconds), -// 2m). Without the floor, a fleet of very tight rules would derive a -// drainTimeout too short to let a healthy in-flight poll land. +// fromFutureTolerance is how far ahead of the runner's clock a `from` may sit +// before check refuses it — the same 60s as SkewHardLimit, since a future `from` +// can only be clock disagreement. Once-per-run input validation, not a coverage +// check, so Check applies it and proveCoverage does not. +const fromFutureTolerance = 60 * time.Second + +// minDrainTimeout floors drainTimeout (otherwise 2 × max intervalSeconds) so a +// fleet of tight rules still lets a healthy in-flight poll land. const minDrainTimeout = 2 * time.Minute -// graceWarnFraction is §13.2's threshold for warning that transitionGrace eats -// too much of the requested window: "approximately one quarter of the window". +// graceWarnFraction is the share of the window above which transitionGrace is +// worth warning about. const graceWarnFraction = 0.25 -// ruleTimings groups the per-rule threshold values §5/§10.1/§14.1 derive from -// a rule's poll cadence and its own evaluation interval. +// ruleTimings groups the per-rule thresholds derived from a rule's poll +// cadence and its own evaluation interval. type ruleTimings struct { pollEvery time.Duration maxGap time.Duration @@ -33,26 +36,21 @@ type ruleTimings struct { } // globalTimings groups the values that apply to the whole run rather than to -// one rule: §13.1's transitionGrace and §19's drainTimeout are each derived -// once, across every non-skipped watched rule, not per rule. +// one rule: transitionGrace and drainTimeout are each derived once, across +// every non-skipped watched rule, not per rule. type globalTimings struct { transitionGrace time.Duration - // graceSource names, and already carries the `for` value of, the rule - // that set transitionGrace (§13.2 requires printing both) — one string - // field rather than a second (rule, duration) pair, matching this - // struct's fixed shape. "none" when no rule contributed (transitionGrace - // is then 0). + // graceSource names, and already carries the `for` value of, the rule that + // set transitionGrace — one string field rather than a second + // (rule, duration) pair, matching this struct's fixed shape. "none" when no + // rule contributed (transitionGrace is then 0). graceSource string drainTimeout time.Duration } -// newRuleTimings derives one rule's thresholds from its fully-resolved poll -// cadence and its evaluation interval (§5, §10.1, §14.1). pollEvery arrives -// already resolved for the caller's mode — the §5 default, the operator's -// --poll-interval override, or (in log mode, a later phase) the cadence -// recorded in the log header. Deriving pollEvery inline here, instead of -// accepting it as an input, would let a caller in the wrong mode compute -// maxGap against the wrong authority — see the "Two authorities" note in P5. +// newRuleTimings derives one rule's thresholds from its resolved cadence and +// evaluation interval. pollEvery is an input — already resolved for the +// caller's mode — so no caller can compute maxGap against the wrong authority. func newRuleTimings(pollEvery time.Duration, intervalSeconds int) ruleTimings { interval := time.Duration(intervalSeconds) * time.Second maxGap := 2 * pollEvery @@ -65,20 +63,16 @@ func newRuleTimings(pollEvery time.Duration, intervalSeconds int) ruleTimings { } } -// defaultPollEvery is §5's default per-rule cadence: half the rule's own +// defaultPollEvery is the default per-rule cadence: half the rule's own // evaluation interval. func defaultPollEvery(intervalSeconds int) time.Duration { return time.Duration(intervalSeconds) * time.Second / 2 } -// DeriveTimings computes every resolved rule's ruleTimings, keyed by UID, -// plus the shared globalTimings, from resolved definitions and watch's -// optional --poll-interval override (0 = no override: use each rule's §5 -// default of half its own interval). Per §5.1, a supplied override is used -// verbatim for every rule and is never clamped down to the default even when -// it exceeds intervalSeconds/2 — that case is reported back as a note, not -// silently corrected or refused, because clamping would defeat the one knob -// §5.1 gives an operator for making a tight schedule fit. +// DeriveTimings computes every resolved rule's ruleTimings (keyed by UID) plus +// the shared globalTimings. A non-zero override is used verbatim for every rule +// and never clamped to the default — an override above intervalSeconds/2 widens +// maxGap and is reported as a note, not corrected. func DeriveTimings(defs []Definition, override time.Duration) (rules map[string]ruleTimings, global globalTimings, notes []string) { rules = make(map[string]ruleTimings, len(defs)) for _, d := range defs { @@ -94,42 +88,35 @@ func DeriveTimings(defs []Definition, override time.Duration) (rules map[string] } rules[d.UID] = newRuleTimings(pollEvery, d.IntervalSeconds) } - return rules, deriveGlobalTimings(defs), notes + // In this mode defs ARE the start-of-step snapshot, so they answer what was + // paused at the window open; only the log-mode counterpart uses the header. + return rules, deriveGlobalTimings(defs, pausedSet(defs)), notes } -// DeriveTimingsFromLog is DeriveTimings' log-mode counterpart, and the two -// authorities of P5 are the whole reason it exists as a separate function. -// pollEvery comes from the header — the cadence the recording ACTUALLY used, -// after any --poll-interval override — and maxGap and healthGrace follow from -// it. Re-deriving pollEvery from defs here would compare gaps recorded at the -// override cadence against thresholds computed from the default: exit 2 on a -// clean window when the override was slower, and, worse, a real recorder gap -// passing silently when it was faster. -// -// evalStaleAfter still comes from defs (2 x intervalSeconds): it is a property -// of the rule's own evaluation cadence and is unaffected by how often the gate -// polled. -// -// Three shapes of header are errors rather than a best-effort derivation, -// because each one would otherwise widen a threshold silently: +// pausedSet is Header.pausedAtStart's counterpart for a set of definitions +// resolved at the start of the step, which is the one moment a definition can +// answer "was this paused when the window opened". +func pausedSet(defs []Definition) map[string]bool { + paused := make(map[string]bool, len(defs)) + for _, d := range defs { + paused[d.UID] = d.IsPaused + } + return paused +} + +// DeriveTimingsFromLog is DeriveTimings' log-mode counterpart: pollEvery comes +// from the header (the cadence actually used), not the definitions — re-deriving +// it here would compare recorded gaps against default-cadence thresholds, an +// exit 2 on a clean window (slower override) or a silently passing recorder gap +// (faster override). evalStaleAfter still comes from defs (2 × intervalSeconds). // -// - a rule with no matching definition — a log that names a rule nobody can -// resolve cannot have that rule's coverage proved; -// - a non-positive recorded cadence — a log that cannot say how often it was -// written cannot have maxGap derived, and defaulting the cadence would -// prove a window that was never observed; -// - the same UID twice — last-one-wins would take whichever cadence happened -// to be written last, and a slower duplicate widens maxGap. That is a -// fail-open reachable through nothing but log corruption. +// Three header shapes are hard errors rather than a best-effort derivation, +// because each would silently widen a threshold: a rule with no matching +// definition, a non-positive recorded cadence, and a UID listed twice +// (last-one-wins would widen maxGap on a corrupt log). // -// It checks only the header-to-defs direction. The opposite direction — a -// resolved definition absent from the header — is NOT this function's to -// judge: it is §19.1 step 3's log-identity validation, and it belongs to P9's -// Check, which is the only caller that knows both sets and can name the -// mismatch. Without that check a definition simply gets no timings entry, and -// a downstream lookup would read a zero maxGap: fail-closed (every gap -// exceeds it) but silent, so P9 must reject the set mismatch by name rather -// than let a rule fail for an unexplained reason. +// It checks only the header-to-defs direction. A definition absent from the +// header is Check's log-identity validation to judge, not this function's. func DeriveTimingsFromLog(h Header, defs []Definition) (rules map[string]ruleTimings, global globalTimings, err error) { byUID := make(map[string]Definition, len(defs)) for _, d := range defs { @@ -155,17 +142,21 @@ func DeriveTimingsFromLog(h Header, defs []Definition) (rules map[string]ruleTim pollEvery := time.Duration(lr.PollEverySeconds * float64(time.Second)) rules[lr.UID] = newRuleTimings(pollEvery, def.IntervalSeconds) } - return rules, deriveGlobalTimings(defs), nil + // The header, not defs, decides which rules are excluded from the grace: + // defs were resolved after the window closed. See deriveGlobalTimings. + return rules, deriveGlobalTimings(defs, h.pausedAtStart()), nil } -// deriveGlobalTimings computes transitionGrace and drainTimeout over defs -// (§5, §13.1, §19). A rule paused before the window opened — skipped, §12 — -// is excluded from the transitionGrace max: its `for` value can never fire -// during the window, so counting it would only inflate the wait past what any -// watched rule actually needs (a judgment call the v2 plan makes explicitly -// for this formula; §19's drainTimeout carries no such exclusion, so it still -// runs over every resolved rule). -func deriveGlobalTimings(defs []Definition) globalTimings { +// deriveGlobalTimings computes transitionGrace and drainTimeout over defs. A +// rule skipped at the window open is excluded from the transitionGrace max (its +// `for` can never fire in-window); drainTimeout runs over every resolved rule. +// +// The exclusion authority is pausedAtStart, never Definition.IsPaused: log-mode +// defs are re-resolved after the window closed. Reading the late definitions +// was a quiet fail-open — a rule paused after `to` would drop out of the max, +// collapse the grace past windowEnd (the classification bound AND collection +// deadline), and pass a window the surfacing poll was never recorded for. +func deriveGlobalTimings(defs []Definition, pausedAtStart map[string]bool) globalTimings { var g globalTimings var maxInterval time.Duration for _, d := range defs { @@ -173,7 +164,7 @@ func deriveGlobalTimings(defs []Definition) globalTimings { if interval > maxInterval { maxInterval = interval } - if d.IsPaused { + if pausedAtStart[d.UID] { continue } if candidate := d.For + interval; candidate > g.transitionGrace { @@ -185,25 +176,19 @@ func deriveGlobalTimings(defs []Definition) globalTimings { return g } -// Scheduler drives one per-rule schedule, never a global cycle (§5): a rule -// at intervalSeconds=10 alongside twenty at 300 keeps its own 5s cadence -// without forcing the same cadence onto the other twenty. +// Scheduler drives one per-rule schedule, never a global cycle: a rule at +// intervalSeconds=10 alongside twenty at 300 keeps its own 5s cadence without +// forcing the same cadence onto the other twenty. type Scheduler struct { next map[string]time.Time every map[string]time.Duration } -// NewScheduler builds a Scheduler over per-rule cadences (keyed by UID), -// staggering each rule's initial next-due time across [0, pollEvery) so the -// fleet does not start phase-aligned (§5's burst-bound proof depends on this: -// an already-staggered fleet only re-aligns by chance, briefly, not by -// construction). -// -// It takes cadences rather than whole ruleTimings on purpose: a scheduler -// decides when to poll and nothing else, so it must not be handed maxGap, -// healthGrace or evalStaleAfter. Those are coverage thresholds, they are -// applied by the pure layer at classification time, and the recorder that -// drives this scheduler never applies them at all. +// NewScheduler builds a Scheduler over per-rule cadences, staggering each +// rule's initial next-due time across [0, pollEvery) so a phase-aligned fleet +// (which would void CheckBudget's burst bound) never arises by construction. +// It takes cadences, not ruleTimings: a scheduler only decides when to poll and +// must not be handed coverage thresholds it never applies. func NewScheduler(every map[string]time.Duration, now time.Time) *Scheduler { s := &Scheduler{ next: make(map[string]time.Time, len(every)), @@ -220,13 +205,10 @@ func NewScheduler(every map[string]time.Duration, now time.Time) *Scheduler { return s } -// Due returns the UIDs whose next-due time has arrived, earliest-due-first. -// Ties (equal next-due time) break by tightest cadence first: the burst-bound -// proof in §5 assumes a newly-due tight rule waits at most for one in-flight -// request, which only holds if a simultaneous batch serves the tightest rule -// ahead of slacker ones. A tie-break that instead followed map iteration -// order would silently void that proof — nothing else would fail until a -// phase-aligned fleet opened a mid-run gap in production. +// Due returns the due UIDs, earliest-due-first. Ties break by tightest cadence +// first: the burst bound assumes a newly-due tight rule waits at most one +// in-flight request, which only holds if a simultaneous batch serves the +// tightest rule first. A map-order tie-break would silently void that. func (s *Scheduler) Due(now time.Time) []string { var due []string for uid, t := range s.next { @@ -261,12 +243,9 @@ func (s *Scheduler) Mark(uid string, now time.Time) error { return nil } -// earliestDue returns the earliest scheduled next-due time, and false when the -// scheduler holds no rules at all. The recorder's loop (P6) waits exactly that -// long instead of waking on a fixed tick: a fixed tick either polls a slack -// rule early — spending request budget the §5 formulas already accounted for — -// or wakes too late for the tightest rule and opens a gap inside its own -// maxGap. +// earliestDue returns the earliest next-due time (false when empty). The loop +// waits exactly that long instead of a fixed tick, which would poll slack rules +// early (wasting budget) or wake late for the tightest rule (opening a gap). func (s *Scheduler) earliestDue() (time.Time, bool) { var earliest time.Time ok := false @@ -279,23 +258,18 @@ func (s *Scheduler) earliestDue() (time.Time, bool) { return earliest, ok } -// CheckBudget applies §5's error-at-start check to a fully resolved schedule. -// t and measured are both keyed by rule UID; measured must carry every UID in -// t; a rule this run never measured can't have its budget proved, and a -// silent zero-duration default would be exactly the kind of pass-on-an- -// unproven-window bug §5 exists to catch. CheckBudget fails when any of three -// conditions holds (sanity-checked against §22.3's mixed-interval regression -// in this phase's tests): +// CheckBudget proves at start that a fully resolved schedule can be served. t +// and measured are keyed by UID, and measured must carry every UID in t (a rule +// never measured cannot have its budget proved). It fails on any of three +// conditions: // -// - utilization: the long-run request rate exceeds what concurrency serves; -// - a single rule's own request cannot fit inside its own cadence; -// - the burst bound: the slowest measured request is slower than the -// fleet's tightest cadence, which — even under earliest-due-first -// ordering — can open a mid-run gap bigger than that rule's maxGap. +// - utilization — the long-run request rate exceeds the concurrency; +// - a single rule's request cannot fit inside its own cadence; +// - the burst bound — the slowest measured request is slower than the fleet's +// tightest cadence, which can open a mid-run gap beyond that rule's maxGap. // -// The message never suggests a single interval (§5.1) — only the three -// controls an operator actually has: concurrency, poll-interval, and the -// alert list. +// The message names only the three operator controls: concurrency, +// poll-interval, and the alert list. func CheckBudget(t map[string]ruleTimings, measured map[string]time.Duration, concurrency int) error { if len(t) == 0 { return nil @@ -365,24 +339,32 @@ func CheckBudget(t map[string]ruleTimings, measured map[string]time.Duration, co return fmt.Errorf("%s", b.String()) } -// StartupSummary formats §13.2's required pre-run print: the total planned -// run time and the rule (with its `for` value) that set transitionGrace, plus -// a warning when the grace eats more than graceWarnFraction of the requested -// window. from/to are the requested classification window. +// graceSourceOrNone is the single "none" default for the grace-source field: +// an empty source means no rule contributed a transitionGrace. Applied here so +// StartupSummary and the human table print the same thing. +func graceSourceOrNone(source string) string { + if source == "" { + return "none" + } + return source +} + +// StartupSummary formats the pre-run print an operator sees before the wait: +// the total planned run time and the rule (with its `for` value) that set +// transitionGrace, plus a warning when the grace eats more than +// graceWarnFraction of the requested window. from/to are the requested +// classification window. func StartupSummary(from, to time.Time, global globalTimings) (summary, warning string) { window := to.Sub(from) total := window + global.transitionGrace + global.drainTimeout - source := global.graceSource - if source == "" { - source = "none" - } + source := graceSourceOrNone(global.graceSource) summary = fmt.Sprintf( - "planned run time: %s (window %s + transitionGrace %s [source: %s] + drainTimeout %s)", - total, window, global.transitionGrace, source, global.drainTimeout) + "planned run time: %s\n window %s + transitionGrace %s + drainTimeout %s\n transitionGrace source: %s", + total, window, global.transitionGrace, global.drainTimeout, source) if window > 0 && float64(global.transitionGrace) > float64(window)*graceWarnFraction { warning = fmt.Sprintf( - "transitionGrace %s is more than %.0f%% of the window %s (source: %s) — the window may be too short for this alert's `for`", + "transitionGrace %s is more than %.0f%% of the window %s — the window may be too short for this alert's `for`\n source: %s", global.transitionGrace, graceWarnFraction*100, window, source) } return summary, warning diff --git a/grafana-alertcheck/internal/gate/schedule_test.go b/grafana-alertcheck/internal/gate/schedule_test.go index e897414bc..ace598650 100644 --- a/grafana-alertcheck/internal/gate/schedule_test.go +++ b/grafana-alertcheck/internal/gate/schedule_test.go @@ -1,9 +1,10 @@ package gate import ( - "strings" "testing" "time" + + "github.com/stretchr/testify/require" ) func TestDeriveTimings_Default(t *testing.T) { @@ -11,22 +12,12 @@ func TestDeriveTimings_Default(t *testing.T) { {UID: "r1", Title: "R1", IntervalSeconds: 60}, } rules, _, notes := DeriveTimings(defs, 0) - if len(notes) != 0 { - t.Fatalf("notes = %v, want none", notes) - } + require.Empty(t, notes) rt := rules["r1"] - if rt.pollEvery != 30*time.Second { - t.Errorf("pollEvery = %s, want 30s", rt.pollEvery) - } - if rt.maxGap != 60*time.Second { - t.Errorf("maxGap = %s, want 60s", rt.maxGap) - } - if rt.healthGrace != 60*time.Second { - t.Errorf("healthGrace = %s, want 60s", rt.healthGrace) - } - if rt.evalStaleAfter != 120*time.Second { - t.Errorf("evalStaleAfter = %s, want 120s", rt.evalStaleAfter) - } + require.Equal(t, 30*time.Second, rt.pollEvery) + require.Equal(t, 60*time.Second, rt.maxGap) + require.Equal(t, 60*time.Second, rt.healthGrace) + require.Equal(t, 120*time.Second, rt.evalStaleAfter) } func TestDeriveTimings_OverrideVerbatimNoClamp(t *testing.T) { @@ -35,23 +26,16 @@ func TestDeriveTimings_OverrideVerbatimNoClamp(t *testing.T) { } rules, _, notes := DeriveTimings(defs, 20*time.Second) rt := rules["r1"] - if rt.pollEvery != 20*time.Second { - t.Fatalf("pollEvery = %s, want the override verbatim (20s), never clamped down to the 5s default", rt.pollEvery) - } - if rt.maxGap != 40*time.Second { - t.Errorf("maxGap = %s, want 2x the override (40s)", rt.maxGap) - } - if len(notes) != 1 || !strings.Contains(notes[0], "R1") { - t.Fatalf("notes = %v, want one note naming R1's exceeded default", notes) - } + require.Equal(t, 20*time.Second, rt.pollEvery, "the override verbatim (20s), never clamped down to the 5s default") + require.Equal(t, 40*time.Second, rt.maxGap) + require.Len(t, notes, 1) + require.Contains(t, notes[0], "R1") } func TestDeriveTimings_OverrideBelowDefaultNoNote(t *testing.T) { defs := []Definition{{UID: "r1", Title: "R1", IntervalSeconds: 60}} // default pollEvery = 30s _, _, notes := DeriveTimings(defs, 5*time.Second) - if len(notes) != 0 { - t.Fatalf("notes = %v, want none when the override tightens rather than exceeds the default", notes) - } + require.Empty(t, notes) } func TestDeriveTimings_TransitionGraceExcludesSkippedRule(t *testing.T) { @@ -61,28 +45,86 @@ func TestDeriveTimings_TransitionGraceExcludesSkippedRule(t *testing.T) { } _, global, _ := DeriveTimings(defs, 0) want := time.Minute + 60*time.Second // r1's for+interval; r2 (skipped) must not win despite its huge `for` - if global.transitionGrace != want { - t.Fatalf("transitionGrace = %s, want %s (paused rule r2 must be excluded from the max)", global.transitionGrace, want) - } - if !strings.Contains(global.graceSource, "Tight") { - t.Errorf("graceSource = %q, want it to name the contributing rule Tight", global.graceSource) - } + require.Equal(t, want, global.transitionGrace) + require.Contains(t, global.graceSource, "Tight") +} + +// `for: 1d` and `for: 1w` parse correctly (parse_ruler_test.go), +// but that alone never proves they flow into transitionGrace — a Prometheus +// duration parser that silently truncated to time.Duration's other units, or +// a transitionGrace derivation that only ever saw hand-built values, could +// each pass every existing test and still be wrong together. This drives the +// real ruler_rules.json fixture (rule0000010, for:1w, DERIVED to exercise the +// w unit — testdata/README.md) through ParseDefinitions and DeriveTimings. +func TestDeriveTimings_RealForOneWeekRuleSetsTransitionGrace(t *testing.T) { + defs := rulerDefs(t) + _, global, notes := DeriveTimings(defs, 0) + require.Empty(t, notes, "no --poll-interval override is given, so no override note should fire") + + want := 7*24*time.Hour + 60*time.Second // rule0000010: for=1w, intervalSeconds=60 + require.Equal(t, want, global.transitionGrace) + require.Contains(t, global.graceSource, "Example Failure Ratio Above 10 Percent Weekly") } func TestDeriveTimings_TransitionGraceZeroWhenAllSkipped(t *testing.T) { defs := []Definition{{UID: "r1", Title: "R1", IntervalSeconds: 60, For: time.Hour, IsPaused: true}} _, global, _ := DeriveTimings(defs, 0) - if global.transitionGrace != 0 { - t.Fatalf("transitionGrace = %s, want 0 when every rule is skipped", global.transitionGrace) - } + require.Zero(t, global.transitionGrace) +} + +// TestDeriveTimingsFromLog_TransitionGraceFollowsTheHeaderNotTheDefinition +// pins the log-mode authority for the grace exclusion. Definitions are +// re-resolved AFTER the window closed, so "paused" in a definition says +// nothing about whether the rule was watched during it. +// +// The fail-open direction is the first case. transitionGrace exists so a +// condition arising just before `to` is still seen when it surfaces at +// to + `for`, and windowEnd is both the classification bound and the +// collection deadline — so a rule somebody paused after `to` dropping out of +// the max collapses the grace, the surfacing poll is never recorded, and the +// run reports clean. +func TestDeriveTimingsFromLog_TransitionGraceFollowsTheHeaderNotTheDefinition(t *testing.T) { + loggedRule := func(uid string, pausedAtStart bool) LoggedRule { + return LoggedRule{UID: uid, Title: uid, IntervalSeconds: 60, PollEverySeconds: 30, IsPaused: pausedAtStart} + } + // The definition says paused in BOTH cases: it is the post-window reading, + // and it must change nothing. + defs := []Definition{{UID: "r1", Title: "R1", IntervalSeconds: 60, For: 5 * time.Minute, IsPaused: true}} + want := 5*time.Minute + 60*time.Second + + t.Run("header says active: the rule stays in the max", func(t *testing.T) { + h := Header{Rules: []LoggedRule{loggedRule("r1", false)}} + _, global, err := DeriveTimingsFromLog(h, defs) + require.NoError(t, err) + require.Equal(t, want, global.transitionGrace, + "the rule was active when the recording opened, so a pause applied afterwards must not shrink the window") + require.Contains(t, global.graceSource, "R1") + }) + + t.Run("header says paused: the rule stays out", func(t *testing.T) { + h := Header{Rules: []LoggedRule{loggedRule("r1", true)}} + _, global, err := DeriveTimingsFromLog(h, defs) + require.NoError(t, err) + require.Zero(t, global.transitionGrace, + "a rule paused before the window opened can never fire during it") + }) + + t.Run("drainTimeout counts every rule either way", func(t *testing.T) { + // drainTimeout carries no pause exclusion, so both headers give the + // same floor-bound value. + for _, pausedAtStart := range []bool{false, true} { + h := Header{Rules: []LoggedRule{loggedRule("r1", pausedAtStart)}} + _, global, err := DeriveTimingsFromLog(h, defs) + require.NoError(t, err) + require.Equalf(t, minDrainTimeout, global.drainTimeout, "pausedAtStart=%v", pausedAtStart) + } + }) } func TestDeriveTimings_DrainTimeoutIncludesPaused(t *testing.T) { defs := []Definition{{UID: "r1", Title: "R1", IntervalSeconds: 10}} _, global, _ := DeriveTimings(defs, 0) - if global.drainTimeout != minDrainTimeout { - t.Fatalf("drainTimeout = %s, want the %s floor", global.drainTimeout, minDrainTimeout) - } + require.Equal(t, minDrainTimeout, global.drainTimeout) } func TestDeriveTimings_DrainTimeoutFloor(t *testing.T) { @@ -92,22 +134,17 @@ func TestDeriveTimings_DrainTimeoutFloor(t *testing.T) { } _, global, _ := DeriveTimings(defs, 0) // double the longest interval (2 * 180s) should be the drain timeout - if global.drainTimeout != 2*180*time.Second { - t.Fatalf("drainTimeout = %s, want %s", global.drainTimeout, 2*180*time.Second) - } + require.Equal(t, 2*180*time.Second, global.drainTimeout) } func TestDeriveTimings_DrainTimeoutAboveFloor(t *testing.T) { defs := []Definition{{UID: "r1", Title: "R1", IntervalSeconds: 300}} // 2x300s = 600s > 2m floor _, global, _ := DeriveTimings(defs, 0) - if global.drainTimeout != 600*time.Second { - t.Fatalf("drainTimeout = %s, want 600s", global.drainTimeout) - } + require.Equal(t, 600*time.Second, global.drainTimeout) } -// TestScheduler_DueOrderingTiesBreakByTightestCadence pins the ordering -// invariant the burst bound depends on (§5): when several rules become due at -// the exact same instant, Due must serve the tightest cadence first, not +// The ordering invariant the burst bound depends on: when several rules become +// due at the exact same instant, Due must serve the tightest cadence first, not // whatever order the underlying map happens to iterate in. A refactor that // loses this ordering must fail here, not in a production phase-aligned gap. func TestScheduler_DueOrderingTiesBreakByTightestCadence(t *testing.T) { @@ -124,9 +161,8 @@ func TestScheduler_DueOrderingTiesBreakByTightestCadence(t *testing.T) { }, } due := s.Due(now) - if len(due) != 4 || due[0] != "tight" { - t.Fatalf("Due = %v, want the tightest-cadence rule (tight) first when all are simultaneously due", due) - } + require.Len(t, due, 4) + require.Equal(t, "tight", due[0], "the tightest-cadence rule must be first when all are simultaneously due") } func TestScheduler_DueExcludesNotYetDue(t *testing.T) { @@ -136,9 +172,8 @@ func TestScheduler_DueExcludesNotYetDue(t *testing.T) { every: map[string]time.Duration{"soon": 10 * time.Second, "later": 10 * time.Second}, } due := s.Due(now) - if len(due) != 1 || due[0] != "soon" { - t.Fatalf("Due = %v, want only [soon]", due) - } + require.Len(t, due, 1) + require.Equal(t, "soon", due[0]) } func TestScheduler_MarkAdvancesNextDue(t *testing.T) { @@ -147,15 +182,9 @@ func TestScheduler_MarkAdvancesNextDue(t *testing.T) { next: map[string]time.Time{"r1": now}, every: map[string]time.Duration{"r1": 30 * time.Second}, } - if err := s.Mark("r1", now); err != nil { - t.Fatalf("Mark: unexpected error: %v", err) - } - if got := s.Due(now); len(got) != 0 { - t.Fatalf("Due right after Mark = %v, want none (next due is 30s out)", got) - } - if got := s.Due(now.Add(30 * time.Second)); len(got) != 1 { - t.Fatalf("Due at next-due time = %v, want [r1]", got) - } + require.NoError(t, s.Mark("r1", now), "Mark must succeed for a known rule") + require.Empty(t, s.Due(now), "next due is 30s out") + require.Len(t, s.Due(now.Add(30*time.Second)), 1) } func TestScheduler_MarkUnknownUIDFails(t *testing.T) { @@ -164,18 +193,16 @@ func TestScheduler_MarkUnknownUIDFails(t *testing.T) { next: map[string]time.Time{"r1": now}, every: map[string]time.Duration{"r1": 30 * time.Second}, } - if err := s.Mark("not-a-rule", now); err == nil { - t.Fatalf("Mark of an unknown uid: want error, got nil (a missing cadence must not read as zero and loop)") - } + err := s.Mark("not-a-rule", now) + require.Error(t, err, "a missing cadence must not read as zero and loop") // The failed Mark must not have inserted a bogus next-due entry. - if _, ok := s.next["not-a-rule"]; ok { - t.Errorf("Mark of an unknown uid inserted a next-due entry") - } + _, ok := s.next["not-a-rule"] + require.False(t, ok, "a failed Mark must not insert a next-due entry") } // TestScheduler_PerRuleCadenceOverTime simulates a run and counts how often -// each rule comes due, pinning §5's core claim: schedules are per rule, never -// a global cycle. A tight rule must be polled at its own cadence regardless +// each rule comes due: schedules are per rule, never a global cycle. A tight +// rule must be polled at its own cadence regardless // of what slower rules in the same fleet need, and a slack rule must never be // forced onto the tight rule's cadence. func TestScheduler_PerRuleCadenceOverTime(t *testing.T) { @@ -193,24 +220,19 @@ func TestScheduler_PerRuleCadenceOverTime(t *testing.T) { now := start.Add(elapsed) for _, uid := range s.Due(now) { counts[uid]++ - if err := s.Mark(uid, now); err != nil { - t.Fatalf("Mark(%q): unexpected error: %v", uid, err) - } + require.NoErrorf(t, s.Mark(uid, now), "Mark(%q)", uid) } } // 900s of runtime: "tight" (10s cadence) polls ~90 times, "slack" (300s // cadence) ~3 times. Assert the ratio holds rather than an exact count, // since the staggered initial offset shifts each by up to one cadence. - if counts["tight"] < 85 || counts["tight"] > 91 { - t.Errorf("tight polled %d times over 900s, want ~90 (its own 10s cadence)", counts["tight"]) - } - if counts["slack"] < 2 || counts["slack"] > 4 { - t.Errorf("slack polled %d times over 900s, want ~3 (its own 300s cadence, not tight's)", counts["slack"]) - } - if counts["slack"] >= counts["tight"] { - t.Fatalf("slack polled as often as tight (%d vs %d) — schedules must be per rule, not a shared global cycle", counts["slack"], counts["tight"]) - } + require.GreaterOrEqual(t, counts["tight"], 85) + require.LessOrEqual(t, counts["tight"], 91) + require.GreaterOrEqual(t, counts["slack"], 2) + require.LessOrEqual(t, counts["slack"], 4) + require.Less(t, counts["slack"], counts["tight"], + "schedules must be per rule, not a shared global cycle") } func TestNewScheduler_StaggersWithinPollEvery(t *testing.T) { @@ -218,16 +240,14 @@ func TestNewScheduler_StaggersWithinPollEvery(t *testing.T) { rules := map[string]time.Duration{"r1": 100 * time.Second} s := NewScheduler(rules, now) offset := s.next["r1"].Sub(now) - if offset < 0 || offset >= 100*time.Second { - t.Fatalf("initial offset = %s, want within [0, 100s)", offset) - } + require.GreaterOrEqual(t, offset, time.Duration(0)) + require.Less(t, offset, 100*time.Second) } func TestScheduler_EarliestDueEmpty(t *testing.T) { s := &Scheduler{next: map[string]time.Time{}, every: map[string]time.Duration{}} - if _, ok := s.earliestDue(); ok { - t.Fatalf("earliestDue on an empty scheduler = ok=true, want false") - } + _, ok := s.earliestDue() + require.False(t, ok) } // A zero next-due time is real, not an empty scheduler. @@ -237,12 +257,8 @@ func TestScheduler_EarliestDueZeroTime(t *testing.T) { every: map[string]time.Duration{"r1": time.Second}, } earliest, ok := s.earliestDue() - if !ok { - t.Fatalf("earliestDue = ok=false, want true (the zero time is a real next-due, not an empty scheduler)") - } - if !earliest.IsZero() { - t.Errorf("earliestDue = %v, want the zero time", earliest) - } + require.True(t, ok, "the zero time is a real next-due, not an empty scheduler") + require.Truef(t, earliest.IsZero(), "earliestDue = %v, want the zero time", earliest) } func TestScheduler_EarliestDuePicksMinimum(t *testing.T) { @@ -255,18 +271,12 @@ func TestScheduler_EarliestDuePicksMinimum(t *testing.T) { every: map[string]time.Duration{"later": time.Minute, "soon": time.Minute}, } earliest, ok := s.earliestDue() - if !ok { - t.Fatalf("earliestDue = ok=false, want true") - } - if !earliest.Equal(now.Add(time.Minute)) { - t.Errorf("earliestDue = %v, want the earliest next-due time", earliest) - } + require.True(t, ok) + require.Truef(t, earliest.Equal(now.Add(time.Minute)), "earliestDue = %v, want the earliest next-due time", earliest) } -// TestCheckBudget_MixedIntervalRegression is §22.3's sanity check from the -// plan: one rule at 10s beside twenty at 300s, all measured ~1.8s, must not -// error at any reasonable concurrency — the exact case a naive worst-case-slot -// simulation would wrongly fail. +// One rule at 10s beside twenty at 300s, all measured ~1.8s, must not error at +// any reasonable concurrency — the exact case a naive worst-case-slot func TestCheckBudget_MixedIntervalRegression(t *testing.T) { timings := map[string]ruleTimings{"tight": {pollEvery: 5 * time.Second}} measured := map[string]time.Duration{"tight": 1800 * time.Millisecond} @@ -275,9 +285,7 @@ func TestCheckBudget_MixedIntervalRegression(t *testing.T) { timings[uid] = ruleTimings{pollEvery: 150 * time.Second} measured[uid] = 1800 * time.Millisecond } - if err := CheckBudget(timings, measured, 1); err != nil { - t.Fatalf("CheckBudget = %v, want nil (utilization 0.6, burst bound 1.8s <= 5s)", err) - } + require.NoError(t, CheckBudget(timings, measured, 1)) } func TestCheckBudget_UtilizationExceeded(t *testing.T) { @@ -287,9 +295,7 @@ func TestCheckBudget_UtilizationExceeded(t *testing.T) { } measured := map[string]time.Duration{"a": 9 * time.Second, "b": 9 * time.Second} err := CheckBudget(timings, measured, 1) - if err == nil { - t.Fatal("CheckBudget = nil, want an error: utilization 1.8 > concurrency 1") - } + require.Error(t, err) assertBudgetMessage(t, err.Error()) } @@ -297,9 +303,7 @@ func TestCheckBudget_SingleRuleExceedsOwnCadence(t *testing.T) { timings := map[string]ruleTimings{"slow": {pollEvery: 5 * time.Second}} measured := map[string]time.Duration{"slow": 6 * time.Second} err := CheckBudget(timings, measured, 10) - if err == nil { - t.Fatal("CheckBudget = nil, want an error: measured 6s exceeds its own 5s poll-interval") - } + require.Error(t, err, "measured 6s exceeds its own 5s poll-interval") assertBudgetMessage(t, err.Error()) } @@ -313,12 +317,8 @@ func TestCheckBudget_BurstBoundViolation(t *testing.T) { } measured := map[string]time.Duration{"tight": 100 * time.Millisecond, "slow": 3 * time.Second} err := CheckBudget(timings, measured, 10) - if err == nil { - t.Fatal("CheckBudget = nil, want a burst-bound error: slow's 3s measured exceeds tight's 2s cadence") - } - if !strings.Contains(err.Error(), "burst bound") { - t.Errorf("error = %q, want it to name the burst bound", err.Error()) - } + require.Error(t, err, "slow's 3s measured exceeds tight's 2s cadence") + require.Contains(t, err.Error(), "burst bound") assertBudgetMessage(t, err.Error()) } @@ -328,17 +328,13 @@ func TestCheckBudget_BurstBoundOKWhenNotExceeded(t *testing.T) { "slow": {pollEvery: 100 * time.Second}, } measured := map[string]time.Duration{"tight": 100 * time.Millisecond, "slow": 1800 * time.Millisecond} - if err := CheckBudget(timings, measured, 10); err != nil { - t.Fatalf("CheckBudget = %v, want nil (1.8s <= 5s tightest cadence)", err) - } + require.NoError(t, CheckBudget(timings, measured, 10)) } func TestCheckBudget_MissingMeasurementIsAnError(t *testing.T) { timings := map[string]ruleTimings{"r1": {pollEvery: 30 * time.Second}} err := CheckBudget(timings, map[string]time.Duration{}, 10) - if err == nil { - t.Fatal("CheckBudget = nil, want an error: r1 was never measured (fail closed, not a silent zero)") - } + require.Error(t, err, "r1 was never measured (fail closed, not a silent zero)") } func TestCheckBudget_MissingMixedMeasurementIsAnError(t *testing.T) { @@ -347,15 +343,12 @@ func TestCheckBudget_MissingMixedMeasurementIsAnError(t *testing.T) { "slow": {pollEvery: 100 * time.Second}, } measured := map[string]time.Duration{"tight": 100 * time.Millisecond} - if err := CheckBudget(timings, measured, 10); err == nil { - t.Fatal("CheckBudget = nil, want an error: slow was never measured (fail closed, not a silent zero)") - } + require.Error(t, CheckBudget(timings, measured, 10), + "slow was never measured (fail closed, not a silent zero)") } func TestCheckBudget_EmptyScheduleIsFine(t *testing.T) { - if err := CheckBudget(nil, nil, 1); err != nil { - t.Fatalf("CheckBudget = %v, want nil for an empty schedule", err) - } + require.NoError(t, CheckBudget(nil, nil, 1)) } func TestCheckBudget_NonPositivePollIntervalIsAnError(t *testing.T) { @@ -363,24 +356,18 @@ func TestCheckBudget_NonPositivePollIntervalIsAnError(t *testing.T) { timings := map[string]ruleTimings{"r1": {pollEvery: pe}} measured := map[string]time.Duration{"r1": time.Second} err := CheckBudget(timings, measured, 1) - if err == nil { - t.Fatalf("CheckBudget(pollEvery=%s) = nil, want error (non-positive poll-interval would divide by zero)", pe) - } - if !strings.Contains(err.Error(), "non-positive") { - t.Errorf("error %q does not name the non-positive poll-interval", err.Error()) - } + require.Errorf(t, err, "pollEvery=%s would divide by zero", pe) + require.Contains(t, err.Error(), "non-positive", "the error must name the non-positive poll-interval") } } -// assertBudgetMessage checks §5.1's required message contents: a measured -// duration is present, and all three controls are named — never a single -// suggested interval. +// assertBudgetMessage checks the message contents: a measured duration is +// present, and all three controls are named — never a single suggested +// interval. func assertBudgetMessage(t *testing.T, msg string) { t.Helper() for _, want := range []string{"measured", "concurrency", "poll-interval", "fewer"} { - if !strings.Contains(msg, want) { - t.Errorf("message %q missing %q", msg, want) - } + require.Contains(t, msg, want) } } @@ -393,15 +380,26 @@ func TestStartupSummary_WarningWhenGraceTooLarge(t *testing.T) { to := from.Add(10 * time.Minute) global := globalTimings{transitionGrace: 5 * time.Minute, graceSource: "R (for=4m30s, interval=30s)", drainTimeout: time.Minute} summary, warning := StartupSummary(from, to, global) - if !strings.Contains(summary, "planned run time") { - t.Errorf("summary = %q, want it to name the planned run time", summary) - } - if warning == "" { - t.Fatal("warning = \"\", want one: transitionGrace (5m) > 1/4 of the 10m window") - } - if !strings.Contains(warning, "R (for=4m30s, interval=30s)") { - t.Errorf("warning = %q, want it to name the grace source", warning) - } + require.Contains(t, summary, "planned run time") + require.NotEmpty(t, warning, "transitionGrace (5m) > 1/4 of the 10m window") + require.Contains(t, warning, "R (for=4m30s, interval=30s)") +} + +// The test above pins the warning formula with a hand-built globalTimings. +// This drives the same warning off the real ruler_rules.json fixture's for:1w +// rule instead, tying ParseDefinitions and DeriveTimings into the warning end +// to end. +func TestStartupSummary_RealForOneWeekRuleTriggersWarning(t *testing.T) { + defs := rulerDefs(t) + _, global, notes := DeriveTimings(defs, 0) + require.Empty(t, notes) + + from := time.Date(2026, 1, 1, 0, 0, 0, 0, time.UTC) + to := from.Add(10 * time.Minute) // transitionGrace (>1w) dwarfs 1/4 of this window + summary, warning := StartupSummary(from, to, global) + require.Contains(t, summary, "planned run time") + require.NotEmpty(t, warning) + require.Contains(t, warning, "Example Failure Ratio Above 10 Percent Weekly") } func TestStartupSummary_NoWarningWhenGraceSmall(t *testing.T) { @@ -409,16 +407,12 @@ func TestStartupSummary_NoWarningWhenGraceSmall(t *testing.T) { to := from.Add(time.Hour) global := globalTimings{transitionGrace: time.Minute, graceSource: "R (for=30s, interval=30s)", drainTimeout: time.Minute} _, warning := StartupSummary(from, to, global) - if warning != "" { - t.Fatalf("warning = %q, want none: 1m grace is well under 1/4 of a 1h window", warning) - } + require.Empty(t, warning) } func TestStartupSummary_NoGraceSourceReadsNone(t *testing.T) { from := time.Date(2026, 1, 1, 0, 0, 0, 0, time.UTC) to := from.Add(time.Hour) summary, _ := StartupSummary(from, to, globalTimings{}) - if !strings.Contains(summary, "none") { - t.Fatalf("summary = %q, want it to read \"none\" when no rule set the grace", summary) - } + require.Contains(t, summary, "none") } diff --git a/grafana-alertcheck/internal/gate/source.go b/grafana-alertcheck/internal/gate/source.go index 6e587b580..721151e21 100644 --- a/grafana-alertcheck/internal/gate/source.go +++ b/grafana-alertcheck/internal/gate/source.go @@ -22,8 +22,8 @@ import ( // across retries. const maxResponseBytes = 25 << 20 // 25 MiB -// Clock is the seam that lets tests advance time without sleeping (§22) — the -// only two operations the gate ever needs from a clock. +// Clock is the seam that lets tests advance time without sleeping — the only +// two operations the gate ever needs from a clock. type Clock interface { Now() time.Time After(d time.Duration) <-chan time.Time @@ -37,18 +37,17 @@ func (SystemClock) After(d time.Duration) <-chan time.Time { return time.After(d // Observation is one successful poll of the state endpoint for a single rule. type Observation struct { - Rules []StateRule // may be empty — an authoritative 2xx saying the rule is absent (§14.5) - GrafanaNow time.Time // the Date header — H4 - Skew time.Duration // serverDate - (t_send+t_headers)/2, signed (§16) + Rules []StateRule // may be empty — an authoritative 2xx saying the rule is absent + GrafanaNow time.Time // the response's Date header + Skew time.Duration // serverDate - (t_send+t_headers)/2, signed SkewBound time.Duration // (t_headers-t_send)/2 — RTT/2 to the response headers Latency time.Duration // t_send through the full body read — see requestResult.Latency } -// TransportError marks a failure worth retrying: a non-2xx response, a -// network failure, or a body that failed to parse. It is never a deleted rule -// (an authoritative 2xx with no matching rule is not this) and never a clock -// problem (a missing/unparseable Date header or an out-of-bounds skew is a -// hard error instead — see doRequest). Never conflate them (§14.5). +// TransportError marks a failure worth retrying: a 5xx/429 response, a network +// failure, or a body that failed to parse. Not a 4xx (wrong auth, missing +// resource), not a deleted rule (an authoritative 2xx) and not a clock problem +// (a hard error — see doRequest). type TransportError struct { Err error Status int // 0 when the failure never got a status (network/transport failure) @@ -63,14 +62,10 @@ func (e *TransportError) Error() string { func (e *TransportError) Unwrap() error { return e.Err } -// RetryExhaustedError is what retryTransport returns once it gives up after -// too many sequential *TransportError failures. It deliberately does not -// implement Unwrap into the underlying *TransportError: once retries are -// exhausted the result is a hard, terminal failure, and -// errors.AsType[*TransportError] must never re-classify it as retryable — -// that is the exact conflation §19.3 case 1 forbids. Cause is still exposed -// as a plain field (and folded into Error()'s text) so a caller can log or -// inspect it; it just cannot flow back into the retry classification. +// RetryExhaustedError is the hard, terminal failure retryTransport returns once +// it gives up. It deliberately omits Unwrap into *TransportError so +// errors.AsType can never re-classify it as retryable; Cause stays a plain +// field for logging only. type RetryExhaustedError struct { Failures int Cause error @@ -81,7 +76,7 @@ func (e *RetryExhaustedError) Error() string { } // Source is everything the gate reads from Grafana. httpSource is the one -// production implementation; every later phase's tests use a scripted fake +// production implementation; the tests use a scripted fake // (source_fake_test.go) instead of real HTTP. type Source interface { Version(ctx context.Context) (string, error) @@ -117,17 +112,7 @@ func parseGrafanaVersion(s string) (grafanaVersion, error) { var v grafanaVersion fields := [3]*int{&v.major, &v.minor, &v.patch} for i, field := range fields { - // Trim any trailing non-digit suffix (prerelease/build metadata, e.g. - // "0+security") rather than requiring an exact numeric match. - digits := parts[i] - j := 0 - for j < len(digits) && digits[j] >= '0' && digits[j] <= '9' { - j++ - } - if j == 0 { - return grafanaVersion{}, fmt.Errorf("unparseable version %q", s) - } - n, err := strconv.Atoi(digits[:j]) + n, err := strconv.Atoi(parts[i]) if err != nil { return grafanaVersion{}, fmt.Errorf("unparseable version %q: %w", s, err) } @@ -137,7 +122,7 @@ func parseGrafanaVersion(s string) (grafanaVersion, error) { } // supportedGrafanaMin and supportedGrafanaMax bound the platform this gate is -// verified against (§2.7 control 2, §21.5): >= 13.0.0, < 14.0.0. +// verified against: >= 13.0.0, < 14.0.0. var ( supportedGrafanaMin = grafanaVersion{13, 0, 0} supportedGrafanaMax = grafanaVersion{14, 0, 0} // exclusive @@ -145,8 +130,9 @@ var ( // CheckGrafanaVersion enforces the supported range. An unparseable or // out-of-range version is a hard error naming both what was found and what is -// supported — trusting an unverified schema is exactly the deprecation risk -// §2.7 control 2 exists to catch. +// supported: the response schemas this gate parses are only verified against +// that range, and trusting an unverified one is how a deprecation turns into a +// silent misread. func CheckGrafanaVersion(version string) error { v, err := parseGrafanaVersion(version) if err != nil { @@ -160,12 +146,12 @@ func CheckGrafanaVersion(version string) error { return nil } -// httpSource is the production Source: stdlib net/http only, bearer auth -// from a token supplied at construction (the caller reads it from the -// environment — §20.2 — this type never touches env itself), and manual -// strict decoding via ParseState/ParseDefinitions (H1). The retry limit and -// backoff parameters are struct fields with production defaults set here, -// not package constants, so a test can shrink them without a hook. +// httpSource is the production Source: stdlib net/http only, bearer auth from +// a token supplied at construction (the caller reads it from the environment; +// this type never touches env itself), and manual strict decoding via +// ParseState/ParseDefinitions. The retry limit and backoff parameters are +// struct fields with production defaults set here, not package constants, so a +// test can shrink them without a hook. type httpSource struct { baseURL string token string @@ -177,9 +163,8 @@ type httpSource struct { backoffCap time.Duration } -// NewHTTPSource builds the production Source. token is never logged and -// never enters an error string (§20.2) — it is used only to set the -// Authorization header. +// NewHTTPSource builds the production Source. token is never logged and never +// enters an error string — it is used only to set the Authorization header. func NewHTTPSource(baseURL, token string, clock Clock) Source { return &httpSource{ baseURL: strings.TrimSuffix(baseURL, "/"), @@ -236,8 +221,8 @@ func (s *httpSource) RuleState(ctx context.Context, title string) (Observation, if parseErr != nil { // Treated as transient, not a schema break: an unparseable 2xx // is far more likely a mid-stream hiccup than a permanent shape - // change, and H1's strict parser already turns a real shape - // change into a loud per-field error the moment it's visible. + // change, and the strict parser already turns a real shape change + // into a loud per-field error the moment it's visible. return Observation{}, &TransportError{Err: fmt.Errorf("parse rule state: %w", parseErr)} } return Observation{ @@ -252,35 +237,28 @@ func (s *httpSource) RuleState(ctx context.Context, title string) (Observation, // requestResult is the outcome of one successful HTTP attempt in doRequest: // the raw body plus everything derived from timing the round trip against -// the response's own clock (§16). +// the response's own clock. type requestResult struct { Body []byte - ServerDate time.Time // the Date header — H4 + ServerDate time.Time // the response's Date header Skew time.Duration // serverDate - (t_send+t_headers)/2, signed SkewBound time.Duration // (t_headers-t_send)/2 — RTT/2 to the response headers - // Latency spans t_send through the full body read (§5.2's budget check - // needs the whole poll's wall time, or a schedule feasibility check that - // only sees header latency goes optimistic — fail-open). It does not - // include the caller's subsequent JSON parse (ParseState/ParseDefinitions - // run outside doRequest); if P4's budget accounting needs parse time - // folded in too, extend here rather than approximating it at the call - // site. + // Latency spans t_send through the full body read: the budget check needs + // the whole poll's wall time (header-only latency would be fail-open). The + // caller's JSON parse runs outside doRequest; extend here if that ever must + // be folded in. Latency time.Duration } -// doRequest performs one HTTP GET and classifies the outcome (§14.5, §16): -// a network failure, a non-2xx status, or a body-read failure is retryable -// (*TransportError); a missing or unparseable Date header, or a skew beyond -// skewHardLimit, is a hard error — retrying can never fix either, so neither -// may enter the backoff loop (H4). +// doRequest performs one HTTP GET and classifies the outcome: a 5xx/429, a +// network failure, or a body-read failure is retryable (*TransportError); a +// 4xx (wrong auth, missing resource — retrying cannot fix it), a missing or +// unparseable Date header, or a skew beyond SkewHardLimit is a hard error, so +// none of those enters the backoff loop. // -// The Date-header/skew check runs for every endpoint this hits, including -// /api/health — broader than §16's own scope, which only discusses the state -// endpoint. Deliberate: a skewed clock discovered only once RuleState starts -// polling is a skew that has already masked whatever /api/health and the -// ruler read reported; failing closed at the first response catches it -// before any of that is trusted, and every response comes with a Date header -// for free. +// The Date/skew check runs on every endpoint (even /api/health): a skew only +// noticed once RuleState starts polling has already masked earlier reads, so it +// fails closed on the first response. func (s *httpSource) doRequest(ctx context.Context, path string) (requestResult, error) { req, buildErr := http.NewRequestWithContext(ctx, http.MethodGet, s.baseURL+path, nil) if buildErr != nil { @@ -312,12 +290,16 @@ func (s *httpSource) doRequest(ctx context.Context, path string) (requestResult, latency := tBodyRead.Sub(tSend) if resp.StatusCode < 200 || resp.StatusCode >= 300 { - return requestResult{}, &TransportError{Err: fmt.Errorf("unexpected status %d", resp.StatusCode), Status: resp.StatusCode} + err := fmt.Errorf("unexpected status %d", resp.StatusCode) + if resp.StatusCode >= 400 && resp.StatusCode < 500 && resp.StatusCode != http.StatusTooManyRequests { + return requestResult{}, err + } + return requestResult{}, &TransportError{Err: err, Status: resp.StatusCode} } dateHeader := resp.Header.Get("Date") if dateHeader == "" { - return requestResult{}, fmt.Errorf("%s: response has no Date header (H4)", path) + return requestResult{}, fmt.Errorf("%s: response has no Date header", path) } serverDate, parseErr := http.ParseTime(dateHeader) if parseErr != nil { @@ -331,19 +313,17 @@ func (s *httpSource) doRequest(ctx context.Context, path string) (requestResult, if absSkew < 0 { absSkew = -absSkew } - if absSkew > skewHardLimit { - return requestResult{}, fmt.Errorf("%s: clock skew %s exceeds hard limit %s (§16)", path, absSkew, skewHardLimit) + if absSkew > SkewHardLimit { + return requestResult{}, fmt.Errorf("%s: clock skew %s exceeds hard limit %s", path, absSkew, SkewHardLimit) } return requestResult{Body: b, ServerDate: serverDate, Skew: signedSkew, SkewBound: bound, Latency: latency}, nil } -// retryTransport runs fn, retrying with backoff only while it fails with a -// *TransportError — any other error is a hard error and returns immediately, -// never retried. failures counts consecutive *TransportError results; -// exceeding maxFailures gives up with a wrapped hard error (§19.3 case 1). -// The wait between attempts goes through clock.After so a test with a fake -// Clock never sleeps on real time (§22). +// retryTransport runs fn, retrying with backoff only on *TransportError — any +// other error returns immediately. failures counts consecutive *TransportError +// results; exceeding maxFailures gives up with a wrapped hard error. Waits go +// through clock.After so a fake Clock never sleeps real time. func retryTransport[T any](ctx context.Context, clock Clock, maxFailures int, backoffBase, backoffCap time.Duration, fn func() (T, error)) (T, error) { var zero T failures := 0 @@ -368,7 +348,7 @@ func retryTransport[T any](ctx context.Context, clock Clock, maxFailures int, ba } // backoffDelay is 1s base, doubling per failure, capped at maxDelay, with -// ±20% jitter (§5's filled-in value for maxSequentialFailures). +// ±20% jitter. func backoffDelay(base, maxDelay time.Duration, failureCount int) time.Duration { d := base for i := 1; i < failureCount && d < maxDelay; i++ { diff --git a/grafana-alertcheck/internal/gate/source_fake_test.go b/grafana-alertcheck/internal/gate/source_fake_test.go index 40a2a8a05..97de92f02 100644 --- a/grafana-alertcheck/internal/gate/source_fake_test.go +++ b/grafana-alertcheck/internal/gate/source_fake_test.go @@ -9,15 +9,13 @@ import ( ) // fakeClock is a manually-advanced Clock — no test in this package sleeps on -// real time (§22). It is goroutine-safe (a concurrent fleet under -race must -// not trip on the double itself), but After always fires immediately, -// regardless of the requested duration or whether Advance was ever called. -// That is sufficient here: every retry/backoff test in this phase only needs -// to avoid a real sleep. It is NOT sufficient for a test that must prove a -// wait did not fire early — e.g. a P4 scheduler test asserting Due() doesn't -// return a rule before its next-due time. Use virtualClock below for that: it -// is the clock P6's recorder-loop tests needed, and it makes a wait and the -// passage of time the same event. +// real time. It is goroutine-safe (a concurrent fleet under -race must not trip +// on the double itself), but After always fires immediately, regardless of the +// requested duration or whether Advance was ever called. That is enough for the +// retry/backoff tests, which only need to avoid a real sleep. It is NOT enough +// for a test that must prove a wait did not fire early — e.g. asserting Due() +// does not return a rule before its next-due time. Use virtualClock below for +// that: it makes a wait and the passage of time the same event. type fakeClock struct { mu sync.Mutex now time.Time @@ -113,12 +111,11 @@ type scriptedObservation struct { err error } -// fakeSource is a scripted Source with no HTTP, goroutine-safe so a phase -// that polls several rules concurrently (P6) can share one instance across -// goroutines without tripping -race. P3 through at least P5 can construct -// one directly instead of talking to HTTP; a phase that needs it to behave -// like a live server under concurrent load beyond simple locking should -// verify that assumption rather than take this comment's word for it. +// fakeSource is a scripted Source with no HTTP, goroutine-safe so a test that +// polls several rules concurrently can share one instance across goroutines +// without tripping -race. A test that needs it to behave like a live server +// under concurrent load beyond simple locking should verify that assumption +// rather than take this comment's word for it. type fakeSource struct { mu sync.Mutex diff --git a/grafana-alertcheck/internal/gate/source_test.go b/grafana-alertcheck/internal/gate/source_test.go index ba864a1a8..e5f24f46f 100644 --- a/grafana-alertcheck/internal/gate/source_test.go +++ b/grafana-alertcheck/internal/gate/source_test.go @@ -7,11 +7,12 @@ import ( "net/http" "net/http/httptest" "net/url" - "strings" "sync" "sync/atomic" "testing" "time" + + "github.com/stretchr/testify/require" ) func healthBody(version string) string { @@ -79,20 +80,18 @@ func TestCheckGrafanaVersion(t *testing.T) { {"", true, nil}, } for _, c := range cases { - err := CheckGrafanaVersion(c.version) - if c.wantErr && err == nil { - t.Errorf("CheckGrafanaVersion(%q): want error, got nil", c.version) - continue - } - if !c.wantErr && err != nil { - t.Errorf("CheckGrafanaVersion(%q): unexpected error: %v", c.version, err) - continue - } - for _, want := range c.wantContains { - if !strings.Contains(err.Error(), want) { - t.Errorf("CheckGrafanaVersion(%q): error %q does not mention %q (the plan requires naming both what was found and what is supported)", c.version, err.Error(), want) + t.Run(c.version, func(t *testing.T) { + err := CheckGrafanaVersion(c.version) + if c.wantErr { + require.Errorf(t, err, "CheckGrafanaVersion(%q)", c.version) + } else { + require.NoErrorf(t, err, "CheckGrafanaVersion(%q)", c.version) } - } + for _, want := range c.wantContains { + require.Contains(t, err.Error(), want, + "must name both what was found and what is supported") + } + }) } } @@ -102,20 +101,14 @@ func TestBackoffDelay(t *testing.T) { maxWithJitter := maxDelay + maxDelay/5 + time.Millisecond for n := 1; n <= 10; n++ { d := backoffDelay(base, maxDelay, n) - if d <= 0 { - t.Fatalf("backoffDelay(_, _, %d) = %v, want > 0", n, d) - } - if d > maxWithJitter { - t.Fatalf("backoffDelay(_, _, %d) = %v, want <= ~%v", n, d, maxWithJitter) - } + require.Positive(t, d) + require.LessOrEqual(t, d, maxWithJitter) } } func TestHTTPSource_Version_HappyPath(t *testing.T) { srv := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) { - if r.URL.Path != "/api/health" { - t.Errorf("path = %q, want /api/health", r.URL.Path) - } + require.Equal(t, "/api/health", r.URL.Path) w.Header().Set("Content-Type", "application/json") _, _ = w.Write([]byte(healthBody("13.1.0"))) })) @@ -124,20 +117,14 @@ func TestHTTPSource_Version_HappyPath(t *testing.T) { clock := newFakeClock(time.Now()) src := NewHTTPSource(srv.URL, "", clock) v, err := src.Version(context.Background()) - if err != nil { - t.Fatalf("Version(): unexpected error: %v", err) - } - if v != "13.1.0" { - t.Fatalf("Version() = %q, want 13.1.0", v) - } + require.NoError(t, err) + require.Equal(t, "13.1.0", v) } func TestHTTPSource_Version_NeverLogsToken(t *testing.T) { const secret = "super-secret-token" srv := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) { - if got := r.Header.Get("Authorization"); got != "Bearer "+secret { - t.Errorf("Authorization = %q, want Bearer %s", got, secret) - } + require.Equal(t, "Bearer "+secret, r.Header.Get("Authorization")) w.WriteHeader(http.StatusInternalServerError) })) defer srv.Close() @@ -145,12 +132,8 @@ func TestHTTPSource_Version_NeverLogsToken(t *testing.T) { clock := newFakeClock(time.Now()) src := NewHTTPSource(srv.URL, secret, clock) _, err := src.Version(context.Background()) - if err == nil { - t.Fatalf("Version(): want error, got nil") - } - if strings.Contains(err.Error(), secret) { - t.Fatalf("error %q leaks the token", err.Error()) - } + require.Error(t, err) + require.NotContains(t, err.Error(), secret) } func TestHTTPSource_RuleState_EmptyIsNotAnError(t *testing.T) { @@ -163,24 +146,16 @@ func TestHTTPSource_RuleState_EmptyIsNotAnError(t *testing.T) { clock := newFakeClock(time.Now()) src := NewHTTPSource(srv.URL, "", clock) obs, err := src.RuleState(context.Background(), "Anything") - if err != nil { - t.Fatalf("RuleState(): unexpected error: %v", err) - } - if len(obs.Rules) != 0 { - t.Fatalf("Rules = %+v, want empty (an authoritative 2xx is not a transport error)", obs.Rules) - } - if obs.GrafanaNow.IsZero() { - t.Fatalf("GrafanaNow is zero, want the response's Date header value (H4)") - } + require.NoError(t, err) + require.Empty(t, obs.Rules, "an authoritative 2xx is not a transport error") + require.False(t, obs.GrafanaNow.IsZero(), "want the response's Date header value") } func TestHTTPSource_RuleState_EscapesRuleName(t *testing.T) { var gotQuery string srv := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) { gotQuery = r.URL.RawQuery - if r.URL.Path != "/api/prometheus/grafana/api/v1/rules" { - t.Errorf("path = %q, want /api/prometheus/grafana/api/v1/rules", r.URL.Path) - } + require.Equal(t, "/api/prometheus/grafana/api/v1/rules", r.URL.Path) w.Header().Set("Content-Type", "application/json") _, _ = w.Write([]byte(emptyStateBody())) })) @@ -189,24 +164,16 @@ func TestHTTPSource_RuleState_EscapesRuleName(t *testing.T) { clock := newFakeClock(time.Now()) src := NewHTTPSource(srv.URL, "", clock) title := "[JD] No Job Proposals & More" - if _, err := src.RuleState(context.Background(), title); err != nil { - t.Fatalf("RuleState(): unexpected error: %v", err) - } - want := "rule_name=" + url.QueryEscape(title) - if gotQuery != want { - t.Fatalf("query = %q, want %q", gotQuery, want) - } + _, err := src.RuleState(context.Background(), title) + require.NoError(t, err) + require.Equal(t, "rule_name="+url.QueryEscape(title), gotQuery) } func TestHTTPSource_Definitions_HappyPath(t *testing.T) { body := readFixture(t, "ruler_rules.json") srv := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) { - if r.URL.Path != "/api/ruler/grafana/api/v1/rules" { - t.Errorf("path = %q, want /api/ruler/grafana/api/v1/rules", r.URL.Path) - } - if r.URL.RawQuery != "" { - t.Errorf("query = %q, want none — Definitions reads the ruler API unfiltered", r.URL.RawQuery) - } + require.Equal(t, "/api/ruler/grafana/api/v1/rules", r.URL.Path) + require.Empty(t, r.URL.RawQuery, "Definitions reads the ruler API unfiltered") w.Header().Set("Content-Type", "application/json") _, _ = w.Write(body) })) @@ -215,12 +182,8 @@ func TestHTTPSource_Definitions_HappyPath(t *testing.T) { clock := newFakeClock(time.Now()) src := NewHTTPSource(srv.URL, "", clock) defs, err := src.Definitions(context.Background()) - if err != nil { - t.Fatalf("Definitions(): unexpected error: %v", err) - } - if len(defs) == 0 { - t.Fatalf("Definitions(): got 0 definitions from a fixture known to have some") - } + require.NoError(t, err) + require.NotEmpty(t, defs) } func TestHTTPSource_Skew(t *testing.T) { @@ -249,14 +212,13 @@ func TestHTTPSource_Skew(t *testing.T) { }) src := NewHTTPSource(srv.URL, "", clock) _, err := src.Version(context.Background()) - if c.wantErr && err == nil { - t.Fatalf("Version(): want error, got nil") - } - if !c.wantErr && err != nil { - t.Fatalf("Version(): unexpected error: %v", err) + if c.wantErr { + require.Error(t, err) + } else { + require.NoError(t, err) } - if c.wantErr && calls.Load() != 1 { - t.Fatalf("calls = %d, want 1 — a skew hard error must never be retried", calls.Load()) + if c.wantErr { + require.Equal(t, int32(1), calls.Load(), "a skew hard error must never be retried") } }) } @@ -273,12 +235,8 @@ func TestHTTPSource_MissingDateHeader(t *testing.T) { clock := newFakeClock(time.Now()) src := NewHTTPSource(srv.URL, "", clock) _, err := src.Version(context.Background()) - if err == nil { - t.Fatalf("Version(): want error, got nil (H4: a missing Date header is a hard error)") - } - if calls.Load() != 1 { - t.Fatalf("calls = %d, want 1 — a missing Date header must never be retried", calls.Load()) - } + require.Error(t, err, "a missing Date header is a hard error") + require.Equal(t, int32(1), calls.Load(), "a missing Date header must never be retried") } func TestHTTPSource_UnparseableDateHeader(t *testing.T) { @@ -293,12 +251,8 @@ func TestHTTPSource_UnparseableDateHeader(t *testing.T) { clock := newFakeClock(time.Now()) src := NewHTTPSource(srv.URL, "", clock) _, err := src.Version(context.Background()) - if err == nil { - t.Fatalf("Version(): want error, got nil (H4: an unparseable Date header is a hard error)") - } - if calls.Load() != 1 { - t.Fatalf("calls = %d, want 1 — an unparseable Date header must never be retried", calls.Load()) - } + require.Error(t, err, "an unparseable Date header is a hard error") + require.Equal(t, int32(1), calls.Load(), "an unparseable Date header must never be retried") } // TestHTTPSource_ObservationTiming pins the arithmetic behind Observation's @@ -333,22 +287,60 @@ func TestHTTPSource_ObservationTiming(t *testing.T) { }) src := NewHTTPSource(srv.URL, "", clock) obs, err := src.RuleState(context.Background(), "Anything") - if err != nil { - t.Fatalf("RuleState(): unexpected error: %v", err) - } - if obs.Skew != c.drift { - t.Errorf("Skew = %v, want %v", obs.Skew, c.drift) - } - if obs.SkewBound != time.Second { - t.Errorf("SkewBound = %v, want 1s (RTT/2 with a 2s round trip to headers)", obs.SkewBound) - } - if obs.Latency != 4*time.Second { - t.Errorf("Latency = %v, want 4s (send through full body read, §5.2) — not just the 2s header round trip", obs.Latency) - } + require.NoError(t, err) + require.Equal(t, c.drift, obs.Skew) + require.Equal(t, time.Second, obs.SkewBound, "RTT/2 with a 2s round trip to headers") + require.Equal(t, 4*time.Second, obs.Latency, "send through full body read — not just the 2s header round trip") }) } } +// A discriminating regression for "the gate compares staleness against the +// Date header, never the runner's clock". lastEvaluation +// sits 100s behind Grafana's TRUE now (obs.GrafanaNow, from the Date header) +// — under the 120s evalStaleAfter limit — but 130s behind the RUNNER's clock. +// An implementation that leaked the runner's clock into the staleness +// comparison, instead of the Date header, would report a false violation +// here; coverage_test.go's TestProveCoverage_SkewTranslationAtWindowBoundary +// cannot catch that, because it sets LastEvaluation equal to GrafanaNow on +// every poll, making staleness zero regardless of which clock is used. +func TestHTTPSourceStalenessNeverFalsePositiveUnderSkew(t *testing.T) { + runnerNow := time.Date(2026, 1, 1, 0, 0, 0, 0, time.UTC) + clock := newFakeClock(runnerNow) + const skew = 30 * time.Second // the runner's clock reads 30s ahead of Grafana's + serverDate := runnerNow.Add(-skew) + lastEval := serverDate.Add(-100 * time.Second) + + def := Definition{UID: "r1", Title: "Rule One"} + srv := rawHTTPServer(t, func(r *http.Request) []byte { + body := fmt.Sprintf(`{"status":"success","data":{"groups":[{"file":"F","name":"G","interval":60,"rules":[`+ + `{"uid":%q,"name":%q,"state":"inactive","health":"ok","isPaused":false,"lastEvaluation":%q}`+ + `]}]}}`, def.UID, def.Title, lastEval.UTC().Format(time.RFC3339)) + return rawResponse(200, "OK", map[string]string{ + "Content-Type": "application/json", + "Date": serverDate.UTC().Format(http.TimeFormat), + }, body) + }) + + src := NewHTTPSource(srv.URL, "", clock) + obs, err := src.RuleState(context.Background(), def.Title) + require.NoError(t, err) + require.True(t, obs.GrafanaNow.Equal(serverDate), + "want the Date header, never the runner's clock") + require.Len(t, obs.Rules, 1) + + rt := newRuleTimings(30*time.Second, 60) // evalStaleAfter = 120s + from := serverDate.Add(-10 * time.Minute) + to := serverDate + polls := denseHealthyPolls(def.UID, from, to, 30*time.Second) + polls[len(polls)-1].LastEvaluation = obs.Rules[0].LastEvaluation // the real, HTTP-sourced value + sentinel := to + + res := proveCoverage(Header{StartedAt: from.Add(-time.Hour)}, polls, &sentinel, rt, def, from, to, 0) + require.False(t, res.Unobservable, + "100s behind Grafana's TRUE now is under the 120s limit — only a runner-clock leak (skewed +30s here) would push this over") +} + func TestHTTPSource_Retry_TransientRecovers(t *testing.T) { var mu sync.Mutex calls := 0 @@ -369,18 +361,12 @@ func TestHTTPSource_Retry_TransientRecovers(t *testing.T) { clock := newFakeClock(time.Now()) src := NewHTTPSource(srv.URL, "", clock) v, err := src.Version(context.Background()) - if err != nil { - t.Fatalf("Version(): unexpected error after a transient failure: %v", err) - } - if v != "13.1.0" { - t.Fatalf("Version() = %q, want 13.1.0", v) - } + require.NoError(t, err) + require.Equal(t, "13.1.0", v) mu.Lock() n := calls mu.Unlock() - if n != 3 { - t.Fatalf("calls = %d, want 3 (2 failures + 1 success)", n) - } + require.Equal(t, 3, n) } func TestHTTPSource_Retry_ExceedsLimit(t *testing.T) { @@ -394,12 +380,9 @@ func TestHTTPSource_Retry_ExceedsLimit(t *testing.T) { clock := newFakeClock(time.Now()) src := NewHTTPSource(srv.URL, "", clock) _, err := src.Version(context.Background()) - if err == nil { - t.Fatalf("Version(): want error, got nil") - } - if n := calls.Load(); n != 6 { - t.Fatalf("calls = %d, want 6 (maxSequentialFailures=5 tolerates 5, gives up on the 6th)", n) - } + require.Error(t, err) + require.Equal(t, int32(6), calls.Load(), + "maxSequentialFailures=5 tolerates 5, gives up on the 6th") assertRetryExhausted(t, err, 6) } @@ -426,18 +409,12 @@ func TestHTTPSource_RuleState_GarbageBodyRetries(t *testing.T) { clock := newFakeClock(time.Now()) src := NewHTTPSource(srv.URL, "", clock) obs, err := src.RuleState(context.Background(), "Anything") - if err != nil { - t.Fatalf("RuleState(): unexpected error after a transient garbage body: %v", err) - } - if len(obs.Rules) != 0 { - t.Fatalf("Rules = %+v, want empty", obs.Rules) - } + require.NoError(t, err) + require.Empty(t, obs.Rules) mu.Lock() n := calls mu.Unlock() - if n != 3 { - t.Fatalf("calls = %d, want 3 (2 unparseable bodies + 1 valid one)", n) - } + require.Equal(t, 3, n) } func TestHTTPSource_Definitions_GarbageBodyGivesUp(t *testing.T) { @@ -452,12 +429,9 @@ func TestHTTPSource_Definitions_GarbageBodyGivesUp(t *testing.T) { clock := newFakeClock(time.Now()) src := NewHTTPSource(srv.URL, "", clock) _, err := src.Definitions(context.Background()) - if err == nil { - t.Fatalf("Definitions(): want error, got nil") - } - if n := calls.Load(); n != 6 { - t.Fatalf("calls = %d, want 6 — a persistently unparseable 2xx body retries like any other transport failure", n) - } + require.Error(t, err) + require.Equal(t, int32(6), calls.Load(), + "a persistently unparseable 2xx body retries like any other transport failure") assertRetryExhausted(t, err, 6) } @@ -481,17 +455,11 @@ func TestHTTPSource_ResponseBodyTooLarge(t *testing.T) { clock := newFakeClock(time.Now()) src := NewHTTPSource(srv.URL, "", clock) _, err := src.Version(context.Background()) - if err == nil { - t.Fatalf("Version(): want error, got nil (an oversized body must fail loudly)") - } - if !strings.Contains(err.Error(), "exceeded") { - t.Fatalf("error %q does not name the size limit", err.Error()) - } + require.Error(t, err, "an oversized body must fail loudly") + require.Contains(t, err.Error(), "exceeded", "the error must name the size limit") // An oversized body is a stable condition, not a transient one: it must // fail hard on the first attempt, never burning retries re-reading it. - if n := calls.Load(); n != 1 { - t.Fatalf("calls = %d, want 1 — an oversized body must never be retried", n) - } + require.Equal(t, int32(1), calls.Load(), "an oversized body must never be retried") } func TestHTTPSource_NetworkFailureRetries(t *testing.T) { @@ -501,9 +469,7 @@ func TestHTTPSource_NetworkFailureRetries(t *testing.T) { clock := newFakeClock(time.Now()) src := NewHTTPSource("http://127.0.0.1:1", "", clock) _, err := src.Version(context.Background()) - if err == nil { - t.Fatalf("Version(): want error, got nil") - } + require.Error(t, err) assertRetryExhausted(t, err, 6) } @@ -511,44 +477,30 @@ func TestHTTPSource_NetworkFailureRetries(t *testing.T) { // it names how many failures it gave up after, and — the regression this // pins — it is never itself classified as a *TransportError. If it were, // something one layer up that also retries on *TransportError would treat an -// already-exhausted give-up as retryable again, the exact conflation §19.3 -// case 1 forbids. +// already-exhausted give-up as retryable again. func assertRetryExhausted(t *testing.T, err error, wantFailures int) { t.Helper() var reErr *RetryExhaustedError - if !errors.As(err, &reErr) { - t.Fatalf("error %v (%T): want a *RetryExhaustedError", err, err) - } - if reErr.Failures != wantFailures { - t.Errorf("RetryExhaustedError.Failures = %d, want %d", reErr.Failures, wantFailures) - } - if !strings.Contains(err.Error(), fmt.Sprintf("gave up after %d", wantFailures)) { - t.Errorf("error %q does not name the failure count", err.Error()) - } - if _, ok := errors.AsType[*TransportError](err); ok { - t.Fatalf("error %v (%T) is classified as *TransportError — an exhausted retry must be a terminal, non-retryable error", err, err) - } + require.ErrorAs(t, err, &reErr) + require.Equal(t, wantFailures, reErr.Failures) + require.Contains(t, err.Error(), fmt.Sprintf("gave up after %d", wantFailures)) + _, ok := errors.AsType[*TransportError](err) + require.False(t, ok, "an exhausted retry must be a terminal, non-retryable error") } func TestFakeClock(t *testing.T) { start := time.Date(2026, 1, 1, 0, 0, 0, 0, time.UTC) c := newFakeClock(start) - if !c.Now().Equal(start) { - t.Fatalf("Now() = %v, want %v", c.Now(), start) - } + require.True(t, c.Now().Equal(start)) c.Advance(5 * time.Minute) want := start.Add(5 * time.Minute) - if !c.Now().Equal(want) { - t.Fatalf("Now() after Advance = %v, want %v", c.Now(), want) - } + require.True(t, c.Now().Equal(want)) select { case fired := <-c.After(time.Hour): - if !fired.Equal(want.Add(time.Hour)) { - t.Fatalf("After fired with %v, want %v", fired, want.Add(time.Hour)) - } + require.True(t, fired.Equal(want.Add(time.Hour))) default: - t.Fatalf("After(1h) did not fire immediately") + require.Fail(t, "After(1h) did not fire immediately") } } @@ -558,37 +510,29 @@ func TestFakeSource(t *testing.T) { f.defs = []Definition{{UID: "u1", Title: "Rule One"}} ctx := context.Background() - if v, err := f.Version(ctx); err != nil || v != "13.1.0" { - t.Fatalf("Version() = (%q, %v), want (13.1.0, nil)", v, err) - } - if defs, err := f.Definitions(ctx); err != nil || len(defs) != 1 { - t.Fatalf("Definitions() = (%v, %v), want one definition", defs, err) - } + v, err := f.Version(ctx) + require.NoError(t, err) + require.Equal(t, "13.1.0", v) + defs, err := f.Definitions(ctx) + require.NoError(t, err) + require.Len(t, defs, 1) f.script("Rule One", Observation{Rules: []StateRule{{UID: "u1"}}}, nil) f.script("Rule One", Observation{}, fmt.Errorf("boom")) f.script("Rule One", Observation{Rules: nil}, nil) obs, err := f.RuleState(ctx, "Rule One") - if err != nil || len(obs.Rules) != 1 { - t.Fatalf("RuleState() call 1 = (%v, %v), want one rule, no error", obs, err) - } - if _, err := f.RuleState(ctx, "Rule One"); err == nil { - t.Fatalf("RuleState() call 2: want the scripted error, got nil") - } + require.NoError(t, err) + require.Len(t, obs.Rules, 1) + _, err = f.RuleState(ctx, "Rule One") + require.Error(t, err, "RuleState() call 2: want the scripted error, got nil") obs, err = f.RuleState(ctx, "Rule One") - if err != nil { - t.Fatalf("RuleState() call 3: unexpected error: %v", err) - } - if obs.Rules != nil { - t.Fatalf("RuleState() call 3: Rules = %v, want nil (last script entry, then repeats)", obs.Rules) - } + require.NoError(t, err) + require.Nil(t, obs.Rules, "last script entry, then repeats") obs, err = f.RuleState(ctx, "Rule One") - if err != nil || obs.Rules != nil { - t.Fatalf("RuleState() call 4: want the last scripted entry to repeat, got (%v, %v)", obs, err) - } + require.NoError(t, err) + require.Nil(t, obs.Rules) - if _, err := f.RuleState(ctx, "Unscripted Rule"); err == nil { - t.Fatalf("RuleState() for an unscripted title: want an error, got nil") - } + _, err = f.RuleState(ctx, "Unscripted Rule") + require.Error(t, err) } diff --git a/grafana-alertcheck/internal/gate/testdata/README.md b/grafana-alertcheck/internal/gate/testdata/README.md index fb437b3ae..1a6bf5877 100644 --- a/grafana-alertcheck/internal/gate/testdata/README.md +++ b/grafana-alertcheck/internal/gate/testdata/README.md @@ -1,6 +1,6 @@ # Fixture provenance -All fixtures are sanitized slices of the real Grafana 13.1.0 payloads captured next to the plan in +All fixtures are sanitized slices of real Grafana 13.1.0 payloads captured into `tmp/` (`tmp/state_all.json`, `tmp/ruler_all.json`, `tmp/health.json` — gitignored, never committed). Renames are consistent across files: the same real folder/rule keeps the same fake identity everywhere it appears (e.g. `folder0000002`/`rule0000002` is the same real paused rule in both @@ -23,45 +23,38 @@ here instead, for every fixture, for consistency. `rule0000002`/"Example Paused Rule". Unmodified: `isPaused:true`, zero `lastEvaluation`, `health:ok`, `state:inactive`, absent `alerts`/`labels`. - **state_health_error.json** — real `health:error` rule ("[JD] No Job Proposals", folder - `job-distributor`), highest priority per §22.1. Renamed to folder `ExampleService`/`folder0000003`, + `job-distributor`). Renamed to folder `ExampleService`/`folder0000003`, rule `rule0000003`/"Example No Data Source". Unmodified: `health:error`, `lastError` text, the single `Error` instance. - **state_health_nodata.json** — real `health:nodata` rule ("ARE test", folder `diegos_playground`). Renamed to folder `ExamplePlayground`/`folder0000004`, rule `rule0000004`/"Example NoData Rule". Unmodified: `health:nodata`, the single `NoData` instance. -- **state_reason_composite.json** — composite of two real instances combined under one rule for P1.2a - coverage: a real `"Normal (Error)"` instance (from a Flux-reconciliation rule; 14 of that state exist +- **state_reason_composite.json** — two real instances combined under one rule to cover composite + state parsing: a real `"Normal (Error)"` instance (from a Flux-reconciliation rule; 14 of that state exist in the capture) and a real `"Normal (NoData)"` instance (from a pod-liveness rule; 1091 of that state exist), plus one plain `"Normal"` instance for contrast. Renamed to folder `ExampleInfra`/`folder0000005`, rule `rule0000005`/"Example Composite Reasons". - **state_missing_optional.json** — derived from `state_one_instance.json`: `alerts`, `totals`, `totalsFiltered` and `labels` all removed. Must parse with `Instances=nil`, `Totals=nil`. -- **state_missing_health.json** — derived from `state_one_instance.json`: the required `health` key - removed. Must be a parse error (H1). -- **state_missing_lasteval.json** — derived from `state_one_instance.json`: the required - `lastEvaluation` key removed. Must be a parse error (H1). -- **state_missing_state.json** — derived from `state_one_instance.json`: the required rule-level - `state` key removed. Must be a parse error (H1). Closes must-error coverage for H1's four required - fields — a review pass found `health`/`lastEvaluation` covered but `state`/`interval` weren't, even - though the code already `req`'d them correctly. -- **state_missing_interval.json** — derived from `state_one_instance.json`: the required group-level - `interval` key removed. Must be a parse error (H1); same review-pass gap as above. +- **state_missing_health.json**, **state_missing_lasteval.json**, **state_missing_state.json**, + **state_missing_interval.json** — derived from `state_one_instance.json`, each with one of the four + required keys removed (`health`, `lastEvaluation`, rule-level `state`, group-level `interval`). Each + must be a parse error. - **state_missing_file.json** / **state_missing_name.json** — derived from `state_one_instance.json`: - the group-level `file`/`name` keys removed respectively. Not part of H1's four (those are `health`, - `state`, `lastEvaluation`, `interval`), but the code treats group identity as strict too, and the same - review pass flagged the gap — closed rather than deferred to a later §22 sweep since the fixture is - the same 10-line edit. + the group-level `file`/`name` keys removed respectively. Not among the four required fields above, + but the parser treats group identity as strict too. - **state_zerotime_unpaused.json** — derived from `state_one_instance.json`: `lastEvaluation` set to - the zero time while `isPaused` stays `false`. Must be a parse error (§2.3). + the zero time while `isPaused` stays `false`. Must be a parse error — only a paused rule may report + the zero time. - **state_unknown_state.json** — derived from `state_one_instance.json`: the instance state hand-edited to `"Weird (NoData)"`, a syntactically valid composite whose base isn't in the 5-value allowlist. Must - be a parse error (P1.2a). + be a parse error. - **state_only_active_instances.json** — derived from a real rule that genuinely had 1 `Alerting` + 22 `Normal` instances (`totals: {alerting:1, normal:22}`, rule `dfhp1t5pkosu8f`, folder `BCM`). `alerts[]` - trimmed to the single `Alerting` instance only, while `totals` is left **unchanged** — reproducing the - §3.2 violation shape (instance list says "only active" while totals disagrees). Renamed to folder - `ExampleTeam`/`folder0000001`, rule `rule0000006`. `ParseState` itself parses this fine; the §3.2 - verification lives in a later phase (P5/P9). + trimmed to the single `Alerting` instance only, while `totals` is left **unchanged** — the shape a + state endpoint that stopped returning normal instances would produce (the instance list says "only + active" while totals disagrees). Renamed to folder `ExampleTeam`/`folder0000001`, rule `rule0000006`. + `ParseState` itself parses this fine; `VerifyNormalInstancesVisible` is what rejects it. ## Ruler endpoint (`/api/ruler/grafana/api/v1/rules`) @@ -69,7 +62,7 @@ here instead, for every fixture, for consistency. - The real true 2-way title collision: namespace `CRE-BCM-Prod-Zone-A`, group `Gateway`, identical folder+group+title, distinct UIDs (`ffvabtvvbozcwf`/`efvabtwbxlvk0b`) — renamed to namespace `Example-Zone-A`, rules `rule0000006a`/`rule0000006b`, both titled "Example No Gateways Available". - Folder/Group/Title alone does **not** disambiguate this pair (§17, §22.2). + Folder/Group/Title alone does **not** disambiguate this pair; only `uid:` does. - The 3 real `is_paused:true` rules, renamed to `rule0000002`/`rule0000007`/`rule0000008`. `rule0000002` intentionally shares its identity (`folder0000002`) with `state_paused.json`. - A real `for:1d` rule (`afs438kjd4v7kd` → `rule0000009`). @@ -81,7 +74,7 @@ here instead, for every fixture, for consistency. Grafana represents a datasource-managed (native Prometheus-format) alerting rule. `ParseDefinitions` must classify it as `KindDatasourceManaged`, parse `Title` from `alert`, and leave `UID` empty (this shape has no uid at all — inventing one would be inventing shape) without rejecting the rule - (rejection is P3's job, only for rules a user actually named). + (rejection is `Resolve`'s job, and only for rules a user actually named). - **ruler_recording.json** — **DERIVED**, no recording rule exists in the capture (verified: 0 rules carry `grafana_alert.record`). Hand-built: a `grafana_alert` block with a `record` sub-object but deliberately *without* `no_data_state`/`exec_err_state`/`is_paused`/`intervalSeconds`/`namespace_uid` diff --git a/grafana-alertcheck/internal/gate/watch.go b/grafana-alertcheck/internal/gate/watch.go index f0f24f247..74dd3945b 100644 --- a/grafana-alertcheck/internal/gate/watch.go +++ b/grafana-alertcheck/internal/gate/watch.go @@ -14,25 +14,19 @@ import ( ) // DaemonChildFlag is the hidden flag the parent passes when it re-execs itself -// as the detached recorder (§4.4). It is deliberately absent from the CLI's +// as the detached recorder. It is deliberately absent from the CLI's // usage text: an operator never types it, and a child started by hand against // a log no parent prepared fails immediately on the header read. const DaemonChildFlag = "--daemon-child" // ReadyFDFlag names the inherited descriptor the child reports readiness on. -// The parent passes the write end of a pipe as descriptor 3 and waits for one -// byte, so "the recorder is running" is a POSITIVE signal from the child -// itself — it has read the header, taken the log's flock and entered its poll -// loop — and not an assumption drawn from surviving a timer. A timer cannot -// tell a healthy child from one that is about to die on a slow runner, and -// getting that wrong means watch returns success over a recording that never -// happened (§4.3). +// "Ready" is a POSITIVE byte from the child (header read, flock taken, in its +// poll loop), never a timer heuristic — a timer can't tell a healthy child from +// one about to die on a slow runner. const ReadyFDFlag = "--ready-fd" -// childReadyTimeout bounds that wait. Everything before the signal is local — -// fork, exec, one read of a log holding a header and a handful of polls — so -// the real figure is milliseconds; this is loose enough for a badly overloaded -// runner and still fails closed rather than hanging the pipeline. +// childReadyTimeout bounds that wait. Everything before the signal is local, so +// it is loose enough for an overloaded runner yet still fails closed. const childReadyTimeout = 30 * time.Second // daemonLogTailBytes bounds how much of a dead child's output the parent @@ -41,32 +35,26 @@ const daemonLogTailBytes = 4096 // WatchConfig is the record step's whole input. // -// It has no To field and must never gain one: watch writes the stopped -// sentinel with its OWN stop time and makes no comparison against `to`, which -// only check knows (§4.5). Passing `to` here would give two components an -// opinion about the same comparison, and the recorder's opinion is the one -// that cannot be trusted — it exits before the grace it would have to wait for. +// It has no To field (and must never gain one): watch writes the sentinel with +// its OWN stop time and makes no `to` comparison — only check knows `to`, and +// the recorder exits before the grace it would have to wait for. // -// It has no States field either, and watch has no --states flag: recording is -// deliberately unfiltered. The reduction keeps every non-normal instance and -// the transition markers key off the same predicate, so neither consults -// States. The payoff is real — because the log is raw evidence, one recording -// can be re-classified under different --states without re-recording — and the -// Header carries no States field for the same reason. +// It has no States field either: recording is unfiltered, so the same raw log +// can be re-classified under different --states without re-recording. type WatchConfig struct { // URL and Token are the connection details. The CLI reads both from the - // environment and never from a flag (§20.2); Token is never logged and - // never enters an error string. + // environment and never from a flag; Token is never logged and never + // enters an error string. URL, Token string - // Alerts are the operator-supplied names, one per line, in any of §17's - // forms. Empty lines are discarded by Resolve. + // Alerts are the operator-supplied names, one per line, in any of the forms + // Resolve accepts. Empty lines are discarded by Resolve. Alerts []string Folder string // Out is the JSONL log path. PidFile and DaemonLog default to // .pid and .daemon.log — the same convention check uses to find - // the recorder it must stop (P9), so nothing has to be wired by hand. + // the recorder it must stop, so nothing has to be wired by hand. Out string PidFile string DaemonLog string @@ -77,11 +65,10 @@ type WatchConfig struct { Until time.Time // PollEvery is the --poll-interval override, used verbatim for every rule - // and never clamped (§5.1). Zero means each rule polls at half its own - // evaluation interval. Whatever this resolves to is written into the header - // as the cadence actually used, and that header value — never a - // re-derivation from the definitions — is what check derives maxGap from - // (P5, "two authorities"). + // and never clamped. Zero means each rule polls at half its own evaluation + // interval. Whatever this resolves to is written into the header as the + // cadence actually used, and that header value — never a re-derivation from + // the definitions — is what check derives maxGap from. PollEvery time.Duration Concurrency int @@ -90,7 +77,7 @@ type WatchConfig struct { // Notes is where the parent prints what an operator has to see before the // deploy step runs: resolve notes, the cadence per rule, the rules it will // not wait for. nil discards them. The library prints nothing else — the - // CLI owns presentation (§20.2). + // CLI owns presentation. Notes io.Writer } @@ -138,20 +125,11 @@ func (cfg WatchConfig) validate() error { return nil } -// Watch is the record step's parent process (§4.3). It returns only once the -// window is genuinely being recorded: -// -// version gate -> resolve definitions and names -> derive timings -> -// open the log and write the header -> ONE observation of every non-skipped -// rule -> verify §3.2 -> check the schedule budget -> detach the child -> -// wait for the child to report that it is recording -> write the pidfile -> -// return. -// -// The first-observation wait is not a convenience. Returning before it would -// leave the deploy inside [from, first_poll] with no evidence — the exact -// blind interval the two-phase model exists to remove — and it is also what -// surfaces auth, name-resolution and parse failures BEFORE deploy.sh runs -// rather than ten minutes later. +// Watch is the record step's parent process, returning only once the window is +// genuinely being recorded (version gate, resolve, write header, one +// observation per rule, budget check, then detach and await the child's +// readiness). The first-observation wait is what surfaces auth, name-resolution +// and parse failures before deploy.sh runs, rather than ten minutes later. func Watch(ctx context.Context, cfg WatchConfig) error { cfg = cfg.withDefaults() if err := cfg.validate(); err != nil { @@ -165,8 +143,8 @@ func Watch(ctx context.Context, cfg WatchConfig) error { } // Hand the log over with Close, never Stop: a sentinel here would tell - // check the recording ended before the child had even started (§4.5). - // Closing also releases the flock the child is about to take. + // check the recording ended before the child had even started. Closing + // also releases the flock the child is about to take. if err := prep.writer.Close(); err != nil { return err } @@ -181,8 +159,8 @@ func Watch(ctx context.Context, cfg WatchConfig) error { // The PARENT writes the pidfile, not the child: check must find the pid the // instant Watch returns, and a child writing its own would race the very - // next step of the pipeline. A deviation from P6's argv list, and the - // reason the child is never given --pidfile at all. + // next step of the pipeline. That is why the child is never given + // --pidfile at all. // // It is written only once the child has reported ready, so no path through // this function leaves a pidfile naming a process that is not recording. @@ -295,8 +273,8 @@ type preparedWatch struct { // prepareWatch is everything the parent does before it detaches. It takes a // Source rather than building one so the paused-rule, first-observation, -// §3.2 and budget behaviours are all testable with a scripted fake — only the -// process spawning needs a real binary. +// instance-visibility and budget behaviours are all testable with a scripted +// fake — only the process spawning needs a real binary. func prepareWatch(ctx context.Context, cfg WatchConfig, src Source) (*preparedWatch, error) { version, err := src.Version(ctx) if err != nil { @@ -325,7 +303,7 @@ func prepareWatch(ctx context.Context, cfg WatchConfig, src Source) (*preparedWa for _, d := range resolved { // A cadence of zero would make the child spin: every rule is due the // instant it was marked. It also cannot be written into the header, - // where check requires a positive value to derive maxGap from (P5). + // where check requires a positive value to derive maxGap from. if rt[d.UID].pollEvery <= 0 { return nil, fmt.Errorf("rule %q (%s) reports intervalSeconds=%d: there is no poll cadence to record at", d.Title, d.UID, d.IntervalSeconds) @@ -363,72 +341,35 @@ func openRecording(ctx context.Context, cfg WatchConfig, src Source, writer *Wri return nil, err } - // A rule whose DEFINITION says is_paused is skipped (§12): it is not - // waited for, not scheduled and never polled. Waiting for one either hangs - // forever or errors before the deploy (§4.3), and recording polls for it - // would report an in-window pause (coverage check 7) for a rule that was - // already paused when the window opened — turning §12's exit 1 into an - // exit 2. The header still names it, with is_paused true, so check reports - // it as skipped from the definitions. + // A rule whose definition says is_paused is skipped (never waited for or + // polled); polling it would report an in-window pause (check 7) for a rule + // already paused at the open. The header still names it, so check reports + // it skipped from the definitions. var active []Definition activeTimings := make(map[string]ruleTimings, len(resolved)) - titles := make(map[string]string, len(resolved)) for _, d := range resolved { if d.IsPaused { - fmt.Fprintf(cfg.Notes, "note: rule %q (%s) is paused: recorded as skipped, not waited for (§4.3)\n", d.Title, d.UID) + fmt.Fprintf(cfg.Notes, "note: rule %q (%s) is paused: recorded as skipped, not waited for\n", d.Title, d.UID) continue } active = append(active, d) activeTimings[d.UID] = rt[d.UID] - titles[d.UID] = d.Title fmt.Fprintf(cfg.Notes, "recording %q (%s) every %s (maxGap %s)\n", d.Title, d.UID, rt[d.UID].pollEvery, rt[d.UID].maxGap) } - uids := make([]string, 0, len(active)) - for _, d := range active { - uids = append(uids, d.UID) - } - observed, err := observeAll(ctx, src, titles, uids, cfg.Concurrency) + polls, measured, err := firstObservations(ctx, src, active, NewReducer(), cfg.Concurrency, cfg.Notes) if err != nil { return nil, err } - - // Verify §3.2 before anything downstream relies on it: if the state - // endpoint ever stops returning normal instances, the reduction's "keep - // the non-normal ones" silently becomes "keep everything it happened to - // send" and the transition markers lose their ground truth. - for _, d := range active { - if err := VerifyNormalInstancesVisible(observed[d.UID].Rules); err != nil { - return nil, err - } - } - - // One poll record per rule, in resolve order so the log is byte-stable for - // a given set of observations. These ARE the log's first heartbeats: they - // predate the deploy step, which is the whole point of §4.3. - reducer := NewReducer() - measured := make(map[string]time.Duration, len(active)) - for _, d := range active { - obs := observed[d.UID] - measured[d.UID] = obs.Latency - poll := reducer.Reduce(d.UID, obs) - if !poll.Found { - // Authoritative, not transient (P2 already retried transport - // failures): the rule resolved in the ruler API but the state - // endpoint does not serve it. Recorded as Found=false, which P7 - // turns into unobservable — a note rather than an error here, - // because the state endpoint can lag a freshly created rule and - // check re-resolves and fails closed either way. - fmt.Fprintf(cfg.Notes, "warning: rule %q (%s) is absent from the state endpoint; recorded as not found\n", d.Title, d.UID) - } - if err := writer.WritePoll(poll); err != nil { + for _, p := range polls { + if err := writer.WritePoll(p); err != nil { return nil, err } } - // Budget last, on the latencies just measured — never on a fixed estimate - // (§5.2). Only the active rules count: a skipped rule is never polled and - // consumes none of the capacity. + // Budget last, on the latencies just measured — never on a fixed estimate. + // Only the active rules count: a skipped rule is never polled and consumes + // none of the capacity. if err := CheckBudget(activeTimings, measured, cfg.Concurrency); err != nil { return nil, err } @@ -438,9 +379,9 @@ func openRecording(ctx context.Context, cfg WatchConfig, src Source, writer *Wri // loggedRules snapshots the resolved definitions into the header's rule list. // Every field but PollEverySeconds is forensic — a resolve-time snapshot that -// makes an uploaded log self-describing (§21.3) — while PollEverySeconds is +// makes an uploaded log self-describing — while PollEverySeconds is // load-bearing: it is the cadence this recording actually used, and check -// derives maxGap from it rather than from the definitions (P5). +// derives maxGap from it rather than from the definitions. func loggedRules(defs []Definition, rt map[string]ruleTimings) []LoggedRule { out := make([]LoggedRule, 0, len(defs)) for _, d := range defs { @@ -460,14 +401,62 @@ func loggedRules(defs []Definition, rt map[string]ruleTimings) []LoggedRule { return out } +// firstObservations takes one observation of every active rule, verifies normal +// instances are visible in those very responses, and reduces each into the +// window's first heartbeat, plus measured latency (the only honest budget +// input). Watches parent and single-step check both share it. Polls return in +// `active` order, so a log written from them is byte-stable. +func firstObservations(ctx context.Context, src Source, active []Definition, reducer *Reducer, + concurrency int, notes io.Writer) ([]Poll, map[string]time.Duration, error) { + + titles := make(map[string]string, len(active)) + uids := make([]string, 0, len(active)) + for _, d := range active { + titles[d.UID] = d.Title + uids = append(uids, d.UID) + } + + observed, err := observeAll(ctx, src, titles, uids, concurrency) + if err != nil { + return nil, nil, err + } + + // Verify this before anything downstream relies on it: if the state + // endpoint ever stops returning normal instances, the reduction's "keep + // the non-normal ones" silently becomes "keep everything it happened to + // send" and the transition markers lose their ground truth. + for _, d := range active { + if err := VerifyNormalInstancesVisible(observed[d.UID].Rules); err != nil { + return nil, nil, err + } + } + + polls := make([]Poll, 0, len(active)) + measured := make(map[string]time.Duration, len(active)) + for _, d := range active { + obs := observed[d.UID] + measured[d.UID] = obs.Latency + poll := reducer.Reduce(d.UID, obs) + if !poll.Found { + // Authoritative, not transient (the transport already retried + // every transient failure): the rule resolved in the ruler API but + // the state endpoint does not serve it. Recorded as Found=false, + // which the coverage proof turns into unobservable — a note rather + // than an error here, because the state endpoint can lag a freshly + // created rule and the coverage proof fails closed either way. + fmt.Fprintf(notes, "warning: rule %q (%s) is absent from the state endpoint; recorded as not found\n", d.Title, d.UID) + } + polls = append(polls, poll) + } + return polls, measured, nil +} + // observeAll polls every rule in uids concurrently, bounded by concurrency, -// and returns one Observation per rule that answered. Every rule is polled by -// TITLE (the ?rule_name= filter, §2.8) and selected out of the response by -// UID (§14.5) — a filtered response can carry several rules sharing one title. -// -// It returns the successful observations alongside the first error in UID -// order, so a caller that wants to keep the good heartbeats can, and the error -// message is the same on every run. +// returning one Observation per rule that answered. Each rule is polled by +// TITLE (the ?rule_name= filter is a title filter) and selected by UID — a +// filtered response can carry several rules sharing a title. Returns the +// successes alongside the first error in UID order, so a caller can keep the +// good heartbeats. func observeAll(ctx context.Context, src Source, titles map[string]string, uids []string, concurrency int) (map[string]Observation, error) { if concurrency < 1 { concurrency = 1 @@ -511,9 +500,9 @@ func observeAll(ctx context.Context, src Source, titles map[string]string, uids // parent already wrote — one source of truth, no parent/child drift, and it // exercises ReadLog's header path — and the connection details come from the // inherited environment. Only the run facts the header does not carry travel -// in argv (§4.4). +// in argv. type DaemonChildConfig struct { - URL, Token string // from the inherited environment, never from argv (§20.2) + URL, Token string // from the inherited environment, never from argv Out string Until time.Time Concurrency int @@ -542,7 +531,7 @@ func RunDaemonChild(ctx context.Context, cfg DaemonChildConfig) error { } // Safe to read: the parent closed its writer before spawning this process, - // and no other writer can hold the log's flock (§4.4 step 4). + // and no other writer can hold the log's flock. header, polls, sentinel, err := ReadLog(cfg.Out) if err != nil { return err @@ -550,7 +539,7 @@ func RunDaemonChild(ctx context.Context, cfg DaemonChildConfig) error { if sentinel != nil { return fmt.Errorf("log %s already carries a stopped sentinel: another recorder finished it", cfg.Out) } - // The header's URL is the log's identity (§19.1 step 3). Checking it here + // The header's URL is the log's identity. Checking it here // catches a child that inherited an environment pointing somewhere else, // before it appends a single poll from the wrong Grafana. if header.URL != cfg.URL { @@ -570,7 +559,7 @@ func RunDaemonChild(ctx context.Context, cfg DaemonChildConfig) error { reducer := NewReducer() reducer.seedFrom(polls) - // SIGTERM is how check stops the recorder (§4.4 step 1); SIGINT is the + // SIGTERM is how check stops the recorder; SIGINT is the // same request from a human at a terminal. Both are clean stops, so both // end with a sentinel. Registered before the readiness report, so a signal // arriving the moment the parent unblocks is already handled. @@ -615,15 +604,10 @@ func reportReady(fd int) error { } // childSchedule derives what the child polls, and how often, from the header -// alone. The cadence comes from PollEverySeconds — the cadence the recording -// actually uses — and is never re-derived from the rule's evaluation interval: -// that is P5's "two authorities", and getting it wrong is fail-open in the -// faster-override direction. Paused rules are excluded here for the same -// reason the parent never polls them (§4.3, §12). -// -// It returns cadences and nothing else. maxGap, healthGrace and evalStaleAfter -// are coverage thresholds applied by the pure layer at classification time, so -// the recorder must not carry them: it would only be able to misuse them. +// alone. Cadence comes from PollEverySeconds (the cadence actually used), never +// re-derived from the evaluation interval; paused rules are excluded. It +// returns cadences only — the recorder must not carry coverage thresholds it +// has no business applying. func childSchedule(h Header) (titles map[string]string, cadence map[string]time.Duration, err error) { titles = make(map[string]string, len(h.Rules)) cadence = make(map[string]time.Duration, len(h.Rules)) @@ -646,7 +630,7 @@ func childSchedule(h Header) (titles map[string]string, cadence map[string]time. // watchLoopConfig is the child's working state: what to poll, how often, and // where to append it. There is no threshold in here and no policy — the child -// records and classifies nothing (H5). +// records and classifies nothing. type watchLoopConfig struct { Src Source Writer *Writer @@ -658,15 +642,13 @@ type watchLoopConfig struct { Clock Clock } -// watchLoop is the child's whole working life: poll the rules that are due, -// reduce each observation to one poll record, append it, and — on a clean stop -// only — finish the log with the stopped sentinel. +// watchLoop is the child's whole working life: poll due rules, reduce each +// observation to a poll record, append it, and — on a clean stop only — finish +// the log with the stopped sentinel. // -// The sentinel policy is the load-bearing part. A clean stop (a signal, or -// Until) writes it; a hard error does NOT. A recorder that died must look -// exactly like a coverage gap to check, because it is one (§4.5) — writing a -// sentinel on the way out of a failure would hand check a "recording finished" -// claim about a window that stopped being observed. +// The sentinel policy is load-bearing: a clean stop (signal or Until) writes +// it; a hard error does NOT. A recorder that died must look exactly like a +// coverage gap to check, because it is one. func watchLoop(ctx context.Context, cfg watchLoopConfig) error { sched := NewScheduler(cfg.Cadence, cfg.Clock.Now()) @@ -709,8 +691,8 @@ func watchLoop(ctx context.Context, cfg watchLoopConfig) error { pollErr := cfg.pollBatch(ctx, due) if ctx.Err() != nil { // Signalled while a poll was in flight. The aborted poll's error is - // not a recorder failure, and a clean stop wins over it (§4.4 step - // 1: finish the in-flight write, then the sentinel). + // not a recorder failure, and a clean stop wins over it: finish the + // in-flight write, then the sentinel. return cfg.Writer.Stop() } if pollErr != nil { @@ -754,7 +736,7 @@ func untilNextPoll(sched *Scheduler, until, now time.Time) (time.Duration, bool) return max(next.Sub(now), 0), true } -// writePidFile records the child's pid where check looks for it (P9's +// writePidFile records the child's pid where check looks for it (its // --pidfile, default .pid). The format is the decimal pid and a newline, // so `kill $(cat log.jsonl.pid)` works and ReadPidFile stays trivial. func writePidFile(path string, pid int) error { @@ -765,7 +747,7 @@ func writePidFile(path string, pid int) error { } // ReadPidFile is the other side of that contract: the pid of the recorder -// check must stop before it may read the log (§4.4 steps 1-4). +// check must stop before it may read the log. func ReadPidFile(path string) (int, error) { b, err := os.ReadFile(path) if err != nil { diff --git a/grafana-alertcheck/internal/gate/watch_daemon_test.go b/grafana-alertcheck/internal/gate/watch_daemon_test.go index 35e9c424a..73a6f3d7a 100644 --- a/grafana-alertcheck/internal/gate/watch_daemon_test.go +++ b/grafana-alertcheck/internal/gate/watch_daemon_test.go @@ -7,6 +7,7 @@ import ( "net/http" "net/http/httptest" "os" + "os/signal" "path/filepath" "slices" "strconv" @@ -14,23 +15,61 @@ import ( "syscall" "testing" "time" + + "github.com/stretchr/testify/require" ) // TestMain doubles this test binary as the detached recorder. Watch spawns // os.Executable(), which under `go test` is this binary, so the one integration // test below exercises the real thing — a real fork/exec, a real setsid, a real // inherited environment, a real SIGTERM — with this function standing in for -// the CLI's `watch --daemon-child` dispatch, which lands in P10. +// the CLI's `watch --daemon-child` dispatch. func TestMain(m *testing.M) { + if path := os.Getenv(lockHolderEnv); path != "" { + os.Exit(runTestLockHolder(path)) + } if slices.Contains(os.Args, DaemonChildFlag) { os.Exit(runTestDaemonChild(os.Args[1:])) } os.Exit(m.Run()) } +// lockHolderEnv turns this test binary into a stand-in recorder that holds the +// log's flock and refuses to die: a process check's stop protocol must wait +// for and, on a timeout, refuse to read around. +// +// It has to be a real second process. flock is what stopRecorder probes, and +// there is no flock(1) on darwin, so a shell one-liner cannot take the lock — +// while a lock taken in the test process itself would be granted to the probe +// on some platforms and prove nothing. +const lockHolderEnv = "GRAFANA_ALERTCHECK_TEST_LOCK_HOLDER" + +// runTestLockHolder takes the log's exclusive lock, reports that it has it on +// stdout, ignores SIGTERM, and waits to be killed. The report is what lets the +// test start only once the lock is genuinely held, rather than racing it. +func runTestLockHolder(path string) int { + signal.Ignore(syscall.SIGTERM) + + f, err := os.OpenFile(path, os.O_APPEND|os.O_CREATE|os.O_WRONLY, 0o644) + if err != nil { + fmt.Fprintln(os.Stderr, err) + return 1 + } + if err := lockExclusive(f); err != nil { + fmt.Fprintln(os.Stderr, err) + return 1 + } + fmt.Println("locked") + + // Long enough to outlive any test that starts it; the test kills it, and + // SIGKILL is not ignorable. + time.Sleep(5 * time.Minute) + return 0 +} + // runTestDaemonChild parses the child argv childArgs() writes, and reads the -// connection details from the environment — never from argv (§20.2). P10's -// `watch` FlagSet does the same four flags. +// connection details from the environment — never from argv. The CLI's `watch` +// FlagSet does the same four flags. func runTestDaemonChild(args []string) int { cfg := DaemonChildConfig{ URL: os.Getenv("GRAFANA_URL"), @@ -79,7 +118,7 @@ func runTestDaemonChild(args []string) int { } // testBearerToken is what every request to grafanaTestServer must carry. The -// child never receives it in argv (§20.2), so a request that arrives +// child never receives it in argv, so a request that arrives // authenticated is proof that the token reached the detached process through // the inherited environment — and a 401 is what a test sees if that ever // breaks. @@ -110,7 +149,7 @@ func grafanaTestServer(t *testing.T) *httptest.Server { _, _ = w.Write(ruler) case strings.HasPrefix(r.URL.Path, "/api/prometheus/"): if r.URL.Query().Get("rule_name") == "" { - // §2.8: the gate must never read the state endpoint unfiltered. + // The gate must never read the state endpoint unfiltered. http.Error(w, "unfiltered state read", http.StatusBadRequest) return } @@ -126,37 +165,25 @@ func grafanaTestServer(t *testing.T) *httptest.Server { func patchedStateBody(t *testing.T) []byte { t.Helper() var body map[string]any - if err := json.Unmarshal(readFixture(t, "state_one_instance.json"), &body); err != nil { - t.Fatalf("unmarshal state fixture: %v", err) - } + require.NoError(t, json.Unmarshal(readFixture(t, "state_one_instance.json"), &body)) data, ok := body["data"].(map[string]any) - if !ok { - t.Fatal("state fixture: no data object") - } + require.True(t, ok, "state fixture: no data object") groups, ok := data["groups"].([]any) - if !ok || len(groups) == 0 { - t.Fatal("state fixture: no groups") - } + require.True(t, ok, "state fixture: no groups") + require.NotEmpty(t, groups, "state fixture: no groups") group, ok := groups[0].(map[string]any) - if !ok { - t.Fatal("state fixture: group 0 is not an object") - } + require.True(t, ok, "state fixture: group 0 is not an object") rules, ok := group["rules"].([]any) - if !ok || len(rules) == 0 { - t.Fatal("state fixture: group 0 has no rules") - } + require.True(t, ok, "state fixture: group 0 has no rules") + require.NotEmpty(t, rules, "state fixture: group 0 has no rules") rule, ok := rules[0].(map[string]any) - if !ok { - t.Fatal("state fixture: rule 0 is not an object") - } + require.True(t, ok, "state fixture: rule 0 is not an object") rule["uid"] = watchActiveUID rule["name"] = watchActiveTitle rule["lastEvaluation"] = time.Now().UTC().Format(time.RFC3339Nano) b, err := json.Marshal(body) - if err != nil { - t.Fatalf("marshal patched state fixture: %v", err) - } + require.NoError(t, err) return b } @@ -172,17 +199,17 @@ func waitFor(t *testing.T, what string, timeout time.Duration, cond func() bool) } time.Sleep(20 * time.Millisecond) } - t.Fatalf("timed out after %s waiting for %s", timeout, what) + require.Fail(t, fmt.Sprintf("timed out after %s waiting for %s", timeout, what)) } -// TestWatchSpawnsADetachedRecorder is P6's one integration test: everything -// from the version gate to the sentinel, through a real detached process. +// The one watch integration test: everything from the version gate to the +// sentinel, through a real detached process. // // It asserts the four things only a real spawn can show — the pidfile points // at a live process, that process is in its own session (setsid, not a bare // `&`), it keeps appending after Watch returned, and SIGTERM makes it finish -// the log in the §4.4 order — and it uses a 200ms --poll-interval to do it in -// about a second, which also exercises the unclamped-override path (§5.1). +// the log in the stop order — and it uses a 200ms --poll-interval to do it in +// about a second, which also exercises the unclamped-override path. func TestWatchSpawnsADetachedRecorder(t *testing.T) { srv := grafanaTestServer(t) t.Setenv("GRAFANA_URL", srv.URL) @@ -200,9 +227,7 @@ func TestWatchSpawnsADetachedRecorder(t *testing.T) { Notes: ¬es, } - if err := Watch(context.Background(), cfg); err != nil { - t.Fatalf("Watch: %v\nnotes:\n%s", err, notes.String()) - } + require.NoError(t, Watch(context.Background(), cfg)) t.Cleanup(func() { if t.Failed() { t.Logf("notes:\n%s", notes.String()) @@ -211,22 +236,16 @@ func TestWatchSpawnsADetachedRecorder(t *testing.T) { }) pid, err := ReadPidFile(out + ".pid") - if err != nil { - t.Fatalf("ReadPidFile: %v", err) - } - if err := syscall.Kill(pid, 0); err != nil { - t.Fatalf("recorder pid %d is not running right after Watch returned: %v", pid, err) - } + require.NoError(t, err) + require.NoError(t, syscall.Kill(pid, 0), "recorder pid %d is not running right after Watch returned", pid) // Setsid, not a bare `&`: a session leader's process group id is its own // pid. Without this the child would still share the parent's process group // and die with the step that started it. - if pgid, err := syscall.Getpgid(pid); err != nil { - t.Errorf("Getpgid(%d): %v", pid, err) - } else if pgid != pid { - t.Errorf("recorder pgid = %d, want %d: it did not get its own session", pgid, pid) - } + pgid, err := syscall.Getpgid(pid) + require.NoError(t, err) + require.Equal(t, pid, pgid, "it did not get its own session") - // The parent already wrote the first heartbeat before it returned (§4.3); + // The parent already wrote the first heartbeat before it returned; // these later ones prove the detached child is the one appending now. waitFor(t, "the detached recorder to append its own polls", 10*time.Second, func() bool { _, polls, _, err := ReadLog(out) @@ -234,35 +253,24 @@ func TestWatchSpawnsADetachedRecorder(t *testing.T) { }) // Stop it exactly the way check does. - if err := syscall.Kill(pid, syscall.SIGTERM); err != nil { - t.Fatalf("SIGTERM %d: %v", pid, err) - } + require.NoError(t, syscall.Kill(pid, syscall.SIGTERM)) waitFor(t, "the stopped sentinel", 10*time.Second, func() bool { _, _, sentinel, err := ReadLog(out) return err == nil && sentinel != nil }) header, polls, sentinel, err := ReadLog(out) - if err != nil { - t.Fatalf("ReadLog: %v", err) - } - if header.URL != srv.URL || header.GrafanaVersion != "13.1.0" { - t.Errorf("header identity = %q/%q, want %q/13.1.0", header.URL, header.GrafanaVersion, srv.URL) - } - if len(header.Rules) != 1 || header.Rules[0].PollEverySeconds != 0.2 { - t.Errorf("header rules = %+v, want one rule recorded at 0.2s", header.Rules) - } + require.NoError(t, err) + require.Equal(t, srv.URL, header.URL) + require.Equal(t, "13.1.0", header.GrafanaVersion) + require.Len(t, header.Rules, 1) + require.Equal(t, float64(0.2), header.Rules[0].PollEverySeconds) for i, p := range polls { - if p.RuleUID != watchActiveUID || !p.Found { - t.Fatalf("poll %d = %+v, want a found observation of %s", i, p, watchActiveUID) - } - if p.GrafanaNow.IsZero() { - t.Fatalf("poll %d has no grafana_now; H4 needs the Date header of its own response", i) - } - } - if sentinel.Before(header.StartedAt) { - t.Errorf("sentinel at %s precedes the record start %s", sentinel, header.StartedAt) + require.Equalf(t, watchActiveUID, p.RuleUID, "poll %d", i) + require.Truef(t, p.Found, "poll %d", i) + require.Falsef(t, p.GrafanaNow.IsZero(), "poll %d has no grafana_now; every poll needs the Date header of its own response", i) } + require.False(t, sentinel.Before(header.StartedAt), "sentinel precedes the record start") waitFor(t, "the recorder to exit", 10*time.Second, func() bool { return syscall.Kill(pid, 0) != nil @@ -280,27 +288,17 @@ func TestDaemonChildRejectsAnAlreadyFinishedLog(t *testing.T) { clock := newFakeClock(testNow) w, err := NewWriter(path, clock) - if err != nil { - t.Fatalf("NewWriter: %v", err) - } - if err := w.WriteHeader(testHeader()); err != nil { - t.Fatalf("WriteHeader: %v", err) - } - if err := w.Stop(); err != nil { - t.Fatalf("Stop: %v", err) - } + require.NoError(t, err) + require.NoError(t, w.WriteHeader(testHeader())) + require.NoError(t, w.Stop()) err = RunDaemonChild(context.Background(), DaemonChildConfig{ URL: testHeader().URL, Out: path, Clock: clock, }) - if err == nil { - t.Fatal("RunDaemonChild: no error against a log that already carries a stopped sentinel") - } - if !strings.Contains(err.Error(), "sentinel") { - t.Errorf("error = %v, want it to name the stopped sentinel", err) - } + require.Error(t, err, "no error against a log that already carries a stopped sentinel") + require.Contains(t, err.Error(), "sentinel") } // TestWatchFailsWhenTheChildCannotStartRecording is the other half of the @@ -328,13 +326,9 @@ func TestWatchFailsWhenTheChildCannotStartRecording(t *testing.T) { Concurrency: 2, Notes: ¬es, }) - if err == nil { - t.Fatal("Watch: no error, but the child could never have started recording") - } - if !strings.Contains(err.Error(), "records url") { - t.Errorf("error does not quote the child's own reason:\n%v", err) - } - if _, statErr := os.Stat(out + ".pid"); !os.IsNotExist(statErr) { - t.Errorf("a pidfile survived a failed detach (%v); pids are reused, so the next step would signal a stranger", statErr) - } + require.Error(t, err, "the child could never have started recording") + require.Contains(t, err.Error(), "records url") + _, statErr := os.Stat(out + ".pid") + require.True(t, os.IsNotExist(statErr), + "a pidfile survived a failed detach; pids are reused, so the next step would signal a stranger") } diff --git a/grafana-alertcheck/internal/gate/watch_process.go b/grafana-alertcheck/internal/gate/watch_process.go index 629606188..169a4f65a 100644 --- a/grafana-alertcheck/internal/gate/watch_process.go +++ b/grafana-alertcheck/internal/gate/watch_process.go @@ -23,7 +23,7 @@ type detachedChild struct { logOffset int64 } -// spawnChild re-execs this binary as the detached recorder (§4.4). A trailing +// spawnChild re-execs this binary as the detached recorder. A trailing // `&` is NOT sufficient: the child would keep the parent's session and process // group, so it would still take the terminal's signals and, on a runner, die // with the step that started it. Setsid gives it a new session AND a new @@ -52,7 +52,7 @@ func spawnChild(cfg WatchConfig) (detachedChild, error) { logOffset = info.Size() } - // The readiness pipe (§ReadyFDFlag): the child gets the write end as + // The readiness pipe: the child gets the write end as // descriptor 3 and reports on it once it holds the log and is polling. readyRead, readyWrite, err := os.Pipe() if err != nil { @@ -64,9 +64,9 @@ func spawnChild(cfg WatchConfig) (detachedChild, error) { cmd.Stdout = logFile cmd.Stderr = logFile cmd.ExtraFiles = []*os.File{readyWrite} // descriptor 3 in the child - // The environment is how the connection details reach the child (§20.2): - // the token must never appear in argv, where it would land in the process - // table and in CI logs. + // The environment is how the connection details reach the child: the token + // must never appear in argv, where it would land in the process table and + // in CI logs. cmd.Env = os.Environ() cmd.SysProcAttr = &syscall.SysProcAttr{Setsid: true} diff --git a/grafana-alertcheck/internal/gate/watch_test.go b/grafana-alertcheck/internal/gate/watch_test.go index fa2af889a..41ee3b09b 100644 --- a/grafana-alertcheck/internal/gate/watch_test.go +++ b/grafana-alertcheck/internal/gate/watch_test.go @@ -6,11 +6,12 @@ import ( "fmt" "os" "path/filepath" - "slices" "strings" "sync" "testing" "time" + + "github.com/stretchr/testify/require" ) // The two fixture rules every prepareWatch test below uses: one live, one @@ -72,12 +73,8 @@ func testStateRule(uid, title string, interval time.Duration, grafanaNow time.Ti func newLoopWriter(t *testing.T, path string, clock Clock) *Writer { t.Helper() w, err := NewWriter(path, clock) - if err != nil { - t.Fatalf("NewWriter: %v", err) - } - if err := w.WriteHeader(testHeader()); err != nil { - t.Fatalf("WriteHeader: %v", err) - } + require.NoError(t, err) + require.NoError(t, w.WriteHeader(testHeader())) return w } @@ -91,9 +88,9 @@ func countPolls(polls []Poll, uid string) int { return n } -// TestWatchLoopPollsEachRuleAtItsOwnCadence is §5's per-rule schedule seen -// from the recorder: a 10s rule beside a 300s one keeps its own 5s cadence -// instead of dragging the slack rule along with it or being slowed to its pace. +// The per-rule schedule seen from the recorder: a 10s rule beside a 300s one +// keeps its own 5s cadence instead of dragging the slack rule along with it or +// being slowed to its pace. func TestWatchLoopPollsEachRuleAtItsOwnCadence(t *testing.T) { const tightUID, slackUID = "tight", "slack" path := filepath.Join(t.TempDir(), "log.jsonl") @@ -122,34 +119,25 @@ func TestWatchLoopPollsEachRuleAtItsOwnCadence(t *testing.T) { Concurrency: 2, Clock: clock, }) - if err != nil { - t.Fatalf("watchLoop: %v", err) - } + require.NoError(t, err) _, polls, sentinel, readErr := ReadLog(path) - if readErr != nil { - t.Fatalf("ReadLog: %v", readErr) - } - if sentinel == nil { - t.Fatal("no stopped sentinel after a clean stop") - } - if sentinel.Before(testNow.Add(300 * time.Second)) { - t.Errorf("sentinel at %s, want >= the stop time %s", sentinel, testNow.Add(300*time.Second)) - } + require.NoError(t, readErr) + require.NotNil(t, sentinel, "no stopped sentinel after a clean stop") + require.False(t, sentinel.Before(testNow.Add(300*time.Second))) // 300s of window at 5s and 150s, minus the initial stagger offset of up to // one cadence: 59-60 and 1-2. The assertion is the ratio, not the exact // count — a single global cycle would give both rules the same number. - if got := countPolls(polls, tightUID); got < 59 || got > 61 { - t.Errorf("tight rule polled %d times, want ~60 (300s at 5s)", got) - } - if got := countPolls(polls, slackUID); got < 1 || got > 3 { - t.Errorf("slack rule polled %d times, want ~2 (300s at 150s)", got) - } + got := countPolls(polls, tightUID) + require.GreaterOrEqual(t, got, 59) + require.LessOrEqual(t, got, 61) + got = countPolls(polls, slackUID) + require.GreaterOrEqual(t, got, 1) + require.LessOrEqual(t, got, 3) } -// TestWatchLoopHardErrorLeavesNoSentinel is §4.5's fail-closed rule from the -// recorder's side: a recorder that dies must look exactly like a coverage gap, -// so it must not sign off the log on its way out. +// Fail-closed from the recorder's side: a recorder that dies must look exactly +// like a coverage gap, so it must not sign off the log on its way out. func TestWatchLoopHardErrorLeavesNoSentinel(t *testing.T) { path := filepath.Join(t.TempDir(), "log.jsonl") clock := newVirtualClock(testNow) @@ -174,26 +162,17 @@ func TestWatchLoopHardErrorLeavesNoSentinel(t *testing.T) { Concurrency: 1, Clock: clock, }) - if !errors.Is(err, boom) { - t.Fatalf("watchLoop error = %v, want %v", err, boom) - } + require.ErrorIs(t, err, boom) _, polls, sentinel, readErr := ReadLog(path) - if readErr != nil { - t.Fatalf("ReadLog: %v", readErr) - } - if sentinel != nil { - t.Errorf("sentinel at %s after a failed recording; check would read that as a finished window", sentinel) - } - if len(polls) != 1 { - t.Errorf("kept %d polls, want the 1 that succeeded before the failure", len(polls)) - } + require.NoError(t, readErr) + require.Nil(t, sentinel, "check would read that as a finished window") + require.Len(t, polls, 1, "want the 1 that succeeded before the failure") } -// TestWatchLoopSignalDuringPollIsACleanStop pins §4.4 step 1: SIGTERM arriving -// while a poll is in flight is a clean stop, so the aborted poll's error must -// not suppress the sentinel — otherwise every normal check run, which stops the -// recorder exactly this way, would end unobservable. +// SIGTERM arriving while a poll is in flight is a clean stop, so the aborted +// poll's error must not suppress the sentinel — otherwise every normal check +// run, which stops the recorder exactly this way, would end unobservable. func TestWatchLoopSignalDuringPollIsACleanStop(t *testing.T) { path := filepath.Join(t.TempDir(), "log.jsonl") clock := newVirtualClock(testNow) @@ -212,7 +191,7 @@ func TestWatchLoopSignalDuringPollIsACleanStop(t *testing.T) { return observation(now, testStateRule("r1", title, time.Minute, now)), nil }) - if err := watchLoop(ctx, watchLoopConfig{ + require.NoError(t, watchLoop(ctx, watchLoopConfig{ Src: src, Writer: w, Reducer: NewReducer(), @@ -220,15 +199,11 @@ func TestWatchLoopSignalDuringPollIsACleanStop(t *testing.T) { Cadence: map[string]time.Duration{"r1": 30 * time.Second}, Concurrency: 1, Clock: clock, - }); err != nil { - t.Fatalf("watchLoop: %v", err) - } + })) - if _, _, sentinel, err := ReadLog(path); err != nil { - t.Fatalf("ReadLog: %v", err) - } else if sentinel == nil { - t.Error("no sentinel after a signalled stop; check would call a fully observed window unobservable") - } + _, _, sentinel, err := ReadLog(path) + require.NoError(t, err) + require.NotNil(t, sentinel, "no sentinel after a signalled stop; check would call a fully observed window unobservable") } // TestWatchLoopWithNothingToPollStillFinishesTheLog covers the every-rule-is- @@ -243,7 +218,7 @@ func TestWatchLoopWithNothingToPollStillFinishesTheLog(t *testing.T) { return Observation{}, fmt.Errorf("nothing should be polled, got %q", title) }) - if err := watchLoop(context.Background(), watchLoopConfig{ + require.NoError(t, watchLoop(context.Background(), watchLoopConfig{ Src: src, Writer: w, Reducer: NewReducer(), @@ -252,20 +227,12 @@ func TestWatchLoopWithNothingToPollStillFinishesTheLog(t *testing.T) { Until: testNow.Add(time.Minute), Concurrency: 1, Clock: clock, - }); err != nil { - t.Fatalf("watchLoop: %v", err) - } + })) _, polls, sentinel, err := ReadLog(path) - if err != nil { - t.Fatalf("ReadLog: %v", err) - } - if len(polls) != 0 { - t.Errorf("wrote %d polls with nothing to poll", len(polls)) - } - if sentinel == nil { - t.Error("no sentinel: check cannot tell this recording from one that died") - } + require.NoError(t, err) + require.Empty(t, polls) + require.NotNil(t, sentinel, "no sentinel: check cannot tell this recording from one that died") } // TestWatchLoopPollBatchKeepsTheHeartbeatsItGot: one rule's failure must not @@ -294,23 +261,16 @@ func TestWatchLoopPollBatchKeepsTheHeartbeatsItGot(t *testing.T) { Concurrency: 2, Clock: clock, } - if err := cfg.pollBatch(context.Background(), []string{"ok", "bad"}); !errors.Is(err, boom) { - t.Fatalf("pollBatch error = %v, want %v", err, boom) - } - if err := w.Close(); err != nil { - t.Fatalf("Close: %v", err) - } + require.ErrorIs(t, cfg.pollBatch(context.Background(), []string{"ok", "bad"}), boom) + require.NoError(t, w.Close()) _, polls, _, err := ReadLog(path) - if err != nil { - t.Fatalf("ReadLog: %v", err) - } - if len(polls) != 1 || polls[0].RuleUID != "ok" { - t.Errorf("polls = %+v, want the one heartbeat that was actually observed", polls) - } + require.NoError(t, err) + require.Len(t, polls, 1) + require.Equal(t, "ok", polls[0].RuleUID) } -// TestReducerSeedFromKeepsMarkersAcrossTheHandoff is H2 at the one seam P6 +// The vanish-versus-clear distinction at the one seam the parent/child handoff // introduces. The parent observes a firing instance; the child starts with a // fresh Reducer and sees the instance gone. Seeded, that is a vanish — a // discontinuity. Unseeded, it is nothing at all, and the instance silently @@ -320,34 +280,27 @@ func TestReducerSeedFromKeepsMarkersAcrossTheHandoff(t *testing.T) { key := instanceKey(firing.Labels) parentPoll := Poll{RuleUID: "r1", Found: true, Abnormal: []Instance{firing}} // The child's first response: the instance is gone from the response - // entirely, which is a vanish and never a clear (§4.7). + // entirely, which is a vanish and never a clear. childObs := observation(testNow, testStateRule("r1", "Example", time.Minute, testNow)) t.Run("seeded", func(t *testing.T) { r := NewReducer() r.seedFrom([]Poll{parentPoll}) p := r.Reduce("r1", childObs) - if !slices.Contains(p.Vanished, key) { - t.Errorf("vanished = %v, want it to contain %q", p.Vanished, key) - } - if len(p.Cleared) != 0 { - t.Errorf("cleared = %v, want none: a vanish is not a recovery", p.Cleared) - } + require.Contains(t, p.Vanished, key) + require.Empty(t, p.Cleared, "a vanish is not a recovery") }) t.Run("unseeded loses the transition", func(t *testing.T) { p := NewReducer().Reduce("r1", childObs) - if len(p.Vanished) != 0 { - t.Fatalf("vanished = %v; this subtest exists to show the seed is what produces the marker", p.Vanished) - } + require.Empty(t, p.Vanished, "this subtest exists to show the seed is what produces the marker") }) t.Run("a not-found poll does not clear the seed", func(t *testing.T) { r := NewReducer() r.seedFrom([]Poll{parentPoll, {RuleUID: "r1", Found: false}}) - if p := r.Reduce("r1", childObs); !slices.Contains(p.Vanished, key) { - t.Errorf("vanished = %v, want it to contain %q: an absent rule leaves the abnormal set untouched", p.Vanished, key) - } + p := r.Reduce("r1", childObs) + require.Contains(t, p.Vanished, key, "an absent rule leaves the abnormal set untouched") }) } @@ -384,59 +337,48 @@ func liveObservation(grafanaNow time.Time) Observation { testInstance(StateNormal, "", "a"))) } -// TestPrepareWatchDoesNotWaitForPausedRules is §22.4's regression test: a rule -// paused in its definition is skipped, never waited for. Waiting for one either -// hangs forever or errors before the deploy — and the header must still name -// it, so check can report it as skipped rather than lose it. +// A rule paused in its definition is skipped, never waited for. Waiting for one +// either hangs forever or errors before the deploy — and the header must still +// name it, so check can report it as skipped rather than lose it. func TestPrepareWatchDoesNotWaitForPausedRules(t *testing.T) { var notes strings.Builder cfg := watchTestConfig(t, ¬es, "uid:"+watchActiveUID, "uid:"+watchPausedUID) src := watchTestSource(t, liveObservation(testNow)) prep, err := prepareWatch(context.Background(), cfg, src) - if err != nil { - t.Fatalf("prepareWatch: %v", err) - } - if err := prep.writer.Close(); err != nil { - t.Fatalf("Close: %v", err) - } + require.NoError(t, err) + require.NoError(t, prep.writer.Close()) header, polls, sentinel, err := ReadLog(cfg.Out) - if err != nil { - t.Fatalf("ReadLog: %v", err) - } - if sentinel != nil { - t.Error("the parent wrote a sentinel; that would tell check the recording ended before the child started") - } + require.NoError(t, err) + require.Nil(t, sentinel, "the parent wrote a sentinel; that would tell check the recording ended before the child started") - if len(header.Rules) != 2 { - t.Fatalf("header names %d rules, want both the live and the paused one", len(header.Rules)) - } + require.Len(t, header.Rules, 2) for _, lr := range header.Rules { - if lr.PollEverySeconds <= 0 { - t.Errorf("header rule %s records poll_every_seconds=%v; check needs a positive cadence to derive maxGap from", lr.UID, lr.PollEverySeconds) - } - if lr.UID == watchPausedUID && !lr.IsPaused { - t.Errorf("header rule %s: is_paused = false, want the resolve-time snapshot to say true", lr.UID) + require.Positive(t, lr.PollEverySeconds, + "check needs a positive cadence to derive maxGap from") + if lr.UID == watchPausedUID { + require.True(t, lr.IsPaused, "want the resolve-time snapshot to say true") } } // One poll, for the live rule only — and it is already in the log before - // prepareWatch returned, which is the whole point of §4.3. - if len(polls) != 1 || polls[0].RuleUID != watchActiveUID { - t.Fatalf("polls = %+v, want exactly one first observation of %s", polls, watchActiveUID) - } - if !polls[0].Found || !polls[0].GrafanaNow.Equal(testNow) { - t.Errorf("first poll = %+v, want a found observation at %s", polls[0], testNow) - } - if !strings.Contains(notes.String(), watchPausedTitle) || !strings.Contains(notes.String(), "paused") { - t.Errorf("notes do not mention the paused rule:\n%s", notes.String()) - } + // prepareWatch returned, which is the whole point of the record step. + require.Len(t, polls, 1) + require.Equal(t, watchActiveUID, polls[0].RuleUID) + require.True(t, polls[0].Found) + require.True(t, polls[0].GrafanaNow.Equal(testNow)) + // The poll record holds the state histogram, asserted through a real + // prepareWatch()/Reducer call rather than log_test.go's hand-built + // Writer/ReadLog round trip. + require.Equal(t, map[string]int{"normal": 1}, polls[0].Histogram) + require.Contains(t, notes.String(), watchPausedTitle) + require.Contains(t, notes.String(), "paused") } -// TestPrepareWatchHeaderRecordsTheOverriddenCadence is P5's "two authorities" -// from the writing side: whatever --poll-interval resolves to is what the -// header records, because that is the only value check may derive maxGap from. +// One authority for the cadence, from the writing side: whatever +// --poll-interval resolves to is what the header records, because that is the +// only value check may derive maxGap from. func TestPrepareWatchHeaderRecordsTheOverriddenCadence(t *testing.T) { var notes strings.Builder cfg := watchTestConfig(t, ¬es, "uid:"+watchActiveUID) @@ -444,24 +386,17 @@ func TestPrepareWatchHeaderRecordsTheOverriddenCadence(t *testing.T) { src := watchTestSource(t, liveObservation(testNow)) prep, err := prepareWatch(context.Background(), cfg, src) - if err != nil { - t.Fatalf("prepareWatch: %v", err) - } + require.NoError(t, err) defer prep.writer.Close() - if got := prep.header.Rules[0].PollEverySeconds; got != 120 { - t.Errorf("header poll_every_seconds = %v, want 120 (the override, used verbatim and never clamped)", got) - } - if got := prep.timings[watchActiveUID].maxGap; got != 240*time.Second { - t.Errorf("maxGap = %s, want 240s (2 x the recorded cadence)", got) - } - if !strings.Contains(notes.String(), "--poll-interval") { - t.Errorf("notes do not report that the override exceeds half the evaluation interval:\n%s", notes.String()) - } + require.Equal(t, float64(120), prep.header.Rules[0].PollEverySeconds, + "the override, used verbatim and never clamped") + require.Equal(t, 240*time.Second, prep.timings[watchActiveUID].maxGap) + require.Contains(t, notes.String(), "--poll-interval") } -// TestPrepareWatchFailsWhenTheScheduleDoesNotFit: the budget check runs on the -// latencies the parent just measured, before the deploy runs (§5.2). +// The budget check runs on the latencies the parent just measured, before the +// deploy runs. func TestPrepareWatchFailsWhenTheScheduleDoesNotFit(t *testing.T) { var notes strings.Builder cfg := watchTestConfig(t, ¬es, "uid:"+watchActiveUID) @@ -471,15 +406,13 @@ func TestPrepareWatchFailsWhenTheScheduleDoesNotFit(t *testing.T) { src := watchTestSource(t, obs) _, err := prepareWatch(context.Background(), cfg, src) - if err == nil { - t.Fatal("prepareWatch: no error on a schedule that cannot hold its own cadence") - } + require.Error(t, err, "a schedule cannot hold its own cadence") assertBudgetMessage(t, err.Error()) } -// TestPrepareWatchVerifiesNormalInstancesAreVisible is the §3.2 check at the -// one place it can still be cheap: the first observation. If the state endpoint -// stops returning normal instances, the reduction's predicate quietly inverts. +// Normal instances are verified visible at the one place it is still cheap: +// the first observation. If the state endpoint stops returning them, the +// reduction's predicate quietly inverts. func TestPrepareWatchVerifiesNormalInstancesAreVisible(t *testing.T) { var notes strings.Builder cfg := watchTestConfig(t, ¬es, "uid:"+watchActiveUID) @@ -489,20 +422,14 @@ func TestPrepareWatchVerifiesNormalInstancesAreVisible(t *testing.T) { src := watchTestSource(t, observation(testNow, rule)) _, err := prepareWatch(context.Background(), cfg, src) - if err == nil { - t.Fatal("prepareWatch: no error when totals claim normal instances the response omitted") - } - if !strings.Contains(err.Error(), "3.2") { - t.Errorf("error does not name §3.2: %v", err) - } + require.Error(t, err, "totals claim normal instances the response omitted") + require.Contains(t, err.Error(), "no longer returns normal instances") // The failure happens before any poll is appended, so the log holds a // header and nothing else. - if _, polls, _, readErr := ReadLog(cfg.Out); readErr != nil { - t.Fatalf("ReadLog: %v", readErr) - } else if len(polls) != 0 { - t.Errorf("wrote %d polls from an observation it refused to trust", len(polls)) - } + _, polls, _, readErr := ReadLog(cfg.Out) + require.NoError(t, readErr) + require.Empty(t, polls) } func TestPrepareWatchRejectsAnUnsupportedGrafana(t *testing.T) { @@ -511,39 +438,29 @@ func TestPrepareWatchRejectsAnUnsupportedGrafana(t *testing.T) { src := watchTestSource(t, liveObservation(testNow)) src.version = "12.4.0" - if _, err := prepareWatch(context.Background(), cfg, src); err == nil { - t.Fatal("prepareWatch: no error on an unsupported grafana version") - } else if !strings.Contains(err.Error(), "12.4.0") || !strings.Contains(err.Error(), "13.0.0") { - t.Errorf("error names neither what was found nor what is supported: %v", err) - } + _, err := prepareWatch(context.Background(), cfg, src) + require.Error(t, err) + require.Contains(t, err.Error(), "12.4.0") + require.Contains(t, err.Error(), "13.0.0") } -// TestPrepareWatchNotesAnAbsentRule: a rule that resolved in the ruler API but -// is absent from the state endpoint is recorded as Found=false — authoritative -// evidence P7 turns into unobservable — not silently dropped. +// A rule that resolved in the ruler API but is absent from the state endpoint +// is recorded as Found=false — authoritative evidence the coverage proof turns +// into unobservable — not silently dropped. func TestPrepareWatchNotesAnAbsentRule(t *testing.T) { var notes strings.Builder cfg := watchTestConfig(t, ¬es, "uid:"+watchActiveUID) src := watchTestSource(t, observation(testNow)) // an authoritative, empty 2xx prep, err := prepareWatch(context.Background(), cfg, src) - if err != nil { - t.Fatalf("prepareWatch: %v", err) - } - if err := prep.writer.Close(); err != nil { - t.Fatalf("Close: %v", err) - } + require.NoError(t, err) + require.NoError(t, prep.writer.Close()) _, polls, _, err := ReadLog(cfg.Out) - if err != nil { - t.Fatalf("ReadLog: %v", err) - } - if len(polls) != 1 || polls[0].Found { - t.Fatalf("polls = %+v, want one poll recorded as not found", polls) - } - if !strings.Contains(notes.String(), "absent from the state endpoint") { - t.Errorf("notes do not warn about the absent rule:\n%s", notes.String()) - } + require.NoError(t, err) + require.Len(t, polls, 1) + require.False(t, polls[0].Found, "want one poll recorded as not found") + require.Contains(t, notes.String(), "absent from the state endpoint") } func TestWatchConfigValidation(t *testing.T) { @@ -571,31 +488,21 @@ func TestWatchConfigValidation(t *testing.T) { cfg := base() tc.mutate(&cfg) err := cfg.withDefaults().validate() - if err == nil { - t.Fatalf("validate: no error, want one naming %q", tc.want) - } - if !strings.Contains(err.Error(), tc.want) { - t.Errorf("validate error = %v, want it to name %q", err, tc.want) - } + require.Errorf(t, err, "validate: no error, want one naming %q", tc.want) + require.Containsf(t, err.Error(), tc.want, "validate error") }) } t.Run("defaults derive the pidfile and daemon log from the log path", func(t *testing.T) { cfg := base().withDefaults() - if cfg.PidFile != cfg.Out+".pid" { - t.Errorf("PidFile = %q, want %q — check finds the recorder by this convention", cfg.PidFile, cfg.Out+".pid") - } - if cfg.DaemonLog == "" { - t.Error("DaemonLog is empty: a detached child would have nowhere to explain a failure") - } - if err := cfg.validate(); err != nil { - t.Errorf("validate: %v", err) - } + require.Equal(t, cfg.Out+".pid", cfg.PidFile) + require.NotEmpty(t, cfg.DaemonLog, "a detached child would have nowhere to explain a failure") + require.NoError(t, cfg.validate()) }) } -// TestChildScheduleUsesTheRecordedCadence is P5's fail-open direction, checked -// on the child's side: a log recorded at 5s on a 300s rule must schedule at 5s. +// The fail-open direction, checked on the child's side: a log recorded at 5s on +// a 300s rule must schedule at 5s. // Re-deriving from the interval would give 150s — and every real 250s hole in // that recording would pass. func TestChildScheduleUsesTheRecordedCadence(t *testing.T) { @@ -605,15 +512,10 @@ func TestChildScheduleUsesTheRecordedCadence(t *testing.T) { }} titles, cadence, err := childSchedule(h) - if err != nil { - t.Fatalf("childSchedule: %v", err) - } - if _, ok := titles["paused"]; ok { - t.Error("the child scheduled a rule that was paused when the window opened (§4.3)") - } - if got := cadence["fast"]; got != 5*time.Second { - t.Errorf("pollEvery = %s, want 5s from the header, not %s from the interval", got, defaultPollEvery(300)) - } + require.NoError(t, err) + _, ok := titles["paused"] + require.False(t, ok, "the child scheduled a rule that was paused when the window opened") + require.Equal(t, 5*time.Second, cadence["fast"]) } func TestChildScheduleRejectsAnUnusableHeader(t *testing.T) { @@ -637,11 +539,9 @@ func TestChildScheduleRejectsAnUnusableHeader(t *testing.T) { }, } { t.Run(tc.name, func(t *testing.T) { - if _, _, err := childSchedule(tc.h); err == nil { - t.Fatalf("childSchedule: no error, want one naming %q", tc.want) - } else if !strings.Contains(err.Error(), tc.want) { - t.Errorf("error = %v, want it to name %q", err, tc.want) - } + _, _, err := childSchedule(tc.h) + require.Errorf(t, err, "childSchedule: no error, want one naming %q", tc.want) + require.Contains(t, err.Error(), tc.want) }) } } @@ -666,85 +566,55 @@ func TestChildArgsCarryNoSecretsAndNoRuleSet(t *testing.T) { joined := strings.Join(args, " ") for _, want := range []string{DaemonChildFlag, "--out /tmp/log.jsonl", "--concurrency 3", "--until ", ReadyFDFlag + " 3"} { - if !strings.Contains(joined, want) { - t.Errorf("child args %q do not contain %q", joined, want) - } + require.Contains(t, joined, want) } for _, forbidden := range []string{"secret-token", "Example", "--folder", "--poll-interval", "--pidfile"} { - if strings.Contains(joined, forbidden) { - t.Errorf("child args %q contain %q, which must not reach argv", joined, forbidden) - } + require.NotContains(t, joined, forbidden) } } func TestPidFileRoundTrip(t *testing.T) { path := filepath.Join(t.TempDir(), "log.jsonl.pid") - if err := writePidFile(path, 4242); err != nil { - t.Fatalf("writePidFile: %v", err) - } + require.NoError(t, writePidFile(path, 4242)) pid, err := ReadPidFile(path) - if err != nil { - t.Fatalf("ReadPidFile: %v", err) - } - if pid != 4242 { - t.Errorf("pid = %d, want 4242", pid) - } + require.NoError(t, err) + require.Equal(t, 4242, pid) t.Run("garbage is an error, never a pid", func(t *testing.T) { bad := filepath.Join(t.TempDir(), "bad.pid") - if err := os.WriteFile(bad, []byte("not-a-pid\n"), 0o644); err != nil { - t.Fatalf("write: %v", err) - } - if _, err := ReadPidFile(bad); err == nil { - t.Error("ReadPidFile: no error on an unparseable pidfile") - } + require.NoError(t, os.WriteFile(bad, []byte("not-a-pid\n"), 0o644)) + _, err := ReadPidFile(bad) + require.Error(t, err, "no error on an unparseable pidfile") }) } func TestDaemonLogTail(t *testing.T) { t.Run("missing file is unreadable", func(t *testing.T) { out := daemonLogTail(filepath.Join(t.TempDir(), "nope.daemon.log"), 0) - if !strings.Contains(out, "unreadable") { - t.Errorf("daemonLogTail = %q, want it to name the file as unreadable", out) - } + require.Contains(t, out, "unreadable") }) t.Run("small file returns its content", func(t *testing.T) { path := filepath.Join(t.TempDir(), "small.daemon.log") - if err := os.WriteFile(path, []byte("line one\nline two\n"), 0o644); err != nil { - t.Fatalf("write: %v", err) - } - if out := daemonLogTail(path, 0); out != "line one\nline two" { - t.Errorf("daemonLogTail = %q, want the full trimmed content", out) - } + require.NoError(t, os.WriteFile(path, []byte("line one\nline two\n"), 0o644)) + require.Equal(t, "line one\nline two", daemonLogTail(path, 0)) }) t.Run("large file keeps only the tail", func(t *testing.T) { path := filepath.Join(t.TempDir(), "large.daemon.log") prefix := strings.Repeat("P", 1000) suffix := strings.Repeat("S", daemonLogTailBytes) - if err := os.WriteFile(path, []byte(prefix+suffix), 0o644); err != nil { - t.Fatalf("write: %v", err) - } - out := daemonLogTail(path, 0) - if out != suffix { - t.Errorf("daemonLogTail = %q, want exactly the trailing %d bytes (the %d leading bytes dropped)", out, daemonLogTailBytes, len(prefix)) - } + require.NoError(t, os.WriteFile(path, []byte(prefix+suffix), 0o644)) + require.Equal(t, suffix, daemonLogTail(path, 0)) }) t.Run("offset skips a previous run's content", func(t *testing.T) { path := filepath.Join(t.TempDir(), "shared.daemon.log") prior := strings.Repeat("p", 2000) - if err := os.WriteFile(path, []byte(prior), 0o644); err != nil { - t.Fatalf("write: %v", err) - } + require.NoError(t, os.WriteFile(path, []byte(prior), 0o644)) from := int64(len(prior)) thisRun := "this run's output\n" - if err := os.WriteFile(path, []byte(prior+thisRun), 0o644); err != nil { - t.Fatalf("write: %v", err) - } - if out := daemonLogTail(path, from); out != "this run's output" { - t.Errorf("daemonLogTail = %q, want only this run's bytes after offset %d", out, from) - } + require.NoError(t, os.WriteFile(path, []byte(prior+thisRun), 0o644)) + require.Equal(t, "this run's output", daemonLogTail(path, from)) }) }