From c6b473c3c8719c6cabc2c8c1ce548a7ba2f09924 Mon Sep 17 00:00:00 2001 From: Bartek Tofel Date: Mon, 31 Aug 2026 11:37:34 +0200 Subject: [PATCH 01/43] chore: implement phase 0 --- .../workflows/grafana-alertcheck-release.yml | 34 +++++++++++++++++++ .github/workflows/test.yaml | 3 ++ .gitignore | 4 ++- grafana-alertcheck/.goreleaser.yaml | 33 ++++++++++++++++++ grafana-alertcheck/README.md | 6 ++++ .../cmd/grafana-alertcheck/main.go | 7 ++++ grafana-alertcheck/go.mod | 3 ++ grafana-alertcheck/internal/gate/gate.go | 2 ++ 8 files changed, 91 insertions(+), 1 deletion(-) create mode 100644 .github/workflows/grafana-alertcheck-release.yml create mode 100644 grafana-alertcheck/.goreleaser.yaml create mode 100644 grafana-alertcheck/README.md create mode 100644 grafana-alertcheck/cmd/grafana-alertcheck/main.go create mode 100644 grafana-alertcheck/go.mod create mode 100644 grafana-alertcheck/internal/gate/gate.go diff --git a/.github/workflows/grafana-alertcheck-release.yml b/.github/workflows/grafana-alertcheck-release.yml new file mode 100644 index 000000000..0f9f2abb1 --- /dev/null +++ b/.github/workflows/grafana-alertcheck-release.yml @@ -0,0 +1,34 @@ +name: Grafana Alertcheck Release + +on: + push: + tags: + - grafana-alertcheck/v* + +jobs: + release: + name: Build and Release + runs-on: ubuntu-latest + environment: integration + permissions: + id-token: write + contents: write + steps: + - name: Checkout repo + uses: actions/checkout@v7 + with: + fetch-depth: 0 + - name: Set up Go + uses: actions/setup-go@v7 + with: + go-version-file: ./grafana-alertcheck/go.mod + cache-dependency-path: ./grafana-alertcheck/go.mod + - name: Goreleaser Release + uses: goreleaser/goreleaser-action@v7 + with: + distribution: goreleaser-pro + version: "~> v2" + args: release --clean -f ./grafana-alertcheck/.goreleaser.yaml + env: + GITHUB_TOKEN: ${{ secrets.GITHUB_TOKEN }} + GORELEASER_KEY: ${{ secrets.GORELEASER_KEY }} diff --git a/.github/workflows/test.yaml b/.github/workflows/test.yaml index 4b4bc20c5..4146daa5f 100644 --- a/.github/workflows/test.yaml +++ b/.github/workflows/test.yaml @@ -37,6 +37,9 @@ jobs: - path: parrot vm: ubuntu-latest regex: ./... + - path: grafana-alertcheck + vm: ubuntu-latest + regex: ./... - path: tools/workflowresultparser vm: ubuntu-latest regex: ./... diff --git a/.gitignore b/.gitignore index 300ded3f9..dffcf974c 100644 --- a/.gitignore +++ b/.gitignore @@ -84,4 +84,6 @@ parrot/parrot # Devenv (generated manually) devenv/ # Generated TOML definitions -book/src/framework/developer_environment/**.toml \ No newline at end of file +book/src/framework/developer_environment/**.toml + +tmp/ \ No newline at end of file diff --git a/grafana-alertcheck/.goreleaser.yaml b/grafana-alertcheck/.goreleaser.yaml new file mode 100644 index 000000000..09e6032cb --- /dev/null +++ b/grafana-alertcheck/.goreleaser.yaml @@ -0,0 +1,33 @@ +# yaml-language-server: $schema=https://goreleaser.com/static/schema-pro.json +version: 2 +project_name: grafana-alertcheck + +dist: grafana-alertcheck/dist + +monorepo: + tag_prefix: grafana-alertcheck/ + dir: grafana-alertcheck + +builds: + - id: grafana-alertcheck + main: ./cmd/grafana-alertcheck/main.go + ldflags: + - -s + - -w + - -X github.com/smartcontractkit/chainlink-testing-framework/grafana-alertcheck.version={{.Version}} + - -X github.com/smartcontractkit/chainlink-testing-framework/grafana-alertcheck.commit={{.ShortCommit}} + - -X github.com/smartcontractkit/chainlink-testing-framework/grafana-alertcheck.date={{.CommitDate}} + - -X github.com/smartcontractkit/chainlink-testing-framework/grafana-alertcheck.builtBy=goreleaser + goos: + - linux + - darwin + goarch: + - amd64 + - arm64 + binary: grafana-alertcheck + env: + - CGO_ENABLED=0 + +before: + hooks: + - sh -c "cd grafana-alertcheck && go mod tidy" diff --git a/grafana-alertcheck/README.md b/grafana-alertcheck/README.md new file mode 100644 index 000000000..43cc03fc4 --- /dev/null +++ b/grafana-alertcheck/README.md @@ -0,0 +1,6 @@ +# grafana-alertcheck + +A CD quality gate for Grafana alerts: bookend a release with `watch` (record) and `check` (classify) to +answer whether any watched alert was in a bad state during the release window. + +Under construction. diff --git a/grafana-alertcheck/cmd/grafana-alertcheck/main.go b/grafana-alertcheck/cmd/grafana-alertcheck/main.go new file mode 100644 index 000000000..150d617d4 --- /dev/null +++ b/grafana-alertcheck/cmd/grafana-alertcheck/main.go @@ -0,0 +1,7 @@ +package main + +import "os" + +func main() { + os.Exit(2) +} diff --git a/grafana-alertcheck/go.mod b/grafana-alertcheck/go.mod new file mode 100644 index 000000000..b0c8511ce --- /dev/null +++ b/grafana-alertcheck/go.mod @@ -0,0 +1,3 @@ +module github.com/smartcontractkit/chainlink-testing-framework/grafana-alertcheck + +go 1.26.6 diff --git a/grafana-alertcheck/internal/gate/gate.go b/grafana-alertcheck/internal/gate/gate.go new file mode 100644 index 000000000..262a101a1 --- /dev/null +++ b/grafana-alertcheck/internal/gate/gate.go @@ -0,0 +1,2 @@ +// Package gate implements the grafana-alertcheck coverage-and-classification gate. +package gate From 2945092d368bc12fa6c051823be5291512a3551d Mon Sep 17 00:00:00 2001 From: Bartek Tofel Date: Tue, 1 Sep 2026 13:24:28 +0200 Subject: [PATCH 02/43] chore: use SHA not tag for goreleaser/goreleaser-action --- .github/workflows/grafana-alertcheck-release.yml | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/.github/workflows/grafana-alertcheck-release.yml b/.github/workflows/grafana-alertcheck-release.yml index 0f9f2abb1..4f946b18f 100644 --- a/.github/workflows/grafana-alertcheck-release.yml +++ b/.github/workflows/grafana-alertcheck-release.yml @@ -24,7 +24,7 @@ jobs: go-version-file: ./grafana-alertcheck/go.mod cache-dependency-path: ./grafana-alertcheck/go.mod - name: Goreleaser Release - uses: goreleaser/goreleaser-action@v7 + uses: goreleaser/goreleaser-action@f06c13b6b1a9625abc9e6e439d9c05a8f2190e94 # v7.2.3 with: distribution: goreleaser-pro version: "~> v2" From afcf2571af8c9bdee168afb50ee8efe99694e766 Mon Sep 17 00:00:00 2001 From: Bartek Tofel Date: Mon, 31 Aug 2026 12:39:42 +0200 Subject: [PATCH 03/43] chore: implement phase 1 Strict parsers for the state and ruler endpoints (H1), a Prometheus-style duration parser, and fixtures sliced from real Grafana 13.1.0 payloads covering every required/optional-field and must-error case, including the "Normal (NoData)"/"Normal (Error)" composite reason states found live in the current fleet capture (not in the original plan's vocabulary). --- grafana-alertcheck/internal/gate/duration.go | 93 ++++ .../internal/gate/duration_test.go | 65 +++ grafana-alertcheck/internal/gate/jsonreq.go | 39 ++ .../internal/gate/jsonreq_test.go | 92 ++++ .../internal/gate/parse_ruler.go | 184 ++++++++ .../internal/gate/parse_ruler_test.go | 116 +++++ .../internal/gate/parse_state.go | 269 ++++++++++++ .../internal/gate/parse_state_test.go | 415 ++++++++++++++++++ .../internal/gate/testdata/README.md | 91 ++++ .../testdata/ruler_datasource_managed.json | 21 + .../gate/testdata/ruler_recording.json | 44 ++ .../internal/gate/testdata/ruler_rules.json | 332 ++++++++++++++ .../gate/testdata/state_health_error.json | 68 +++ .../gate/testdata/state_health_nodata.json | 62 +++ .../gate/testdata/state_missing_file.json | 73 +++ .../gate/testdata/state_missing_health.json | 73 +++ .../gate/testdata/state_missing_interval.json | 73 +++ .../gate/testdata/state_missing_lasteval.json | 73 +++ .../gate/testdata/state_missing_name.json | 73 +++ .../gate/testdata/state_missing_optional.json | 41 ++ .../gate/testdata/state_missing_state.json | 73 +++ .../gate/testdata/state_one_instance.json | 74 ++++ .../testdata/state_only_active_instances.json | 79 ++++ .../internal/gate/testdata/state_paused.json | 40 ++ .../gate/testdata/state_reason_composite.json | 99 +++++ .../gate/testdata/state_unknown_state.json | 74 ++++ .../testdata/state_zerotime_unpaused.json | 74 ++++ 27 files changed, 2810 insertions(+) create mode 100644 grafana-alertcheck/internal/gate/duration.go create mode 100644 grafana-alertcheck/internal/gate/duration_test.go create mode 100644 grafana-alertcheck/internal/gate/jsonreq.go create mode 100644 grafana-alertcheck/internal/gate/jsonreq_test.go create mode 100644 grafana-alertcheck/internal/gate/parse_ruler.go create mode 100644 grafana-alertcheck/internal/gate/parse_ruler_test.go create mode 100644 grafana-alertcheck/internal/gate/parse_state.go create mode 100644 grafana-alertcheck/internal/gate/parse_state_test.go create mode 100644 grafana-alertcheck/internal/gate/testdata/README.md create mode 100644 grafana-alertcheck/internal/gate/testdata/ruler_datasource_managed.json create mode 100644 grafana-alertcheck/internal/gate/testdata/ruler_recording.json create mode 100644 grafana-alertcheck/internal/gate/testdata/ruler_rules.json create mode 100644 grafana-alertcheck/internal/gate/testdata/state_health_error.json create mode 100644 grafana-alertcheck/internal/gate/testdata/state_health_nodata.json create mode 100644 grafana-alertcheck/internal/gate/testdata/state_missing_file.json create mode 100644 grafana-alertcheck/internal/gate/testdata/state_missing_health.json create mode 100644 grafana-alertcheck/internal/gate/testdata/state_missing_interval.json create mode 100644 grafana-alertcheck/internal/gate/testdata/state_missing_lasteval.json create mode 100644 grafana-alertcheck/internal/gate/testdata/state_missing_name.json create mode 100644 grafana-alertcheck/internal/gate/testdata/state_missing_optional.json create mode 100644 grafana-alertcheck/internal/gate/testdata/state_missing_state.json create mode 100644 grafana-alertcheck/internal/gate/testdata/state_one_instance.json create mode 100644 grafana-alertcheck/internal/gate/testdata/state_only_active_instances.json create mode 100644 grafana-alertcheck/internal/gate/testdata/state_paused.json create mode 100644 grafana-alertcheck/internal/gate/testdata/state_reason_composite.json create mode 100644 grafana-alertcheck/internal/gate/testdata/state_unknown_state.json create mode 100644 grafana-alertcheck/internal/gate/testdata/state_zerotime_unpaused.json diff --git a/grafana-alertcheck/internal/gate/duration.go b/grafana-alertcheck/internal/gate/duration.go new file mode 100644 index 000000000..011c27f14 --- /dev/null +++ b/grafana-alertcheck/internal/gate/duration.go @@ -0,0 +1,93 @@ +package gate + +import ( + "fmt" + "math" + "strconv" + "strings" + "time" +) + +// promDurationUnit is one accepted unit in a Prometheus-style duration string. +// rank increases with unit size; ParsePromDuration requires strictly decreasing +// rank across concatenated components (e.g. "1h30m", never "30m1h"). +type promDurationUnit struct { + suffix string + mult time.Duration + rank int +} + +// Longest suffix must be tried first ("ms" before "m") — see the matching loop below. +var promDurationUnits = []promDurationUnit{ + {"ms", time.Millisecond, 0}, + {"s", time.Second, 1}, + {"m", time.Minute, 2}, + {"h", time.Hour, 3}, + {"d", 24 * time.Hour, 4}, + {"w", 7 * 24 * time.Hour, 5}, + {"y", 365 * 24 * time.Hour, 6}, +} + +// ParsePromDuration parses a Grafana/Prometheus-style duration ("1h30m", "1d", "1w"). +// Unlike time.ParseDuration, it accepts "d" and "w" (§11.8). "" and "0" are 0. +func ParsePromDuration(s string) (time.Duration, error) { + if s == "" || s == "0" { + return 0, nil + } + if strings.HasPrefix(s, "-") { + return 0, fmt.Errorf("invalid duration %q: negative durations are not supported", s) + } + + var total time.Duration + prevRank := len(promDurationUnits) // sentinel higher than any real rank + rest := s + for rest != "" { + i := 0 + for i < len(rest) && rest[i] >= '0' && rest[i] <= '9' { + i++ + } + if i == 0 { + return 0, fmt.Errorf("invalid duration %q: expected a number", s) + } + numPart := rest[:i] + rest = rest[i:] + + matched := -1 + matchLen := 0 + for idx, u := range promDurationUnits { + if strings.HasPrefix(rest, u.suffix) && len(u.suffix) > matchLen { + matched = idx + matchLen = len(u.suffix) + } + } + if matched == -1 { + return 0, fmt.Errorf("invalid duration %q: unrecognized unit", s) + } + u := promDurationUnits[matched] + if u.rank >= prevRank { + return 0, fmt.Errorf("invalid duration %q: units must appear in descending order", s) + } + prevRank = u.rank + + n, err := strconv.ParseInt(numPart, 10, 64) + if err != nil { + return 0, fmt.Errorf("invalid duration %q: %w", s, err) + } + // n and u.mult are both non-negative here (the leading "-" check + // above rejects negative input), so overflow of either the + // multiplication or the running sum can only wrap upward past + // math.MaxInt64 — check both explicitly rather than let a duration + // like "300y" silently become negative garbage that would later + // feed transitionGrace. + if n != 0 && n > math.MaxInt64/int64(u.mult) { + return 0, fmt.Errorf("invalid duration %q: overflows time.Duration", s) + } + delta := time.Duration(n) * u.mult + if total > math.MaxInt64-delta { + return 0, fmt.Errorf("invalid duration %q: overflows time.Duration", s) + } + total += delta + rest = rest[matchLen:] + } + return total, nil +} diff --git a/grafana-alertcheck/internal/gate/duration_test.go b/grafana-alertcheck/internal/gate/duration_test.go new file mode 100644 index 000000000..73a0c00e1 --- /dev/null +++ b/grafana-alertcheck/internal/gate/duration_test.go @@ -0,0 +1,65 @@ +package gate + +import ( + "testing" + "time" +) + +func TestParsePromDuration(t *testing.T) { + cases := []struct { + in string + want time.Duration + }{ + {"", 0}, + {"0", 0}, + {"0s", 0}, + {"500ms", 500 * time.Millisecond}, + {"1s", time.Second}, + {"1m", time.Minute}, + {"2m", 2 * time.Minute}, + {"3m", 3 * time.Minute}, + {"5m", 5 * time.Minute}, + {"10m", 10 * time.Minute}, + {"15m", 15 * time.Minute}, + {"20m", 20 * time.Minute}, + {"30m", 30 * time.Minute}, + {"1h", time.Hour}, + {"6h", 6 * time.Hour}, + {"12h", 12 * time.Hour}, + {"1d", 24 * time.Hour}, + {"1w", 7 * 24 * time.Hour}, + {"1y", 365 * 24 * time.Hour}, + {"1h30m", time.Hour + 30*time.Minute}, + {"1s500ms", time.Second + 500*time.Millisecond}, + {"2d12h", 2*24*time.Hour + 12*time.Hour}, + } + for _, c := range cases { + got, err := ParsePromDuration(c.in) + if err != nil { + t.Errorf("ParsePromDuration(%q): unexpected error: %v", c.in, err) + continue + } + if got != c.want { + t.Errorf("ParsePromDuration(%q) = %v, want %v", c.in, got, c.want) + } + } +} + +func TestParsePromDuration_Errors(t *testing.T) { + cases := []string{ + "5", // bare number, no unit + "-5m", // negative + "5x", // unknown unit + "30m1h", // ascending order (must be descending) + "1h1h", // duplicate unit + "m", // unit with no number + "1.5h", // fractional number not supported by this grammar + "1 h", // whitespace + "300y", // overflows time.Duration (int64 nanoseconds) — must error, not wrap negative + } + for _, in := range cases { + if _, err := ParsePromDuration(in); err == nil { + t.Errorf("ParsePromDuration(%q): expected an error, got none", in) + } + } +} diff --git a/grafana-alertcheck/internal/gate/jsonreq.go b/grafana-alertcheck/internal/gate/jsonreq.go new file mode 100644 index 000000000..cea19b6fe --- /dev/null +++ b/grafana-alertcheck/internal/gate/jsonreq.go @@ -0,0 +1,39 @@ +package gate + +import ( + "bytes" + "encoding/json" + "fmt" +) + +// req decodes m[key] into *dst. It returns an error when key is absent from m +// or explicitly JSON null, so a caller can never mistake absence for a zero +// value (H1) — json.Unmarshal treats "null" as a documented no-op for +// non-pointer targets (string, bool, int, ...), so without this check a +// required field sent as null would silently pass through as its zero value. +func req[T any](m map[string]json.RawMessage, key string, dst *T) error { + raw, ok := m[key] + if !ok || isJSONNull(raw) { + return fmt.Errorf("required field %q is absent", key) + } + if err := json.Unmarshal(raw, dst); err != nil { + return fmt.Errorf("field %q: %w", key, err) + } + return nil +} + +func isJSONNull(raw json.RawMessage) bool { + return string(bytes.TrimSpace(raw)) == "null" +} + +// opt decodes m[key] into *dst when present, leaving *dst untouched when key is absent. +func opt[T any](m map[string]json.RawMessage, key string, dst *T) error { + raw, ok := m[key] + if !ok { + return nil + } + if err := json.Unmarshal(raw, dst); err != nil { + return fmt.Errorf("field %q: %w", key, err) + } + return nil +} diff --git a/grafana-alertcheck/internal/gate/jsonreq_test.go b/grafana-alertcheck/internal/gate/jsonreq_test.go new file mode 100644 index 000000000..bf3881e4b --- /dev/null +++ b/grafana-alertcheck/internal/gate/jsonreq_test.go @@ -0,0 +1,92 @@ +package gate + +import ( + "encoding/json" + "testing" +) + +func rawMap(t *testing.T, jsonObj string) map[string]json.RawMessage { + t.Helper() + var m map[string]json.RawMessage + if err := json.Unmarshal([]byte(jsonObj), &m); err != nil { + t.Fatalf("rawMap: %v", err) + } + return m +} + +func TestReq(t *testing.T) { + m := rawMap(t, `{"present":"hello","wrongtype":123,"nullval":null}`) + + t.Run("present key decodes", func(t *testing.T) { + var s string + if err := req(m, "present", &s); err != nil { + t.Fatalf("unexpected error: %v", err) + } + if s != "hello" { + t.Errorf("got %q, want hello", s) + } + }) + + t.Run("absent key errors", func(t *testing.T) { + var s string + if err := req(m, "missing", &s); err == nil { + t.Fatalf("expected an error, got none") + } + }) + + t.Run("wrong type errors", func(t *testing.T) { + var s string + if err := req(m, "wrongtype", &s); err == nil { + t.Fatalf("expected an error, got none") + } + }) + + t.Run("explicit JSON null errors, never a zero value", func(t *testing.T) { + var s string + err := req(m, "nullval", &s) + if err == nil { + t.Fatalf("expected an error, got none (s=%q) — a null required field must not silently become a zero value", s) + } + }) +} + +func TestOpt(t *testing.T) { + m := rawMap(t, `{"present":"hello","wrongtype":123,"nullval":null}`) + + t.Run("present key decodes", func(t *testing.T) { + var s string + if err := opt(m, "present", &s); err != nil { + t.Fatalf("unexpected error: %v", err) + } + if s != "hello" { + t.Errorf("got %q, want hello", s) + } + }) + + t.Run("absent key leaves dst untouched", func(t *testing.T) { + s := "unchanged" + if err := opt(m, "missing", &s); err != nil { + t.Fatalf("unexpected error: %v", err) + } + if s != "unchanged" { + t.Errorf("got %q, want unchanged", s) + } + }) + + t.Run("wrong type errors", func(t *testing.T) { + var s string + if err := opt(m, "wrongtype", &s); err == nil { + t.Fatalf("expected an error, got none") + } + }) + + t.Run("explicit JSON null leaves dst at its zero value", func(t *testing.T) { + var s string + if err := opt(m, "nullval", &s); err != nil { + t.Fatalf("unexpected error: %v", err) + } + if s != "" { + t.Errorf("got %q, want empty string", s) + } + }) +} diff --git a/grafana-alertcheck/internal/gate/parse_ruler.go b/grafana-alertcheck/internal/gate/parse_ruler.go new file mode 100644 index 000000000..06f51407e --- /dev/null +++ b/grafana-alertcheck/internal/gate/parse_ruler.go @@ -0,0 +1,184 @@ +package gate + +import ( + "encoding/json" + "fmt" + "sort" + "time" +) + +// RuleKind classifies a ruler-endpoint rule by shape, not by name (P1.3). +// P3 rejects KindDatasourceManaged and KindRecording, but only for rules a +// user actually named — ParseDefinitions itself never rejects. +type RuleKind int + +const ( + KindGrafanaManaged RuleKind = iota + KindDatasourceManaged + KindRecording +) + +// Definition is one rule from the ruler endpoint +// (/api/ruler/grafana/api/v1/rules). IntervalSeconds, NoDataState and +// ExecErrState live inside the grafana_alert block and are only populated for +// KindGrafanaManaged — a datasource-managed rule has no such block by +// definition (§11.6 drops relativeTimeRange/keep_firing_for entirely; neither +// is parsed here). +type Definition struct { + UID, Title, Folder, FolderUID, Group string + For time.Duration + IntervalSeconds int + NoDataState string + ExecErrState string + IsPaused bool + Kind RuleKind +} + +// ParseDefinitions strictly parses a ruler-endpoint response body +// (map[namespace][]group) into its rule definitions. +func ParseDefinitions(body []byte) ([]Definition, error) { + var namespaces map[string][]json.RawMessage + if err := json.Unmarshal(body, &namespaces); err != nil { + return nil, fmt.Errorf("ruler response: %w", err) + } + + // Map iteration order is nondeterministic; sort namespace names so + // ParseDefinitions' output order is stable across calls (P3's candidate + // listings and any golden test depend on that). + names := make([]string, 0, len(namespaces)) + for name := range namespaces { + names = append(names, name) + } + sort.Strings(names) + + var defs []Definition + for _, folder := range names { + for gi, groupRaw := range namespaces[folder] { + var group map[string]json.RawMessage + if err := json.Unmarshal(groupRaw, &group); err != nil { + return nil, fmt.Errorf("ruler response: namespace %q: group %d: %w", folder, gi, err) + } + + var groupName string + if err := req(group, "name", &groupName); err != nil { + return nil, fmt.Errorf("ruler response: namespace %q: group %d: %w", folder, gi, err) + } + + var rulesRaw []json.RawMessage + if err := req(group, "rules", &rulesRaw); err != nil { + return nil, fmt.Errorf("ruler response: namespace %q: group %q: %w", folder, groupName, err) + } + + for ri, ruleRaw := range rulesRaw { + def, err := parseDefinition(ruleRaw, folder, groupName) + if err != nil { + return nil, fmt.Errorf("ruler response: namespace %q: group %q: rule %d: %w", folder, groupName, ri, err) + } + defs = append(defs, def) + } + } + } + return defs, nil +} + +func parseDefinition(raw json.RawMessage, folder, group string) (Definition, error) { + var m map[string]json.RawMessage + if err := json.Unmarshal(raw, &m); err != nil { + return Definition{}, fmt.Errorf("%w", err) + } + + var forStr string + if err := opt(m, "for", &forStr); err != nil { + return Definition{}, err + } + forDur, err := ParsePromDuration(forStr) + if err != nil { + return Definition{}, fmt.Errorf("for: %w", err) + } + + var gaRaw json.RawMessage + if err := opt(m, "grafana_alert", &gaRaw); err != nil { + return Definition{}, err + } + if gaRaw == nil { + // No grafana_alert block at all: a datasource-managed (native + // Prometheus-format) rule. Its identity is the Prometheus rule + // name — "alert" for an alerting rule, "record" for a recording + // one — never a synthetic UID (Grafana's ruler API gives this + // shape no uid at all; inventing one would be inventing shape). + def := Definition{Folder: folder, Group: group, For: forDur, Kind: KindDatasourceManaged} + if err := opt(m, "alert", &def.Title); err != nil { + return Definition{}, err + } + if def.Title == "" { + if err := opt(m, "record", &def.Title); err != nil { + return Definition{}, err + } + } + return def, nil + } + + var ga map[string]json.RawMessage + if err := json.Unmarshal(gaRaw, &ga); err != nil { + return Definition{}, fmt.Errorf("grafana_alert: %w", err) + } + + // uid identifies the rule regardless of kind — every grafana_alert + // object Grafana emits, alerting or recording, carries one. + var uid string + if err := req(ga, "uid", &uid); err != nil { + return Definition{}, fmt.Errorf("grafana_alert: %w", err) + } + def := Definition{Folder: folder, Group: group, For: forDur, UID: uid} + + // Classify by the presence of "record" before requiring anything else. + // no_data_state/exec_err_state/is_paused/intervalSeconds are alerting-only + // concepts a recording rule may not carry at all — its real shape is + // unverified (none exist in the fleet capture) — and P3 refuses this + // Kind categorically before any of this would gate a release. Strict- + // parsing a recording rule into a hard error over fields it was never + // going to use would brick `list` and every resolve for rules nobody + // named (§11.6, "do not reject here"). + var record json.RawMessage + if err := opt(ga, "record", &record); err != nil { + return Definition{}, fmt.Errorf("rule %q: grafana_alert: %w", uid, err) + } + if record != nil { + def.Kind = KindRecording + if err := opt(ga, "title", &def.Title); err != nil { + return Definition{}, fmt.Errorf("rule %q: grafana_alert: %w", uid, err) + } + if err := opt(ga, "namespace_uid", &def.FolderUID); err != nil { + return Definition{}, fmt.Errorf("rule %q: grafana_alert: %w", uid, err) + } + if err := opt(ga, "intervalSeconds", &def.IntervalSeconds); err != nil { + return Definition{}, fmt.Errorf("rule %q: grafana_alert: %w", uid, err) + } + if err := opt(ga, "is_paused", &def.IsPaused); err != nil { + return Definition{}, fmt.Errorf("rule %q: grafana_alert: %w", uid, err) + } + return def, nil + } + + def.Kind = KindGrafanaManaged + if err := req(ga, "title", &def.Title); err != nil { + return Definition{}, fmt.Errorf("rule %q: grafana_alert: %w", uid, err) + } + if err := req(ga, "namespace_uid", &def.FolderUID); err != nil { + return Definition{}, fmt.Errorf("rule %q: grafana_alert: %w", uid, err) + } + if err := req(ga, "intervalSeconds", &def.IntervalSeconds); err != nil { + return Definition{}, fmt.Errorf("rule %q: grafana_alert: %w", uid, err) + } + if err := req(ga, "no_data_state", &def.NoDataState); err != nil { + return Definition{}, fmt.Errorf("rule %q: grafana_alert: %w", uid, err) + } + if err := req(ga, "exec_err_state", &def.ExecErrState); err != nil { + return Definition{}, fmt.Errorf("rule %q: grafana_alert: %w", uid, err) + } + if err := req(ga, "is_paused", &def.IsPaused); err != nil { + return Definition{}, fmt.Errorf("rule %q: grafana_alert: %w", uid, err) + } + + return def, nil +} diff --git a/grafana-alertcheck/internal/gate/parse_ruler_test.go b/grafana-alertcheck/internal/gate/parse_ruler_test.go new file mode 100644 index 000000000..56d35b663 --- /dev/null +++ b/grafana-alertcheck/internal/gate/parse_ruler_test.go @@ -0,0 +1,116 @@ +package gate + +import ( + "testing" + "time" +) + +func TestParseDefinitions_RulerRules(t *testing.T) { + defs, err := ParseDefinitions(readFixture(t, "ruler_rules.json")) + if err != nil { + t.Fatalf("ParseDefinitions: unexpected error: %v", err) + } + + byUID := map[string]Definition{} + for _, d := range defs { + if d.Kind != KindGrafanaManaged { + t.Errorf("rule %q: Kind = %v, want KindGrafanaManaged", d.UID, d.Kind) + } + byUID[d.UID] = d + } + + // The real 2-way duplicate title: same folder, same group, same title, + // distinct UIDs (§17, §22.2). + a, ok := byUID["rule0000006a"] + if !ok { + t.Fatalf("missing rule0000006a") + } + b, ok := byUID["rule0000006b"] + if !ok { + t.Fatalf("missing rule0000006b") + } + if a.Title != b.Title || a.Folder != b.Folder || a.Group != b.Group { + t.Errorf("duplicate-title pair should share Title/Folder/Group: a=%+v b=%+v", a, b) + } + if a.UID == b.UID { + t.Errorf("duplicate-title pair should have distinct UIDs") + } + + // The 3 real paused rules. + pausedUIDs := []string{"rule0000002", "rule0000007", "rule0000008"} + for _, uid := range pausedUIDs { + d, ok := byUID[uid] + if !ok { + t.Fatalf("missing paused rule %q", uid) + } + if !d.IsPaused { + t.Errorf("rule %q: IsPaused = false, want true", uid) + } + } + + // for:1d and the derived for:1w rule. + dayRule, ok := byUID["rule0000009"] + if !ok || dayRule.For != 24*time.Hour { + t.Fatalf("rule0000009: For = %v, want 24h (ok=%v)", dayRule.For, ok) + } + weekRule, ok := byUID["rule0000010"] + if !ok || weekRule.For != 7*24*time.Hour { + t.Fatalf("rule0000010: For = %v, want 168h (ok=%v)", weekRule.For, ok) + } + + // Identity shared with testdata/state_paused.json. + shared := byUID["rule0000002"] + if shared.FolderUID != "folder0000002" { + t.Errorf("rule0000002: FolderUID = %q, want folder0000002", shared.FolderUID) + } +} + +func TestParseDefinitions_DatasourceManaged(t *testing.T) { + defs, err := ParseDefinitions(readFixture(t, "ruler_datasource_managed.json")) + if err != nil { + t.Fatalf("ParseDefinitions: unexpected error: %v", err) + } + if len(defs) != 1 { + t.Fatalf("got %d definitions, want 1", len(defs)) + } + if defs[0].Kind != KindDatasourceManaged { + t.Errorf("Kind = %v, want KindDatasourceManaged", defs[0].Kind) + } + if defs[0].For != 5*time.Minute { + t.Errorf("For = %v, want 5m", defs[0].For) + } + // A datasource-managed rule has no uid in this shape; its only identity + // is the Prometheus "alert" name — a synthetic UID would be invented + // shape, and an empty Title would make P3's refusal-by-name unreachable. + if defs[0].Title != "ExampleTargetDown" { + t.Errorf("Title = %q, want ExampleTargetDown", defs[0].Title) + } + if defs[0].UID != "" { + t.Errorf("UID = %q, want empty (this shape has no uid)", defs[0].UID) + } +} + +func TestParseDefinitions_Recording(t *testing.T) { + defs, err := ParseDefinitions(readFixture(t, "ruler_recording.json")) + if err != nil { + t.Fatalf("ParseDefinitions: unexpected error: %v", err) + } + if len(defs) != 1 { + t.Fatalf("got %d definitions, want 1", len(defs)) + } + d := defs[0] + if d.Kind != KindRecording { + t.Errorf("Kind = %v, want KindRecording", d.Kind) + } + if d.UID != "rule0000011" { + t.Errorf("UID = %q, want rule0000011", d.UID) + } + // The fixture deliberately omits no_data_state/exec_err_state/is_paused/ + // intervalSeconds/namespace_uid — alerting-only concepts a recording + // rule may not carry. Requiring them would brick ParseDefinitions for + // every named rule in the same response over one recording rule + // elsewhere in the fleet; they must come back as zero values, not errors. + if d.NoDataState != "" || d.ExecErrState != "" || d.IsPaused || d.IntervalSeconds != 0 || d.FolderUID != "" { + t.Errorf("expected zero-valued alert-only fields for a recording rule, got %+v", d) + } +} diff --git a/grafana-alertcheck/internal/gate/parse_state.go b/grafana-alertcheck/internal/gate/parse_state.go new file mode 100644 index 000000000..ebd30d7a5 --- /dev/null +++ b/grafana-alertcheck/internal/gate/parse_state.go @@ -0,0 +1,269 @@ +package gate + +import ( + "encoding/json" + "fmt" + "sort" + "strings" + "time" +) + +// State is the canonical instance state (P1.2a). It is distinct from the raw, +// unnormalized vocabularies the API uses at the rule level and at the instance +// level — see normalizeInstanceState. +type State string + +const ( + StateNormal State = "normal" + StateFiring State = "firing" + StatePending State = "pending" + StateNodata State = "nodata" + StateError State = "error" +) + +// Instance is one entry of a rule's alerts[]. State is always canonical; Reason +// is the opaque suffix of a "State (Reason)" composite ("" when the API gave a +// bare state). Reason is reporting-only except for the H2 MissingSeries routing +// done downstream in the log markers. +type Instance struct { + Labels map[string]string + State State + Reason string + ActiveAt time.Time + Value string +} + +// StateRule is one rule from the state endpoint +// (/api/prometheus/grafana/api/v1/rules), fully and strictly parsed (H1). +type StateRule struct { + UID, Title, Folder, Group string + Interval time.Duration + // State and Health are raw, lowercase, and reporting-only — never + // classified (P1.2a). State in particular is never normalized. + State, Health string + LastError string + LastEvaluation time.Time + IsPaused bool + Instances []Instance + Totals map[string]int +} + +// ParseState strictly parses a state-endpoint response body into its rules. +// A missing or unparseable required field (health, state, lastEvaluation on +// each rule; interval on each group) is an error, never a zero value (H1). +func ParseState(body []byte) ([]StateRule, error) { + var top map[string]json.RawMessage + if err := json.Unmarshal(body, &top); err != nil { + return nil, fmt.Errorf("state response: %w", err) + } + + var dataRaw json.RawMessage + if err := req(top, "data", &dataRaw); err != nil { + return nil, fmt.Errorf("state response: %w", err) + } + var data map[string]json.RawMessage + if err := json.Unmarshal(dataRaw, &data); err != nil { + return nil, fmt.Errorf("state response: data: %w", err) + } + + var groupsRaw []json.RawMessage + if err := req(data, "groups", &groupsRaw); err != nil { + return nil, fmt.Errorf("state response: %w", err) + } + + var rules []StateRule + for gi, groupRaw := range groupsRaw { + var group map[string]json.RawMessage + if err := json.Unmarshal(groupRaw, &group); err != nil { + return nil, fmt.Errorf("state response: group %d: %w", gi, err) + } + + var folder, groupName string + if err := req(group, "file", &folder); err != nil { + return nil, fmt.Errorf("state response: group %d: %w", gi, err) + } + if err := req(group, "name", &groupName); err != nil { + return nil, fmt.Errorf("state response: group %d (folder %q): %w", gi, folder, err) + } + + var intervalSeconds float64 + if err := req(group, "interval", &intervalSeconds); err != nil { + return nil, fmt.Errorf("state response: group %q: %w", groupName, err) + } + interval := time.Duration(intervalSeconds * float64(time.Second)) + + var rulesRaw []json.RawMessage + if err := req(group, "rules", &rulesRaw); err != nil { + return nil, fmt.Errorf("state response: group %q: %w", groupName, err) + } + + for ri, ruleRaw := range rulesRaw { + rule, err := parseStateRule(ruleRaw, folder, groupName, interval) + if err != nil { + return nil, fmt.Errorf("state response: group %q (folder %q): rule %d: %w", groupName, folder, ri, err) + } + rules = append(rules, rule) + } + } + return rules, nil +} + +func parseStateRule(raw json.RawMessage, folder, group string, interval time.Duration) (StateRule, error) { + var m map[string]json.RawMessage + if err := json.Unmarshal(raw, &m); err != nil { + return StateRule{}, fmt.Errorf("rule: %w", err) + } + + var uid, name string + if err := req(m, "uid", &uid); err != nil { + return StateRule{}, fmt.Errorf("rule: %w", err) + } + if err := req(m, "name", &name); err != nil { + return StateRule{}, fmt.Errorf("rule %q: %w", uid, err) + } + + r := StateRule{UID: uid, Title: name, Folder: folder, Group: group, Interval: interval} + + if err := req(m, "state", &r.State); err != nil { + return StateRule{}, fmt.Errorf("rule %q: %w", uid, err) + } + if err := req(m, "health", &r.Health); err != nil { + return StateRule{}, fmt.Errorf("rule %q: %w", uid, err) + } + // isPaused is not one of H1's four named required fields, but this parser + // extends that contract to it: the zero-time rule below can't tell a + // paused rule from a broken one without it, and it's the primary + // in-window pause detector (H2/§12.2) — a silent false default would be + // exactly the fail-open bug H1 exists to kill. + if err := req(m, "isPaused", &r.IsPaused); err != nil { + return StateRule{}, fmt.Errorf("rule %q: %w", uid, err) + } + + var lastEvalStr string + if err := req(m, "lastEvaluation", &lastEvalStr); err != nil { + return StateRule{}, fmt.Errorf("rule %q: %w", uid, err) + } + lastEval, err := time.Parse(time.RFC3339, lastEvalStr) + if err != nil { + return StateRule{}, fmt.Errorf("rule %q: lastEvaluation: %w", uid, err) + } + // The zero-time rule (§2.3): only a paused rule may report the zero time. + if lastEval.IsZero() && !r.IsPaused { + return StateRule{}, fmt.Errorf("rule %q: lastEvaluation is the zero time but isPaused is false", uid) + } + r.LastEvaluation = lastEval + + if err := opt(m, "lastError", &r.LastError); err != nil { + return StateRule{}, fmt.Errorf("rule %q: %w", uid, err) + } + + if err := opt(m, "totals", &r.Totals); err != nil { + return StateRule{}, fmt.Errorf("rule %q: %w", uid, err) + } + + var alertsRaw []json.RawMessage + if err := opt(m, "alerts", &alertsRaw); err != nil { + return StateRule{}, fmt.Errorf("rule %q: %w", uid, err) + } + if alertsRaw != nil { + instances := make([]Instance, 0, len(alertsRaw)) + for ii, ar := range alertsRaw { + inst, err := parseInstance(ar) + if err != nil { + return StateRule{}, fmt.Errorf("rule %q: instance %d: %w", uid, ii, err) + } + instances = append(instances, inst) + } + r.Instances = instances + } + + return r, nil +} + +func parseInstance(raw json.RawMessage) (Instance, error) { + var m map[string]json.RawMessage + if err := json.Unmarshal(raw, &m); err != nil { + return Instance{}, fmt.Errorf("%w", err) + } + + var rawState string + if err := req(m, "state", &rawState); err != nil { + return Instance{}, err + } + state, reason, err := normalizeInstanceState(rawState) + if err != nil { + return Instance{}, err + } + + // activeAt is also not in H1's named list, extended here for the same + // reason as StateRule.IsPaused: it's the onset time BadFor (P8) measures + // from, so a silently zeroed one would misclassify how long an instance + // has been bad rather than failing loudly. + var activeAtStr string + if err := req(m, "activeAt", &activeAtStr); err != nil { + return Instance{}, err + } + activeAt, err := time.Parse(time.RFC3339, activeAtStr) + if err != nil { + return Instance{}, fmt.Errorf("activeAt: %w", err) + } + + inst := Instance{State: state, Reason: reason, ActiveAt: activeAt} + + if err := opt(m, "value", &inst.Value); err != nil { + return Instance{}, err + } + if err := opt(m, "labels", &inst.Labels); err != nil { + return Instance{}, err + } + + return inst, nil +} + +// baseInstanceStates is the strict 5-value allowlist for the base of an +// instance state (P1.2a). Anything else — including an unrecognized base +// inside a "Base (Reason)" composite — is a parse error (H1, §2.7 control 3). +var baseInstanceStates = map[string]State{ + "Normal": StateNormal, + "Alerting": StateFiring, + "Pending": StatePending, + "NoData": StateNodata, + "Error": StateError, +} + +// normalizeInstanceState normalizes an instance-level state string into its +// canonical base and an opaque reason. The composite "State (Reason)" form is +// parsed structurally — split on the first " (" with a trailing ")" — never by +// enumerating composites, because Grafana's reason vocabulary grows across +// versions and an unknown reason must not break parsing. +func normalizeInstanceState(s string) (State, string, error) { + base, reason := s, "" + if i := strings.Index(s, " ("); i != -1 && strings.HasSuffix(s, ")") { + base = s[:i] + reason = s[i+2 : len(s)-1] + } + state, ok := baseInstanceStates[base] + if !ok { + return "", "", fmt.Errorf("unrecognized instance state %q", s) + } + return state, reason, nil +} + +// instanceKey is a stable identity for an instance's label set: a sorted +// "k=v\n" join. Used to correlate an instance across polls without hashing. +func instanceKey(labels map[string]string) string { + keys := make([]string, 0, len(labels)) + for k := range labels { + keys = append(keys, k) + } + sort.Strings(keys) + + var b strings.Builder + for _, k := range keys { + b.WriteString(k) + b.WriteByte('=') + b.WriteString(labels[k]) + b.WriteByte('\n') + } + return b.String() +} diff --git a/grafana-alertcheck/internal/gate/parse_state_test.go b/grafana-alertcheck/internal/gate/parse_state_test.go new file mode 100644 index 000000000..7cf336526 --- /dev/null +++ b/grafana-alertcheck/internal/gate/parse_state_test.go @@ -0,0 +1,415 @@ +package gate + +import ( + "bytes" + "encoding/json" + "fmt" + "maps" + "os" + "path/filepath" + "strings" + "testing" + "time" +) + +func readFixture(t *testing.T, name string) []byte { + t.Helper() + b, err := os.ReadFile(filepath.Join("testdata", name)) + if err != nil { + t.Fatalf("reading fixture %s: %v", name, err) + } + return b +} + +func TestParseState_HappyPaths(t *testing.T) { + cases := []struct { + fixture string + wantRules int + wantInstances int + checkFirst func(t *testing.T, r StateRule) + }{ + { + fixture: "state_one_instance.json", + wantRules: 1, + wantInstances: 1, + checkFirst: func(t *testing.T, r StateRule) { + if r.UID != "rule0000001" { + t.Errorf("UID = %q, want rule0000001", r.UID) + } + if r.Folder != "ExampleTeam" || r.Group != "Example Service - Prod" { + t.Errorf("Folder/Group = %q/%q, want ExampleTeam/Example Service - Prod", r.Folder, r.Group) + } + if r.Health != "ok" || r.State != "inactive" { + t.Errorf("Health/State = %q/%q, want ok/inactive", r.Health, r.State) + } + if r.Interval.Seconds() != 60 { + t.Errorf("Interval = %v, want 60s", r.Interval) + } + if r.IsPaused { + t.Errorf("IsPaused = true, want false") + } + if len(r.Instances) != 1 || r.Instances[0].State != StateNormal { + t.Fatalf("Instances = %+v, want one normal instance", r.Instances) + } + + inst := r.Instances[0] + wantLabels := map[string]string{ + "alertname": "Example High CPU Usage", + "env": "prod", + "grafana_folder": "ExampleTeam", + "app_instance": "example-app-instance", + "service": "example-svc", + "severity": "critical", + "team": "example-team", + } + if !maps.Equal(inst.Labels, wantLabels) { + t.Errorf("Labels = %+v, want %+v", inst.Labels, wantLabels) + } + wantActiveAt, err := time.Parse(time.RFC3339, "2026-08-31T08:02:50Z") + if err != nil { + t.Fatalf("test setup: %v", err) + } + if !inst.ActiveAt.Equal(wantActiveAt) { + t.Errorf("ActiveAt = %v, want %v", inst.ActiveAt, wantActiveAt) + } + if inst.Value != "" { + t.Errorf("Value = %q, want empty string", inst.Value) + } + }, + }, + { + fixture: "state_paused.json", + wantRules: 1, + wantInstances: 0, + checkFirst: func(t *testing.T, r StateRule) { + if !r.IsPaused { + t.Errorf("IsPaused = false, want true") + } + if !r.LastEvaluation.IsZero() { + t.Errorf("LastEvaluation = %v, want zero time", r.LastEvaluation) + } + if r.Health != "ok" || r.State != "inactive" { + t.Errorf("Health/State = %q/%q, want ok/inactive", r.Health, r.State) + } + }, + }, + { + fixture: "state_health_error.json", + wantRules: 1, + wantInstances: 1, + checkFirst: func(t *testing.T, r StateRule) { + if r.Health != "error" { + t.Errorf("Health = %q, want error", r.Health) + } + if r.LastError == "" { + t.Errorf("LastError is empty, want a message") + } + if len(r.Instances) != 1 || r.Instances[0].State != StateError { + t.Fatalf("Instances = %+v, want one error instance", r.Instances) + } + }, + }, + { + fixture: "state_health_nodata.json", + wantRules: 1, + wantInstances: 1, + checkFirst: func(t *testing.T, r StateRule) { + if r.Health != "nodata" { + t.Errorf("Health = %q, want nodata", r.Health) + } + if len(r.Instances) != 1 || r.Instances[0].State != StateNodata { + t.Fatalf("Instances = %+v, want one nodata instance", r.Instances) + } + }, + }, + { + fixture: "state_reason_composite.json", + wantRules: 1, + wantInstances: 3, + checkFirst: func(t *testing.T, r StateRule) { + byReason := map[string]Instance{} + for _, inst := range r.Instances { + byReason[inst.Reason] = inst + } + errInst, ok := byReason["Error"] + if !ok || errInst.State != StateNormal { + t.Errorf(`want an instance with State=normal Reason="Error", got %+v`, byReason["Error"]) + } + nodataInst, ok := byReason["NoData"] + if !ok || nodataInst.State != StateNormal { + t.Errorf(`want an instance with State=normal Reason="NoData", got %+v`, byReason["NoData"]) + } + plain, ok := byReason[""] + if !ok || plain.State != StateNormal { + t.Errorf(`want a plain State=normal Reason="" instance, got %+v`, byReason[""]) + } + }, + }, + { + fixture: "state_missing_optional.json", + wantRules: 1, + wantInstances: 0, + checkFirst: func(t *testing.T, r StateRule) { + if r.Instances != nil { + t.Errorf("Instances = %+v, want nil", r.Instances) + } + if r.Totals != nil { + t.Errorf("Totals = %+v, want nil", r.Totals) + } + }, + }, + { + fixture: "state_only_active_instances.json", + wantRules: 1, + wantInstances: 1, + checkFirst: func(t *testing.T, r StateRule) { + if len(r.Instances) != 1 || r.Instances[0].State != StateFiring { + t.Fatalf("Instances = %+v, want one firing instance", r.Instances) + } + if r.Totals["normal"] == 0 { + t.Errorf(`Totals["normal"] = 0, want >0 (this is the §3.2 mismatch the fixture exists to capture)`) + } + }, + }, + } + + for _, c := range cases { + t.Run(c.fixture, func(t *testing.T) { + rules, err := ParseState(readFixture(t, c.fixture)) + if err != nil { + t.Fatalf("ParseState(%s): unexpected error: %v", c.fixture, err) + } + if len(rules) != c.wantRules { + t.Fatalf("ParseState(%s): got %d rules, want %d", c.fixture, len(rules), c.wantRules) + } + if got := len(rules[0].Instances); got != c.wantInstances { + t.Fatalf("ParseState(%s): got %d instances, want %d", c.fixture, got, c.wantInstances) + } + if c.checkFirst != nil { + c.checkFirst(t, rules[0]) + } + }) + } +} + +// TestParseState_MustError is the H1 regression suite: it doesn't just check +// err != nil (a stray comma in a fixture would keep that green forever while +// the actual check regressed) — it asserts the error names the specific +// offending field or value, so a real H1 check going missing fails loudly +// here instead of surviving unnoticed. +func TestParseState_MustError(t *testing.T) { + cases := []struct { + fixture string + wantContains []string + }{ + {"state_missing_health.json", []string{`"health"`}}, + {"state_missing_lasteval.json", []string{`"lastEvaluation"`}}, + {"state_missing_state.json", []string{`"state"`}}, + {"state_missing_interval.json", []string{`"interval"`}}, + {"state_missing_file.json", []string{`"file"`}}, + {"state_missing_name.json", []string{`"name"`}}, + {"state_zerotime_unpaused.json", []string{"zero time", "isPaused"}}, + {"state_unknown_state.json", []string{`"Weird (NoData)"`}}, + } + for _, c := range cases { + t.Run(c.fixture, func(t *testing.T) { + _, err := ParseState(readFixture(t, c.fixture)) + if err == nil { + t.Fatalf("ParseState(%s): expected an error, got none", c.fixture) + } + for _, want := range c.wantContains { + if !strings.Contains(err.Error(), want) { + t.Errorf("ParseState(%s): error %q does not mention %q", c.fixture, err.Error(), want) + } + } + }) + } +} + +func TestParseNormalizeInstanceState(t *testing.T) { + cases := []struct { + in string + wantState State + wantReason string + wantErr bool + }{ + {"Normal", StateNormal, "", false}, + {"Alerting", StateFiring, "", false}, + {"Pending", StatePending, "", false}, + {"NoData", StateNodata, "", false}, + {"Error", StateError, "", false}, + {"Normal (NoData)", StateNormal, "NoData", false}, + {"Normal (Error)", StateNormal, "Error", false}, + {"Normal (MissingSeries)", StateNormal, "MissingSeries", false}, + {"Weird (NoData)", "", "", true}, + {"Weird", "", "", true}, + {"", "", "", true}, + } + for _, c := range cases { + state, reason, err := normalizeInstanceState(c.in) + if c.wantErr { + if err == nil { + t.Errorf("normalizeInstanceState(%q): expected an error, got none", c.in) + } + continue + } + if err != nil { + t.Errorf("normalizeInstanceState(%q): unexpected error: %v", c.in, err) + continue + } + if state != c.wantState || reason != c.wantReason { + t.Errorf("normalizeInstanceState(%q) = (%q, %q), want (%q, %q)", c.in, state, reason, c.wantState, c.wantReason) + } + } +} + +func TestInstanceKey(t *testing.T) { + a := instanceKey(map[string]string{"b": "2", "a": "1"}) + b := instanceKey(map[string]string{"a": "1", "b": "2"}) + if a != b { + t.Errorf("instanceKey order-independence: %q != %q", a, b) + } + if a != "a=1\nb=2\n" { + t.Errorf("instanceKey = %q, want a=1\\nb=2\\n", a) + } + + diff := instanceKey(map[string]string{"a": "1", "b": "3"}) + if a == diff { + t.Errorf("instanceKey should differ when a label value differs") + } + + if instanceKey(nil) != "" { + t.Errorf("instanceKey(nil) = %q, want empty string", instanceKey(nil)) + } +} + +// synthesizeHighCardinalityState builds a state response with a single rule +// holding `alerting` Alerting instances and `normal` Normal instances, by +// cloning the one real instance in state_one_instance.json. It is never +// committed (§3.2, §22.3, §22.6) — the 2446-instance rule this stands in for +// is ~600 KB and exists only to prove the parser and (in later phases) the +// reducer don't choke on real fleet cardinality. +func synthesizeHighCardinalityState(t *testing.T, alerting, normal int) []byte { + t.Helper() + base := readFixture(t, "state_one_instance.json") + + var top map[string]json.RawMessage + if err := json.Unmarshal(base, &top); err != nil { + t.Fatalf("synthesize: %v", err) + } + var data map[string]json.RawMessage + if err := json.Unmarshal(top["data"], &data); err != nil { + t.Fatalf("synthesize: %v", err) + } + var groups []map[string]json.RawMessage + if err := json.Unmarshal(data["groups"], &groups); err != nil { + t.Fatalf("synthesize: %v", err) + } + var rules []map[string]json.RawMessage + if err := json.Unmarshal(groups[0]["rules"], &rules); err != nil { + t.Fatalf("synthesize: %v", err) + } + var alerts []map[string]json.RawMessage + if err := json.Unmarshal(rules[0]["alerts"], &alerts); err != nil { + t.Fatalf("synthesize: %v", err) + } + template := alerts[0] + + newAlerts := make([]map[string]json.RawMessage, 0, alerting+normal) + for i := range alerting { + inst := cloneRawMap(template) + inst["state"] = mustRaw(t, "Alerting") + inst["labels"] = mustRaw(t, map[string]string{"instance": fmt.Sprintf("alerting-%d", i)}) + newAlerts = append(newAlerts, inst) + } + for i := range normal { + inst := cloneRawMap(template) + inst["state"] = mustRaw(t, "Normal") + inst["labels"] = mustRaw(t, map[string]string{"instance": fmt.Sprintf("normal-%d", i)}) + newAlerts = append(newAlerts, inst) + } + + rules[0]["alerts"] = mustRaw(t, newAlerts) + rules[0]["totals"] = mustRaw(t, map[string]int{"alerting": alerting, "normal": normal}) + rules[0]["totalsFiltered"] = rules[0]["totals"] + groups[0]["rules"] = mustRaw(t, rules) + data["groups"] = mustRaw(t, groups) + top["data"] = mustRaw(t, data) + + out, err := json.Marshal(top) + if err != nil { + t.Fatalf("synthesize: %v", err) + } + return out +} + +func cloneRawMap(m map[string]json.RawMessage) map[string]json.RawMessage { + out := make(map[string]json.RawMessage, len(m)) + for k, v := range m { + cp := make(json.RawMessage, len(v)) + copy(cp, v) + out[k] = cp + } + return out +} + +func mustRaw(t *testing.T, v any) json.RawMessage { + t.Helper() + b, err := json.Marshal(v) + if err != nil { + t.Fatalf("marshal: %v", err) + } + return json.RawMessage(b) +} + +func TestParseState_HighCardinality(t *testing.T) { + body := synthesizeHighCardinalityState(t, 445, 2004) + if !bytes.Contains(body, []byte("alerting-0")) { + t.Fatalf("synthesized body missing expected content") + } + + rules, err := ParseState(body) + if err != nil { + t.Fatalf("ParseState: unexpected error: %v", err) + } + if len(rules) != 1 { + t.Fatalf("got %d rules, want 1", len(rules)) + } + r := rules[0] + if len(r.Instances) != 445+2004 { + t.Fatalf("got %d instances, want %d", len(r.Instances), 445+2004) + } + + var firing, normal int + for _, inst := range r.Instances { + switch inst.State { + case StateFiring: + firing++ + case StateNormal: + normal++ + default: + t.Fatalf("unexpected instance state %q", inst.State) + } + } + if firing != 445 || normal != 2004 { + t.Fatalf("got firing=%d normal=%d, want firing=445 normal=2004", firing, normal) + } + + // Each synthesized instance carries a distinct "instance" label; confirm + // Labels actually made it through parsing (not just State) by checking + // instanceKey produces one unique key per instance, with no collisions. + seen := make(map[string]bool, len(r.Instances)) + for _, inst := range r.Instances { + if inst.Labels == nil { + t.Fatalf("instance has nil Labels") + } + k := instanceKey(inst.Labels) + if seen[k] { + t.Fatalf("duplicate instance key %q", k) + } + seen[k] = true + } + if len(seen) != 445+2004 { + t.Fatalf("got %d unique instance keys, want %d", len(seen), 445+2004) + } +} diff --git a/grafana-alertcheck/internal/gate/testdata/README.md b/grafana-alertcheck/internal/gate/testdata/README.md new file mode 100644 index 000000000..fb437b3ae --- /dev/null +++ b/grafana-alertcheck/internal/gate/testdata/README.md @@ -0,0 +1,91 @@ +# Fixture provenance + +All fixtures are sanitized slices of the real Grafana 13.1.0 payloads captured next to the plan in +`tmp/` (`tmp/state_all.json`, `tmp/ruler_all.json`, `tmp/health.json` — gitignored, never committed). +Renames are consistent across files: the same real folder/rule keeps the same fake identity everywhere +it appears (e.g. `folder0000002`/`rule0000002` is the same real paused rule in both +`state_paused.json` and `ruler_rules.json`). + +An earlier revision embedded these notes as a top-level `_fixture_note` JSON key. That works for the +state-endpoint fixtures (the parser decodes the top level generically, so a stray key is just ignored), +but breaks the ruler-endpoint fixtures: `ParseDefinitions` decodes the whole body directly as +`map[string][]group`, and a `_fixture_note` string value can't unmarshal as `[]group`. Notes now live +here instead, for every fixture, for consistency. + +## State endpoint (`/api/prometheus/grafana/api/v1/rules`) + +- **state_one_instance.json** — real rule (`bfhp23rgt18u8f`, folder `BCM`, "[PROD][ACE] High CPU + Utilization") with exactly one live instance. Renamed to folder `ExampleTeam`/`folder0000001`, rule + `rule0000001`/"Example High CPU Usage", datasource `datasource000001`. Also the template the + high-cardinality test helper (`synthesizeHighCardinalityState`) clones from. +- **state_paused.json** — one of the 3 real `isPaused:true` rules (`ac153fee-...`, folder `Mercury`, + "Rubberbanding"). Renamed to folder `ExampleMetrics`/`folder0000002`, rule + `rule0000002`/"Example Paused Rule". Unmodified: `isPaused:true`, zero `lastEvaluation`, `health:ok`, + `state:inactive`, absent `alerts`/`labels`. +- **state_health_error.json** — real `health:error` rule ("[JD] No Job Proposals", folder + `job-distributor`), highest priority per §22.1. Renamed to folder `ExampleService`/`folder0000003`, + rule `rule0000003`/"Example No Data Source". Unmodified: `health:error`, `lastError` text, the single + `Error` instance. +- **state_health_nodata.json** — real `health:nodata` rule ("ARE test", folder `diegos_playground`). + Renamed to folder `ExamplePlayground`/`folder0000004`, rule `rule0000004`/"Example NoData Rule". + Unmodified: `health:nodata`, the single `NoData` instance. +- **state_reason_composite.json** — composite of two real instances combined under one rule for P1.2a + coverage: a real `"Normal (Error)"` instance (from a Flux-reconciliation rule; 14 of that state exist + in the capture) and a real `"Normal (NoData)"` instance (from a pod-liveness rule; 1091 of that state + exist), plus one plain `"Normal"` instance for contrast. Renamed to folder + `ExampleInfra`/`folder0000005`, rule `rule0000005`/"Example Composite Reasons". +- **state_missing_optional.json** — derived from `state_one_instance.json`: `alerts`, `totals`, + `totalsFiltered` and `labels` all removed. Must parse with `Instances=nil`, `Totals=nil`. +- **state_missing_health.json** — derived from `state_one_instance.json`: the required `health` key + removed. Must be a parse error (H1). +- **state_missing_lasteval.json** — derived from `state_one_instance.json`: the required + `lastEvaluation` key removed. Must be a parse error (H1). +- **state_missing_state.json** — derived from `state_one_instance.json`: the required rule-level + `state` key removed. Must be a parse error (H1). Closes must-error coverage for H1's four required + fields — a review pass found `health`/`lastEvaluation` covered but `state`/`interval` weren't, even + though the code already `req`'d them correctly. +- **state_missing_interval.json** — derived from `state_one_instance.json`: the required group-level + `interval` key removed. Must be a parse error (H1); same review-pass gap as above. +- **state_missing_file.json** / **state_missing_name.json** — derived from `state_one_instance.json`: + the group-level `file`/`name` keys removed respectively. Not part of H1's four (those are `health`, + `state`, `lastEvaluation`, `interval`), but the code treats group identity as strict too, and the same + review pass flagged the gap — closed rather than deferred to a later §22 sweep since the fixture is + the same 10-line edit. +- **state_zerotime_unpaused.json** — derived from `state_one_instance.json`: `lastEvaluation` set to + the zero time while `isPaused` stays `false`. Must be a parse error (§2.3). +- **state_unknown_state.json** — derived from `state_one_instance.json`: the instance state hand-edited + to `"Weird (NoData)"`, a syntactically valid composite whose base isn't in the 5-value allowlist. Must + be a parse error (P1.2a). +- **state_only_active_instances.json** — derived from a real rule that genuinely had 1 `Alerting` + 22 + `Normal` instances (`totals: {alerting:1, normal:22}`, rule `dfhp1t5pkosu8f`, folder `BCM`). `alerts[]` + trimmed to the single `Alerting` instance only, while `totals` is left **unchanged** — reproducing the + §3.2 violation shape (instance list says "only active" while totals disagrees). Renamed to folder + `ExampleTeam`/`folder0000001`, rule `rule0000006`. `ParseState` itself parses this fine; the §3.2 + verification lives in a later phase (P5/P9). + +## Ruler endpoint (`/api/ruler/grafana/api/v1/rules`) + +- **ruler_rules.json** — contains: + - The real true 2-way title collision: namespace `CRE-BCM-Prod-Zone-A`, group `Gateway`, identical + folder+group+title, distinct UIDs (`ffvabtvvbozcwf`/`efvabtwbxlvk0b`) — renamed to namespace + `Example-Zone-A`, rules `rule0000006a`/`rule0000006b`, both titled "Example No Gateways Available". + Folder/Group/Title alone does **not** disambiguate this pair (§17, §22.2). + - The 3 real `is_paused:true` rules, renamed to `rule0000002`/`rule0000007`/`rule0000008`. + `rule0000002` intentionally shares its identity (`folder0000002`) with `state_paused.json`. + - A real `for:1d` rule (`afs438kjd4v7kd` → `rule0000009`). + - **`rule0000010` is DERIVED**: no `for:1w` rule exists anywhere in the capture (verified). Built by + copying the `for:1d` rule and changing `for` to `1w` and its identity, to exercise the `w` unit. +- **ruler_datasource_managed.json** — **DERIVED**, no datasource-managed rule exists in the capture + (verified: 0 rules lack a `grafana_alert` block). Hand-built minimal shape: a rule object with no + `grafana_alert` key at all and an `alert` field carrying its Prometheus-format name — exactly how + Grafana represents a datasource-managed (native Prometheus-format) alerting rule. `ParseDefinitions` + must classify it as `KindDatasourceManaged`, parse `Title` from `alert`, and leave `UID` empty (this + shape has no uid at all — inventing one would be inventing shape) without rejecting the rule + (rejection is P3's job, only for rules a user actually named). +- **ruler_recording.json** — **DERIVED**, no recording rule exists in the capture (verified: 0 rules + carry `grafana_alert.record`). Hand-built: a `grafana_alert` block with a `record` sub-object but + deliberately *without* `no_data_state`/`exec_err_state`/`is_paused`/`intervalSeconds`/`namespace_uid` + — those are alerting-only concepts a recording rule may not carry, and since none exist in the + capture that shape is unverified either way. `ParseDefinitions` must classify it as `KindRecording` + and must not require those fields for this Kind (requiring them bricks `ParseDefinitions` for every + named rule in the same response over one recording rule elsewhere in the fleet). diff --git a/grafana-alertcheck/internal/gate/testdata/ruler_datasource_managed.json b/grafana-alertcheck/internal/gate/testdata/ruler_datasource_managed.json new file mode 100644 index 000000000..a909f7754 --- /dev/null +++ b/grafana-alertcheck/internal/gate/testdata/ruler_datasource_managed.json @@ -0,0 +1,21 @@ +{ + "ExampleMetrics": [ + { + "name": "Datasource Managed Group", + "interval": "1m", + "rules": [ + { + "alert": "ExampleTargetDown", + "expr": "up == 0", + "for": "5m", + "labels": { + "severity": "warning" + }, + "annotations": { + "summary": "target down" + } + } + ] + } + ] +} diff --git a/grafana-alertcheck/internal/gate/testdata/ruler_recording.json b/grafana-alertcheck/internal/gate/testdata/ruler_recording.json new file mode 100644 index 000000000..07881626c --- /dev/null +++ b/grafana-alertcheck/internal/gate/testdata/ruler_recording.json @@ -0,0 +1,44 @@ +{ + "ExampleMetrics": [ + { + "name": "Recording Group", + "interval": "1m", + "rules": [ + { + "expr": "", + "for": "0s", + "labels": {}, + "annotations": {}, + "grafana_alert": { + "title": "example:recorded_metric:rate5m", + "condition": "", + "data": [ + { + "refId": "A", + "queryType": "", + "relativeTimeRange": { + "from": 600, + "to": 0 + }, + "datasourceUid": "datasource000010", + "model": { + "expr": "rate(example_metric_total[5m])", + "refId": "A" + } + } + ], + "record": { + "metric": "example:recorded_metric:rate5m", + "from": "A" + }, + "updated": "2026-08-01T00:00:00Z", + "version": 1, + "uid": "rule0000011", + "rule_group": "Recording Group", + "guid": "example-guid-0011" + } + } + ] + } + ] +} diff --git a/grafana-alertcheck/internal/gate/testdata/ruler_rules.json b/grafana-alertcheck/internal/gate/testdata/ruler_rules.json new file mode 100644 index 000000000..4eceeadce --- /dev/null +++ b/grafana-alertcheck/internal/gate/testdata/ruler_rules.json @@ -0,0 +1,332 @@ +{ + "Example-Zone-A": [ + { + "name": "Gateway", + "interval": "1m", + "rules": [ + { + "expr": "", + "for": "5m", + "keep_firing_for": "0s", + "labels": { + "env": "production", + "severity": "critical", + "team": "example-team", + "zone": "zone-a" + }, + "annotations": { + "description": "Node(s) have no gateways configured.", + "runbook_url": "https://example.com/runbook", + "summary": "No gateways available for 5m" + }, + "grafana_alert": { + "title": "Example No Gateways Available", + "condition": "C", + "data": [ + { + "refId": "A", + "queryType": "", + "relativeTimeRange": { + "from": 600, + "to": 0 + }, + "datasourceUid": "datasource000006", + "model": { + "expr": "sum(example_no_gateways_available_count[5m])", + "refId": "A" + } + } + ], + "updated": "2026-08-20T09:54:36Z", + "intervalSeconds": 60, + "version": 16, + "uid": "rule0000006a", + "namespace_uid": "folder0000006", + "rule_group": "Gateway", + "no_data_state": "OK", + "exec_err_state": "OK", + "is_paused": false, + "guid": "example-guid-0006a" + } + }, + { + "expr": "", + "for": "5m", + "keep_firing_for": "0s", + "labels": { + "env": "production", + "severity": "critical", + "team": "example-team", + "zone": "zone-a" + }, + "annotations": { + "description": "Node(s) have no gateways configured.", + "runbook_url": "https://example.com/runbook", + "summary": "No gateways available for 5m" + }, + "grafana_alert": { + "title": "Example No Gateways Available", + "condition": "C", + "data": [ + { + "refId": "A", + "queryType": "", + "relativeTimeRange": { + "from": 600, + "to": 0 + }, + "datasourceUid": "datasource000006", + "model": { + "expr": "sum(example_no_gateways_available_count[5m])", + "refId": "A" + } + } + ], + "updated": "2026-08-20T09:54:40Z", + "intervalSeconds": 60, + "version": 3, + "uid": "rule0000006b", + "namespace_uid": "folder0000006", + "rule_group": "Gateway", + "no_data_state": "OK", + "exec_err_state": "OK", + "is_paused": false, + "guid": "example-guid-0006b" + } + } + ] + }, + { + "name": "EVM Capabilities", + "interval": "1m", + "rules": [ + { + "expr": "", + "for": "1d", + "keep_firing_for": "0s", + "labels": { + "env": "production", + "severity": "warning", + "team": "example-team" + }, + "annotations": { + "description": "Chain has an elevated failure ratio.", + "summary": "Failure ratio > 10% for 1d" + }, + "grafana_alert": { + "title": "Example Failure Ratio Above 10 Percent", + "condition": "C", + "data": [ + { + "refId": "A", + "queryType": "", + "relativeTimeRange": { + "from": 600, + "to": 0 + }, + "datasourceUid": "datasource000009", + "model": { + "expr": "sum(example_failure_ratio)", + "refId": "A" + } + } + ], + "updated": "2026-08-10T17:24:04Z", + "intervalSeconds": 60, + "version": 5, + "uid": "rule0000009", + "namespace_uid": "folder0000006", + "rule_group": "EVM Capabilities", + "no_data_state": "OK", + "exec_err_state": "OK", + "is_paused": false, + "guid": "example-guid-0009" + } + }, + { + "expr": "", + "for": "1w", + "keep_firing_for": "0s", + "labels": { + "env": "production", + "severity": "warning", + "team": "example-team" + }, + "annotations": { + "description": "DERIVED fixture: no real for:1w rule exists in the capture — copied from the for:1d rule above with 'for' changed to exercise the w unit.", + "summary": "Failure ratio elevated for 1w" + }, + "grafana_alert": { + "title": "Example Failure Ratio Above 10 Percent Weekly", + "condition": "C", + "data": [ + { + "refId": "A", + "queryType": "", + "relativeTimeRange": { + "from": 600, + "to": 0 + }, + "datasourceUid": "datasource000009", + "model": { + "expr": "sum(example_failure_ratio)", + "refId": "A" + } + } + ], + "updated": "2026-08-10T17:24:04Z", + "intervalSeconds": 60, + "version": 1, + "uid": "rule0000010", + "namespace_uid": "folder0000006", + "rule_group": "EVM Capabilities", + "no_data_state": "OK", + "exec_err_state": "OK", + "is_paused": false, + "guid": "example-guid-0010" + } + } + ] + } + ], + "ExampleObservability": [ + { + "name": "Example Auth Production", + "interval": "1m", + "rules": [ + { + "expr": "", + "for": "5m", + "keep_firing_for": "0s", + "labels": { + "env": "production" + }, + "annotations": {}, + "grafana_alert": { + "title": "example_workflow_paused_rule", + "condition": "C", + "data": [ + { + "refId": "A", + "queryType": "", + "relativeTimeRange": { + "from": 600, + "to": 0 + }, + "datasourceUid": "datasource000007", + "model": { + "expr": "sum(example_auth_client_requests)", + "refId": "A" + } + } + ], + "updated": "2026-08-01T00:00:00Z", + "intervalSeconds": 60, + "version": 1, + "uid": "rule0000007", + "namespace_uid": "folder0000007", + "rule_group": "Example Auth Production", + "no_data_state": "Alerting", + "exec_err_state": "Error", + "is_paused": true, + "guid": "example-guid-0007" + } + } + ] + } + ], + "ExampleFeeds": [ + { + "name": "Example Feeds Annotations", + "interval": "30s", + "rules": [ + { + "expr": "", + "for": "1m", + "keep_firing_for": "2m", + "labels": { + "env": "production", + "severity": "critical", + "team": "example-feeds-pm" + }, + "annotations": { + "summary": "[TEST ONLY] example depeg alert" + }, + "grafana_alert": { + "title": "TEMP - Example depeg alert", + "condition": "C", + "data": [ + { + "refId": "A", + "queryType": "", + "relativeTimeRange": { + "from": 600, + "to": 0 + }, + "datasourceUid": "datasource000008", + "model": { + "expr": "avg(example_feed_answer)", + "refId": "A" + } + } + ], + "updated": "2026-08-01T00:00:00Z", + "intervalSeconds": 30, + "version": 1, + "uid": "rule0000008", + "namespace_uid": "folder0000008", + "rule_group": "Example Feeds Annotations", + "no_data_state": "NoData", + "exec_err_state": "Error", + "is_paused": true, + "guid": "example-guid-0008" + } + } + ] + } + ], + "ExampleMetrics": [ + { + "name": "Example Paused Rule", + "interval": "5m", + "rules": [ + { + "expr": "", + "for": "5m", + "keep_firing_for": "0s", + "labels": {}, + "annotations": {}, + "grafana_alert": { + "title": "Example Paused Rule", + "condition": "C", + "data": [ + { + "refId": "A", + "queryType": "", + "relativeTimeRange": { + "from": 600, + "to": 0 + }, + "datasourceUid": "datasource000002", + "model": { + "expr": "example_metric{env=\"production\"}", + "refId": "A" + } + } + ], + "updated": "2026-08-01T00:00:00Z", + "intervalSeconds": 300, + "version": 1, + "uid": "rule0000002", + "namespace_uid": "folder0000002", + "rule_group": "Example Paused Rule", + "no_data_state": "NoData", + "exec_err_state": "Error", + "is_paused": true, + "guid": "example-guid-0002" + } + } + ] + } + ] +} diff --git a/grafana-alertcheck/internal/gate/testdata/state_health_error.json b/grafana-alertcheck/internal/gate/testdata/state_health_error.json new file mode 100644 index 000000000..92a50fc83 --- /dev/null +++ b/grafana-alertcheck/internal/gate/testdata/state_health_error.json @@ -0,0 +1,68 @@ +{ + "status": "success", + "data": { + "groups": [ + { + "name": "ExampleService", + "file": "ExampleService", + "folderUid": "folder0000003", + "rules": [ + { + "state": "inactive", + "name": "Example No Data Source", + "query": "sum(increase(example_requests_total{env=\"staging\", method=\"Example\"}[30d]))", + "queriedDatasourceUIDs": [ + "datasource000003" + ], + "duration": 300, + "annotations": { + "description": "Just a test", + "summary": "Just a test" + }, + "alerts": [ + { + "labels": { + "alertname": "Example No Data Source", + "grafana_folder": "ExampleService", + "product": "example-service" + }, + "annotations": { + "Error": "failed to build query 'A': data source not found", + "description": "Just a test", + "summary": "Just a test" + }, + "state": "Error", + "activeAt": "2026-08-20T15:45:00Z", + "value": "" + } + ], + "totals": { + "error": 1 + }, + "totalsFiltered": { + "error": 1 + }, + "uid": "rule0000003", + "folderUid": "folder0000003", + "labels": { + "product": "example-service" + }, + "health": "error", + "lastError": "failed to build query 'A': data source not found", + "type": "alerting", + "lastEvaluation": "2026-08-31T09:15:00Z", + "evaluationTime": 0.00109723, + "isPaused": false + } + ], + "totals": { + "error": 1, + "inactive": 1 + }, + "interval": 300, + "lastEvaluation": "2026-08-31T09:15:00Z", + "evaluationTime": 0.00109723 + } + ] + } +} diff --git a/grafana-alertcheck/internal/gate/testdata/state_health_nodata.json b/grafana-alertcheck/internal/gate/testdata/state_health_nodata.json new file mode 100644 index 000000000..fc3b6949c --- /dev/null +++ b/grafana-alertcheck/internal/gate/testdata/state_health_nodata.json @@ -0,0 +1,62 @@ +{ + "status": "success", + "data": { + "groups": [ + { + "name": "example_group", + "file": "ExamplePlayground", + "folderUid": "folder0000004", + "rules": [ + { + "state": "inactive", + "name": "Example NoData Rule", + "query": "sum(example_satisfied_slo == 1)", + "queriedDatasourceUIDs": [ + "datasource000004" + ], + "duration": 300, + "annotations": { + "description": "{{ range $value := query \"sum by (id) (example_satisfied_slo == 1)\" }}\n - {{ $value.Labels.id }}\n{{ end }}" + }, + "alerts": [ + { + "labels": { + "alertname": "Example NoData Rule", + "datasource_uid": "datasource000004", + "grafana_folder": "ExamplePlayground", + "ref_id": "A" + }, + "annotations": { + "description": "" + }, + "state": "NoData", + "activeAt": "2026-08-24T09:06:30Z", + "value": "" + } + ], + "totals": { + "nodata": 1 + }, + "totalsFiltered": { + "nodata": 1 + }, + "uid": "rule0000004", + "folderUid": "folder0000004", + "health": "nodata", + "type": "alerting", + "lastEvaluation": "2026-08-31T09:16:30Z", + "evaluationTime": 0.34196819, + "isPaused": false + } + ], + "totals": { + "inactive": 1, + "nodata": 1 + }, + "interval": 300, + "lastEvaluation": "2026-08-31T09:16:30Z", + "evaluationTime": 0.34196819 + } + ] + } +} diff --git a/grafana-alertcheck/internal/gate/testdata/state_missing_file.json b/grafana-alertcheck/internal/gate/testdata/state_missing_file.json new file mode 100644 index 000000000..bc7beb4c7 --- /dev/null +++ b/grafana-alertcheck/internal/gate/testdata/state_missing_file.json @@ -0,0 +1,73 @@ +{ + "status": "success", + "data": { + "groups": [ + { + "name": "Example Service - Prod", + "folderUid": "folder0000001", + "rules": [ + { + "state": "inactive", + "name": "Example High CPU Usage", + "query": "max by (app_instance) (example_cpu_utilization_ratio{env=\"prod\", app=\"example-app\"})", + "queriedDatasourceUIDs": [ + "datasource000001" + ], + "duration": 300, + "annotations": { + "description": "{{ index $labels \"app_instance\" }} is using {{ index $values \"B\" }} of its CPU limit.", + "runbook_url": "https://example.com/runbook", + "summary": "{{ index $labels \"app_instance\" }} CPU utilization exceeded 90% of limit" + }, + "alerts": [ + { + "labels": { + "alertname": "Example High CPU Usage", + "env": "prod", + "grafana_folder": "ExampleTeam", + "app_instance": "example-app-instance", + "service": "example-svc", + "severity": "critical", + "team": "example-team" + }, + "annotations": { + "description": "example-app-instance is using 0.0075 of its CPU limit.", + "runbook_url": "https://example.com/runbook", + "summary": "example-app-instance CPU utilization exceeded 90% of limit" + }, + "state": "Normal", + "activeAt": "2026-08-31T08:02:50Z", + "value": "" + } + ], + "totals": { + "normal": 1 + }, + "totalsFiltered": { + "normal": 1 + }, + "uid": "rule0000001", + "folderUid": "folder0000001", + "labels": { + "env": "prod", + "service": "example-svc", + "severity": "critical", + "team": "example-team" + }, + "health": "ok", + "type": "alerting", + "lastEvaluation": "2026-08-31T09:16:50Z", + "evaluationTime": 0.003748832, + "isPaused": false + } + ], + "totals": { + "inactive": 14 + }, + "interval": 60, + "lastEvaluation": "2026-08-31T09:15:50Z", + "evaluationTime": 0.005293504 + } + ] + } +} diff --git a/grafana-alertcheck/internal/gate/testdata/state_missing_health.json b/grafana-alertcheck/internal/gate/testdata/state_missing_health.json new file mode 100644 index 000000000..d450a32f6 --- /dev/null +++ b/grafana-alertcheck/internal/gate/testdata/state_missing_health.json @@ -0,0 +1,73 @@ +{ + "status": "success", + "data": { + "groups": [ + { + "name": "Example Service - Prod", + "file": "ExampleTeam", + "folderUid": "folder0000001", + "rules": [ + { + "state": "inactive", + "name": "Example High CPU Usage", + "query": "max by (app_instance) (example_cpu_utilization_ratio{env=\"prod\", app=\"example-app\"})", + "queriedDatasourceUIDs": [ + "datasource000001" + ], + "duration": 300, + "annotations": { + "description": "{{ index $labels \"app_instance\" }} is using {{ index $values \"B\" }} of its CPU limit.", + "runbook_url": "https://example.com/runbook", + "summary": "{{ index $labels \"app_instance\" }} CPU utilization exceeded 90% of limit" + }, + "alerts": [ + { + "labels": { + "alertname": "Example High CPU Usage", + "env": "prod", + "grafana_folder": "ExampleTeam", + "app_instance": "example-app-instance", + "service": "example-svc", + "severity": "critical", + "team": "example-team" + }, + "annotations": { + "description": "example-app-instance is using 0.0075 of its CPU limit.", + "runbook_url": "https://example.com/runbook", + "summary": "example-app-instance CPU utilization exceeded 90% of limit" + }, + "state": "Normal", + "activeAt": "2026-08-31T08:02:50Z", + "value": "" + } + ], + "totals": { + "normal": 1 + }, + "totalsFiltered": { + "normal": 1 + }, + "uid": "rule0000001", + "folderUid": "folder0000001", + "labels": { + "env": "prod", + "service": "example-svc", + "severity": "critical", + "team": "example-team" + }, + "type": "alerting", + "lastEvaluation": "2026-08-31T09:16:50Z", + "evaluationTime": 0.003748832, + "isPaused": false + } + ], + "totals": { + "inactive": 14 + }, + "interval": 60, + "lastEvaluation": "2026-08-31T09:15:50Z", + "evaluationTime": 0.005293504 + } + ] + } +} diff --git a/grafana-alertcheck/internal/gate/testdata/state_missing_interval.json b/grafana-alertcheck/internal/gate/testdata/state_missing_interval.json new file mode 100644 index 000000000..4bc7dec18 --- /dev/null +++ b/grafana-alertcheck/internal/gate/testdata/state_missing_interval.json @@ -0,0 +1,73 @@ +{ + "status": "success", + "data": { + "groups": [ + { + "name": "Example Service - Prod", + "file": "ExampleTeam", + "folderUid": "folder0000001", + "rules": [ + { + "state": "inactive", + "name": "Example High CPU Usage", + "query": "max by (app_instance) (example_cpu_utilization_ratio{env=\"prod\", app=\"example-app\"})", + "queriedDatasourceUIDs": [ + "datasource000001" + ], + "duration": 300, + "annotations": { + "description": "{{ index $labels \"app_instance\" }} is using {{ index $values \"B\" }} of its CPU limit.", + "runbook_url": "https://example.com/runbook", + "summary": "{{ index $labels \"app_instance\" }} CPU utilization exceeded 90% of limit" + }, + "alerts": [ + { + "labels": { + "alertname": "Example High CPU Usage", + "env": "prod", + "grafana_folder": "ExampleTeam", + "app_instance": "example-app-instance", + "service": "example-svc", + "severity": "critical", + "team": "example-team" + }, + "annotations": { + "description": "example-app-instance is using 0.0075 of its CPU limit.", + "runbook_url": "https://example.com/runbook", + "summary": "example-app-instance CPU utilization exceeded 90% of limit" + }, + "state": "Normal", + "activeAt": "2026-08-31T08:02:50Z", + "value": "" + } + ], + "totals": { + "normal": 1 + }, + "totalsFiltered": { + "normal": 1 + }, + "uid": "rule0000001", + "folderUid": "folder0000001", + "labels": { + "env": "prod", + "service": "example-svc", + "severity": "critical", + "team": "example-team" + }, + "health": "ok", + "type": "alerting", + "lastEvaluation": "2026-08-31T09:16:50Z", + "evaluationTime": 0.003748832, + "isPaused": false + } + ], + "totals": { + "inactive": 14 + }, + "lastEvaluation": "2026-08-31T09:15:50Z", + "evaluationTime": 0.005293504 + } + ] + } +} diff --git a/grafana-alertcheck/internal/gate/testdata/state_missing_lasteval.json b/grafana-alertcheck/internal/gate/testdata/state_missing_lasteval.json new file mode 100644 index 000000000..4a2e8fcb9 --- /dev/null +++ b/grafana-alertcheck/internal/gate/testdata/state_missing_lasteval.json @@ -0,0 +1,73 @@ +{ + "status": "success", + "data": { + "groups": [ + { + "name": "Example Service - Prod", + "file": "ExampleTeam", + "folderUid": "folder0000001", + "rules": [ + { + "state": "inactive", + "name": "Example High CPU Usage", + "query": "max by (app_instance) (example_cpu_utilization_ratio{env=\"prod\", app=\"example-app\"})", + "queriedDatasourceUIDs": [ + "datasource000001" + ], + "duration": 300, + "annotations": { + "description": "{{ index $labels \"app_instance\" }} is using {{ index $values \"B\" }} of its CPU limit.", + "runbook_url": "https://example.com/runbook", + "summary": "{{ index $labels \"app_instance\" }} CPU utilization exceeded 90% of limit" + }, + "alerts": [ + { + "labels": { + "alertname": "Example High CPU Usage", + "env": "prod", + "grafana_folder": "ExampleTeam", + "app_instance": "example-app-instance", + "service": "example-svc", + "severity": "critical", + "team": "example-team" + }, + "annotations": { + "description": "example-app-instance is using 0.0075 of its CPU limit.", + "runbook_url": "https://example.com/runbook", + "summary": "example-app-instance CPU utilization exceeded 90% of limit" + }, + "state": "Normal", + "activeAt": "2026-08-31T08:02:50Z", + "value": "" + } + ], + "totals": { + "normal": 1 + }, + "totalsFiltered": { + "normal": 1 + }, + "uid": "rule0000001", + "folderUid": "folder0000001", + "labels": { + "env": "prod", + "service": "example-svc", + "severity": "critical", + "team": "example-team" + }, + "health": "ok", + "type": "alerting", + "evaluationTime": 0.003748832, + "isPaused": false + } + ], + "totals": { + "inactive": 14 + }, + "interval": 60, + "lastEvaluation": "2026-08-31T09:15:50Z", + "evaluationTime": 0.005293504 + } + ] + } +} diff --git a/grafana-alertcheck/internal/gate/testdata/state_missing_name.json b/grafana-alertcheck/internal/gate/testdata/state_missing_name.json new file mode 100644 index 000000000..eb5e8eb2a --- /dev/null +++ b/grafana-alertcheck/internal/gate/testdata/state_missing_name.json @@ -0,0 +1,73 @@ +{ + "status": "success", + "data": { + "groups": [ + { + "file": "ExampleTeam", + "folderUid": "folder0000001", + "rules": [ + { + "state": "inactive", + "name": "Example High CPU Usage", + "query": "max by (app_instance) (example_cpu_utilization_ratio{env=\"prod\", app=\"example-app\"})", + "queriedDatasourceUIDs": [ + "datasource000001" + ], + "duration": 300, + "annotations": { + "description": "{{ index $labels \"app_instance\" }} is using {{ index $values \"B\" }} of its CPU limit.", + "runbook_url": "https://example.com/runbook", + "summary": "{{ index $labels \"app_instance\" }} CPU utilization exceeded 90% of limit" + }, + "alerts": [ + { + "labels": { + "alertname": "Example High CPU Usage", + "env": "prod", + "grafana_folder": "ExampleTeam", + "app_instance": "example-app-instance", + "service": "example-svc", + "severity": "critical", + "team": "example-team" + }, + "annotations": { + "description": "example-app-instance is using 0.0075 of its CPU limit.", + "runbook_url": "https://example.com/runbook", + "summary": "example-app-instance CPU utilization exceeded 90% of limit" + }, + "state": "Normal", + "activeAt": "2026-08-31T08:02:50Z", + "value": "" + } + ], + "totals": { + "normal": 1 + }, + "totalsFiltered": { + "normal": 1 + }, + "uid": "rule0000001", + "folderUid": "folder0000001", + "labels": { + "env": "prod", + "service": "example-svc", + "severity": "critical", + "team": "example-team" + }, + "health": "ok", + "type": "alerting", + "lastEvaluation": "2026-08-31T09:16:50Z", + "evaluationTime": 0.003748832, + "isPaused": false + } + ], + "totals": { + "inactive": 14 + }, + "interval": 60, + "lastEvaluation": "2026-08-31T09:15:50Z", + "evaluationTime": 0.005293504 + } + ] + } +} diff --git a/grafana-alertcheck/internal/gate/testdata/state_missing_optional.json b/grafana-alertcheck/internal/gate/testdata/state_missing_optional.json new file mode 100644 index 000000000..9ad5c87f5 --- /dev/null +++ b/grafana-alertcheck/internal/gate/testdata/state_missing_optional.json @@ -0,0 +1,41 @@ +{ + "status": "success", + "data": { + "groups": [ + { + "name": "Example Service - Prod", + "file": "ExampleTeam", + "folderUid": "folder0000001", + "rules": [ + { + "state": "inactive", + "name": "Example High CPU Usage", + "query": "max by (app_instance) (example_cpu_utilization_ratio{env=\"prod\", app=\"example-app\"})", + "queriedDatasourceUIDs": [ + "datasource000001" + ], + "duration": 300, + "annotations": { + "description": "{{ index $labels \"app_instance\" }} is using {{ index $values \"B\" }} of its CPU limit.", + "runbook_url": "https://example.com/runbook", + "summary": "{{ index $labels \"app_instance\" }} CPU utilization exceeded 90% of limit" + }, + "uid": "rule0000001", + "folderUid": "folder0000001", + "health": "ok", + "type": "alerting", + "lastEvaluation": "2026-08-31T09:16:50Z", + "evaluationTime": 0.003748832, + "isPaused": false + } + ], + "totals": { + "inactive": 14 + }, + "interval": 60, + "lastEvaluation": "2026-08-31T09:15:50Z", + "evaluationTime": 0.005293504 + } + ] + } +} diff --git a/grafana-alertcheck/internal/gate/testdata/state_missing_state.json b/grafana-alertcheck/internal/gate/testdata/state_missing_state.json new file mode 100644 index 000000000..8db78b789 --- /dev/null +++ b/grafana-alertcheck/internal/gate/testdata/state_missing_state.json @@ -0,0 +1,73 @@ +{ + "status": "success", + "data": { + "groups": [ + { + "name": "Example Service - Prod", + "file": "ExampleTeam", + "folderUid": "folder0000001", + "rules": [ + { + "name": "Example High CPU Usage", + "query": "max by (app_instance) (example_cpu_utilization_ratio{env=\"prod\", app=\"example-app\"})", + "queriedDatasourceUIDs": [ + "datasource000001" + ], + "duration": 300, + "annotations": { + "description": "{{ index $labels \"app_instance\" }} is using {{ index $values \"B\" }} of its CPU limit.", + "runbook_url": "https://example.com/runbook", + "summary": "{{ index $labels \"app_instance\" }} CPU utilization exceeded 90% of limit" + }, + "alerts": [ + { + "labels": { + "alertname": "Example High CPU Usage", + "env": "prod", + "grafana_folder": "ExampleTeam", + "app_instance": "example-app-instance", + "service": "example-svc", + "severity": "critical", + "team": "example-team" + }, + "annotations": { + "description": "example-app-instance is using 0.0075 of its CPU limit.", + "runbook_url": "https://example.com/runbook", + "summary": "example-app-instance CPU utilization exceeded 90% of limit" + }, + "state": "Normal", + "activeAt": "2026-08-31T08:02:50Z", + "value": "" + } + ], + "totals": { + "normal": 1 + }, + "totalsFiltered": { + "normal": 1 + }, + "uid": "rule0000001", + "folderUid": "folder0000001", + "labels": { + "env": "prod", + "service": "example-svc", + "severity": "critical", + "team": "example-team" + }, + "health": "ok", + "type": "alerting", + "lastEvaluation": "2026-08-31T09:16:50Z", + "evaluationTime": 0.003748832, + "isPaused": false + } + ], + "totals": { + "inactive": 14 + }, + "interval": 60, + "lastEvaluation": "2026-08-31T09:15:50Z", + "evaluationTime": 0.005293504 + } + ] + } +} diff --git a/grafana-alertcheck/internal/gate/testdata/state_one_instance.json b/grafana-alertcheck/internal/gate/testdata/state_one_instance.json new file mode 100644 index 000000000..09afe0fbc --- /dev/null +++ b/grafana-alertcheck/internal/gate/testdata/state_one_instance.json @@ -0,0 +1,74 @@ +{ + "status": "success", + "data": { + "groups": [ + { + "name": "Example Service - Prod", + "file": "ExampleTeam", + "folderUid": "folder0000001", + "rules": [ + { + "state": "inactive", + "name": "Example High CPU Usage", + "query": "max by (app_instance) (example_cpu_utilization_ratio{env=\"prod\", app=\"example-app\"})", + "queriedDatasourceUIDs": [ + "datasource000001" + ], + "duration": 300, + "annotations": { + "description": "{{ index $labels \"app_instance\" }} is using {{ index $values \"B\" }} of its CPU limit.", + "runbook_url": "https://example.com/runbook", + "summary": "{{ index $labels \"app_instance\" }} CPU utilization exceeded 90% of limit" + }, + "alerts": [ + { + "labels": { + "alertname": "Example High CPU Usage", + "env": "prod", + "grafana_folder": "ExampleTeam", + "app_instance": "example-app-instance", + "service": "example-svc", + "severity": "critical", + "team": "example-team" + }, + "annotations": { + "description": "example-app-instance is using 0.0075 of its CPU limit.", + "runbook_url": "https://example.com/runbook", + "summary": "example-app-instance CPU utilization exceeded 90% of limit" + }, + "state": "Normal", + "activeAt": "2026-08-31T08:02:50Z", + "value": "" + } + ], + "totals": { + "normal": 1 + }, + "totalsFiltered": { + "normal": 1 + }, + "uid": "rule0000001", + "folderUid": "folder0000001", + "labels": { + "env": "prod", + "service": "example-svc", + "severity": "critical", + "team": "example-team" + }, + "health": "ok", + "type": "alerting", + "lastEvaluation": "2026-08-31T09:16:50Z", + "evaluationTime": 0.003748832, + "isPaused": false + } + ], + "totals": { + "inactive": 14 + }, + "interval": 60, + "lastEvaluation": "2026-08-31T09:15:50Z", + "evaluationTime": 0.005293504 + } + ] + } +} diff --git a/grafana-alertcheck/internal/gate/testdata/state_only_active_instances.json b/grafana-alertcheck/internal/gate/testdata/state_only_active_instances.json new file mode 100644 index 000000000..ac39c69b0 --- /dev/null +++ b/grafana-alertcheck/internal/gate/testdata/state_only_active_instances.json @@ -0,0 +1,79 @@ +{ + "status": "success", + "data": { + "groups": [ + { + "name": "example-reporting-stage", + "file": "ExampleTeam", + "folderUid": "folder0000001", + "rules": [ + { + "state": "firing", + "name": "Example Missing Pipeline Events", + "query": "max by (pipeline, event_name) (example_oldest_missing_event_age_ms{cluster=\"example-cluster\"})", + "queriedDatasourceUIDs": [ + "datasource000001" + ], + "duration": 300, + "annotations": { + "description": "The oldest un-materialized event has been missing for more than 30 minutes.", + "runbook_url": "https://example.com/runbook", + "summary": "Pipeline event stuck" + }, + "activeAt": "2026-08-31T04:38:10Z", + "alerts": [ + { + "labels": { + "alertname": "Example Missing Pipeline Events", + "env": "stage", + "event_name": "PolicyConfigured", + "grafana_folder": "ExampleTeam", + "pipeline": "policy_configured", + "service": "example-svc", + "severity": "critical", + "team": "example-team" + }, + "annotations": { + "description": "The oldest un-materialized event for pipeline policy_configured has been missing for more than 30 minutes.", + "runbook_url": "https://example.com/runbook", + "summary": "Pipeline policy_configured has a PolicyConfigured event stuck for 18747998ms" + }, + "state": "Alerting", + "activeAt": "2026-08-31T04:38:10Z", + "value": "1e+00" + } + ], + "totals": { + "alerting": 1, + "normal": 22 + }, + "totalsFiltered": { + "alerting": 1, + "normal": 22 + }, + "uid": "rule0000006", + "folderUid": "folder0000001", + "labels": { + "env": "stage", + "service": "example-svc", + "severity": "critical", + "team": "example-team" + }, + "health": "ok", + "type": "alerting", + "lastEvaluation": "2026-08-31T09:16:10Z", + "evaluationTime": 0.016299343, + "isPaused": false + } + ], + "totals": { + "firing": 1, + "inactive": 9 + }, + "interval": 60, + "lastEvaluation": "2026-08-31T09:16:10Z", + "evaluationTime": 0.003291541 + } + ] + } +} diff --git a/grafana-alertcheck/internal/gate/testdata/state_paused.json b/grafana-alertcheck/internal/gate/testdata/state_paused.json new file mode 100644 index 000000000..e44d367f5 --- /dev/null +++ b/grafana-alertcheck/internal/gate/testdata/state_paused.json @@ -0,0 +1,40 @@ +{ + "status": "success", + "data": { + "groups": [ + { + "name": "Example Paused Rule", + "file": "ExampleMetrics", + "folderUid": "folder0000002", + "rules": [ + { + "state": "inactive", + "name": "Example Paused Rule", + "query": "example_metric{env=\"production\"}", + "queriedDatasourceUIDs": [ + "datasource000002" + ], + "duration": 300, + "annotations": { + "__dashboardUid__": "example-dashboard-uid", + "__panelId__": "1" + }, + "uid": "rule0000002", + "folderUid": "folder0000002", + "health": "ok", + "type": "alerting", + "lastEvaluation": "0001-01-01T00:00:00Z", + "evaluationTime": 0, + "isPaused": true + } + ], + "totals": { + "inactive": 1 + }, + "interval": 300, + "lastEvaluation": "0001-01-01T00:00:00Z", + "evaluationTime": 0 + } + ] + } +} diff --git a/grafana-alertcheck/internal/gate/testdata/state_reason_composite.json b/grafana-alertcheck/internal/gate/testdata/state_reason_composite.json new file mode 100644 index 000000000..b28a6d50d --- /dev/null +++ b/grafana-alertcheck/internal/gate/testdata/state_reason_composite.json @@ -0,0 +1,99 @@ +{ + "status": "success", + "data": { + "groups": [ + { + "name": "example-infra", + "file": "ExampleInfra", + "folderUid": "folder0000005", + "rules": [ + { + "state": "inactive", + "name": "Example Composite Reasons", + "query": "sum(increase(example_reconcile_total{result=\"error\"}[1m])) by (cluster)", + "queriedDatasourceUIDs": [ + "datasource000005" + ], + "duration": 300, + "annotations": { + "summary": "{{ cluster }} is having reconciliation issues." + }, + "alerts": [ + { + "labels": { + "alertname": "Example Composite Reasons", + "grafana_folder": "ExampleInfra", + "cluster": "example-cluster-a", + "severity": "error", + "team": "example-infra" + }, + "annotations": { + "Error": "failed to build query 'B': data source not found", + "summary": "example-cluster-a is having reconciliation issues." + }, + "state": "Normal (Error)", + "activeAt": "2025-08-07T11:48:10Z", + "value": "" + }, + { + "labels": { + "alertname": "Example Composite Reasons", + "grafana_folder": "ExampleInfra", + "cluster": "example-cluster-b", + "severity": "warning", + "team": "example-infra" + }, + "annotations": { + "Error": "unexpected response with status code 429", + "description": "example-cluster-b query returned no data recently.", + "summary": "example-cluster-b is not reporting" + }, + "state": "Normal (NoData)", + "activeAt": "2026-08-31T08:02:50Z", + "value": "" + }, + { + "labels": { + "alertname": "Example Composite Reasons", + "grafana_folder": "ExampleInfra", + "cluster": "example-cluster-c", + "severity": "warning", + "team": "example-infra" + }, + "annotations": { + "summary": "example-cluster-c is healthy" + }, + "state": "Normal", + "activeAt": "2026-08-31T08:02:50Z", + "value": "" + } + ], + "totals": { + "normal": 3 + }, + "totalsFiltered": { + "normal": 3 + }, + "uid": "rule0000005", + "folderUid": "folder0000005", + "labels": { + "severity": "error", + "team": "example-infra" + }, + "health": "ok", + "type": "alerting", + "lastEvaluation": "2026-08-31T09:16:10Z", + "evaluationTime": 0.001283254, + "isPaused": false + } + ], + "totals": { + "inactive": 1 + }, + "interval": 60, + "lastEvaluation": "2026-08-31T09:16:10Z", + "evaluationTime": 0.001283254 + } + ] + } +} diff --git a/grafana-alertcheck/internal/gate/testdata/state_unknown_state.json b/grafana-alertcheck/internal/gate/testdata/state_unknown_state.json new file mode 100644 index 000000000..4cda28ca3 --- /dev/null +++ b/grafana-alertcheck/internal/gate/testdata/state_unknown_state.json @@ -0,0 +1,74 @@ +{ + "status": "success", + "data": { + "groups": [ + { + "name": "Example Service - Prod", + "file": "ExampleTeam", + "folderUid": "folder0000001", + "rules": [ + { + "state": "inactive", + "name": "Example High CPU Usage", + "query": "max by (app_instance) (example_cpu_utilization_ratio{env=\"prod\", app=\"example-app\"})", + "queriedDatasourceUIDs": [ + "datasource000001" + ], + "duration": 300, + "annotations": { + "description": "{{ index $labels \"app_instance\" }} is using {{ index $values \"B\" }} of its CPU limit.", + "runbook_url": "https://example.com/runbook", + "summary": "{{ index $labels \"app_instance\" }} CPU utilization exceeded 90% of limit" + }, + "alerts": [ + { + "labels": { + "alertname": "Example High CPU Usage", + "env": "prod", + "grafana_folder": "ExampleTeam", + "app_instance": "example-app-instance", + "service": "example-svc", + "severity": "critical", + "team": "example-team" + }, + "annotations": { + "description": "example-app-instance is using 0.0075 of its CPU limit.", + "runbook_url": "https://example.com/runbook", + "summary": "example-app-instance CPU utilization exceeded 90% of limit" + }, + "state": "Weird (NoData)", + "activeAt": "2026-08-31T08:02:50Z", + "value": "" + } + ], + "totals": { + "normal": 1 + }, + "totalsFiltered": { + "normal": 1 + }, + "uid": "rule0000001", + "folderUid": "folder0000001", + "labels": { + "env": "prod", + "service": "example-svc", + "severity": "critical", + "team": "example-team" + }, + "health": "ok", + "type": "alerting", + "lastEvaluation": "2026-08-31T09:16:50Z", + "evaluationTime": 0.003748832, + "isPaused": false + } + ], + "totals": { + "inactive": 14 + }, + "interval": 60, + "lastEvaluation": "2026-08-31T09:15:50Z", + "evaluationTime": 0.005293504 + } + ] + } +} diff --git a/grafana-alertcheck/internal/gate/testdata/state_zerotime_unpaused.json b/grafana-alertcheck/internal/gate/testdata/state_zerotime_unpaused.json new file mode 100644 index 000000000..b9a0fdd1a --- /dev/null +++ b/grafana-alertcheck/internal/gate/testdata/state_zerotime_unpaused.json @@ -0,0 +1,74 @@ +{ + "status": "success", + "data": { + "groups": [ + { + "name": "Example Service - Prod", + "file": "ExampleTeam", + "folderUid": "folder0000001", + "rules": [ + { + "state": "inactive", + "name": "Example High CPU Usage", + "query": "max by (app_instance) (example_cpu_utilization_ratio{env=\"prod\", app=\"example-app\"})", + "queriedDatasourceUIDs": [ + "datasource000001" + ], + "duration": 300, + "annotations": { + "description": "{{ index $labels \"app_instance\" }} is using {{ index $values \"B\" }} of its CPU limit.", + "runbook_url": "https://example.com/runbook", + "summary": "{{ index $labels \"app_instance\" }} CPU utilization exceeded 90% of limit" + }, + "alerts": [ + { + "labels": { + "alertname": "Example High CPU Usage", + "env": "prod", + "grafana_folder": "ExampleTeam", + "app_instance": "example-app-instance", + "service": "example-svc", + "severity": "critical", + "team": "example-team" + }, + "annotations": { + "description": "example-app-instance is using 0.0075 of its CPU limit.", + "runbook_url": "https://example.com/runbook", + "summary": "example-app-instance CPU utilization exceeded 90% of limit" + }, + "state": "Normal", + "activeAt": "2026-08-31T08:02:50Z", + "value": "" + } + ], + "totals": { + "normal": 1 + }, + "totalsFiltered": { + "normal": 1 + }, + "uid": "rule0000001", + "folderUid": "folder0000001", + "labels": { + "env": "prod", + "service": "example-svc", + "severity": "critical", + "team": "example-team" + }, + "health": "ok", + "type": "alerting", + "lastEvaluation": "0001-01-01T00:00:00Z", + "evaluationTime": 0.003748832, + "isPaused": false + } + ], + "totals": { + "inactive": 14 + }, + "interval": 60, + "lastEvaluation": "2026-08-31T09:15:50Z", + "evaluationTime": 0.005293504 + } + ] + } +} From 11d16870173d244712d404c66ed6985aaa8ad281 Mon Sep 17 00:00:00 2001 From: Bartek Tofel Date: Mon, 7 Sep 2026 11:54:26 +0200 Subject: [PATCH 04/43] chore: apply code review comments --- grafana-alertcheck/internal/gate/jsonreq.go | 5 +++- .../internal/gate/parse_ruler.go | 3 +++ .../internal/gate/parse_ruler_test.go | 11 ++++++++ .../internal/gate/parse_state.go | 25 ++++++------------- .../internal/gate/parse_state_test.go | 21 +++++++++++++--- 5 files changed, 43 insertions(+), 22 deletions(-) diff --git a/grafana-alertcheck/internal/gate/jsonreq.go b/grafana-alertcheck/internal/gate/jsonreq.go index cea19b6fe..bfddca382 100644 --- a/grafana-alertcheck/internal/gate/jsonreq.go +++ b/grafana-alertcheck/internal/gate/jsonreq.go @@ -13,9 +13,12 @@ import ( // required field sent as null would silently pass through as its zero value. func req[T any](m map[string]json.RawMessage, key string, dst *T) error { raw, ok := m[key] - if !ok || isJSONNull(raw) { + if !ok { return fmt.Errorf("required field %q is absent", key) } + if isJSONNull(raw) { + return fmt.Errorf("required field %q is null", key) + } if err := json.Unmarshal(raw, dst); err != nil { return fmt.Errorf("field %q: %w", key, err) } diff --git a/grafana-alertcheck/internal/gate/parse_ruler.go b/grafana-alertcheck/internal/gate/parse_ruler.go index 06f51407e..c27e5b9c4 100644 --- a/grafana-alertcheck/internal/gate/parse_ruler.go +++ b/grafana-alertcheck/internal/gate/parse_ruler.go @@ -115,6 +115,9 @@ func parseDefinition(raw json.RawMessage, folder, group string) (Definition, err return Definition{}, err } } + if def.Title == "" { + return Definition{}, fmt.Errorf("datasource-managed rule has no alert or record name") + } return def, nil } diff --git a/grafana-alertcheck/internal/gate/parse_ruler_test.go b/grafana-alertcheck/internal/gate/parse_ruler_test.go index 56d35b663..dda66b77e 100644 --- a/grafana-alertcheck/internal/gate/parse_ruler_test.go +++ b/grafana-alertcheck/internal/gate/parse_ruler_test.go @@ -114,3 +114,14 @@ func TestParseDefinitions_Recording(t *testing.T) { t.Errorf("expected zero-valued alert-only fields for a recording rule, got %+v", d) } } + +// A datasource-managed rule with neither an "alert" nor a "record" name has no +// identity (its only name is the Prometheus rule name), and an empty Title +// would make P3's refusal-by-name unreachable. It must fail parsing, not hand +// back a silently unusable Definition. +func TestParseDefinitions_DatasourceManagedNoName(t *testing.T) { + body := []byte(`{"ExampleMetrics":[{"name":"g","rules":[{"expr":"up == 0","for":"5m"}]}]}`) + if _, err := ParseDefinitions(body); err == nil { + t.Fatalf("ParseDefinitions: expected error for datasource-managed rule with no alert/record, got nil") + } +} diff --git a/grafana-alertcheck/internal/gate/parse_state.go b/grafana-alertcheck/internal/gate/parse_state.go index ebd30d7a5..d49dca0fa 100644 --- a/grafana-alertcheck/internal/gate/parse_state.go +++ b/grafana-alertcheck/internal/gate/parse_state.go @@ -3,7 +3,6 @@ package gate import ( "encoding/json" "fmt" - "sort" "strings" "time" ) @@ -249,21 +248,13 @@ func normalizeInstanceState(s string) (State, string, error) { return state, reason, nil } -// instanceKey is a stable identity for an instance's label set: a sorted -// "k=v\n" join. Used to correlate an instance across polls without hashing. +// instanceKey is a stable identity for an instance's label set. It is the +// JSON encoding of the label map, which encoding/json deterministically emits +// with sorted keys, so the string is order-independent and collision-free (a +// value containing "=" or "\n" can't be mistaken for a key/value boundary). +// Used to correlate an instance across polls without hashing. Marshal cannot +// fail for map[string]string, so the error is discarded. func instanceKey(labels map[string]string) string { - keys := make([]string, 0, len(labels)) - for k := range labels { - keys = append(keys, k) - } - sort.Strings(keys) - - var b strings.Builder - for _, k := range keys { - b.WriteString(k) - b.WriteByte('=') - b.WriteString(labels[k]) - b.WriteByte('\n') - } - return b.String() + b, _ := json.Marshal(labels) + return string(b) } diff --git a/grafana-alertcheck/internal/gate/parse_state_test.go b/grafana-alertcheck/internal/gate/parse_state_test.go index 7cf336526..e370c0f5e 100644 --- a/grafana-alertcheck/internal/gate/parse_state_test.go +++ b/grafana-alertcheck/internal/gate/parse_state_test.go @@ -269,8 +269,8 @@ func TestInstanceKey(t *testing.T) { if a != b { t.Errorf("instanceKey order-independence: %q != %q", a, b) } - if a != "a=1\nb=2\n" { - t.Errorf("instanceKey = %q, want a=1\\nb=2\\n", a) + if a != `{"a":"1","b":"2"}` { + t.Errorf("instanceKey = %q, want %q", a, `{"a":"1","b":"2"}`) } diff := instanceKey(map[string]string{"a": "1", "b": "3"}) @@ -278,8 +278,21 @@ func TestInstanceKey(t *testing.T) { t.Errorf("instanceKey should differ when a label value differs") } - if instanceKey(nil) != "" { - t.Errorf("instanceKey(nil) = %q, want empty string", instanceKey(nil)) + if instanceKey(nil) != "null" { + t.Errorf("instanceKey(nil) = %q, want \"null\"", instanceKey(nil)) + } +} + +// TestInstanceKey_NoCollision guards against ambiguous identities: label +// values may legally contain "\n" or "=", and a naive "k=v\n" join would +// collide e.g. {a:"1\nb=2"} with {a:"1",b:"2"}. The JSON encoding must keep +// such sets distinct. +func TestInstanceKey_NoCollision(t *testing.T) { + if instanceKey(map[string]string{"a": "1\nb=2"}) == instanceKey(map[string]string{"a": "1", "b": "2"}) { + t.Errorf("instanceKey collided for sets {a:1\\nb=2} and {a:1,b:2}") + } + if instanceKey(map[string]string{"a": "1=b"}) == instanceKey(map[string]string{"a": "1", "b": ""}) { + t.Errorf("instanceKey collided for a value containing '='") } } From 8385e14512ffdd93bb5371c42e058c229f5cd0cc Mon Sep 17 00:00:00 2001 From: Bartek Tofel Date: Mon, 31 Aug 2026 13:43:36 +0200 Subject: [PATCH 05/43] chore: implement phase 2 Fix retry-error conflation, measure full poll latency, and harden Source test doubles for concurrency. --- grafana-alertcheck/internal/gate/source.go | 373 ++++++++++++ .../internal/gate/source_fake_test.go | 142 +++++ .../internal/gate/source_test.go | 559 ++++++++++++++++++ 3 files changed, 1074 insertions(+) create mode 100644 grafana-alertcheck/internal/gate/source.go create mode 100644 grafana-alertcheck/internal/gate/source_fake_test.go create mode 100644 grafana-alertcheck/internal/gate/source_test.go diff --git a/grafana-alertcheck/internal/gate/source.go b/grafana-alertcheck/internal/gate/source.go new file mode 100644 index 000000000..46de818bb --- /dev/null +++ b/grafana-alertcheck/internal/gate/source.go @@ -0,0 +1,373 @@ +package gate + +import ( + "context" + "encoding/json" + "errors" + "fmt" + "io" + "math/rand/v2" + "net/http" + "net/url" + "strconv" + "strings" + "time" +) + +// skewHardLimit is one of §5's filled-in values (basis: §16; §22.11 asserts +// 120s errors, 30s does not). It belongs in schedule.go's named-constants +// block once P4 exists; defined here because P2 needs it first. +const skewHardLimit = 60 * time.Second + +// Clock is the seam that lets tests advance time without sleeping (§22) — the +// only two operations the gate ever needs from a clock. +type Clock interface { + Now() time.Time + After(d time.Duration) <-chan time.Time +} + +// SystemClock is the production Clock: the real wall clock. +type SystemClock struct{} + +func (SystemClock) Now() time.Time { return time.Now() } +func (SystemClock) After(d time.Duration) <-chan time.Time { return time.After(d) } + +// Observation is one successful poll of the state endpoint for a single rule. +type Observation struct { + Rules []StateRule // may be empty — an authoritative 2xx saying the rule is absent (§14.5) + GrafanaNow time.Time // the Date header — H4 + Skew time.Duration // serverDate - (t_send+t_headers)/2, signed (§16) + SkewBound time.Duration // (t_headers-t_send)/2 — RTT/2 to the response headers + Latency time.Duration // t_send through the full body read — see requestResult.Latency +} + +// TransportError marks a failure worth retrying: a non-2xx response, a +// network failure, or a body that failed to parse. It is never a deleted rule +// (an authoritative 2xx with no matching rule is not this) and never a clock +// problem (a missing/unparseable Date header or an out-of-bounds skew is a +// hard error instead — see doRequest). Never conflate them (§14.5). +type TransportError struct { + Err error + Status int // 0 when the failure never got a status (network/transport failure) +} + +func (e *TransportError) Error() string { + if e.Status != 0 { + return fmt.Sprintf("transport error: status %d: %v", e.Status, e.Err) + } + return fmt.Sprintf("transport error: %v", e.Err) +} + +func (e *TransportError) Unwrap() error { return e.Err } + +// RetryExhaustedError is what retryTransport returns once it gives up after +// too many sequential *TransportError failures. It deliberately does not +// implement Unwrap into the underlying *TransportError: once retries are +// exhausted the result is a hard, terminal failure, and +// errors.AsType[*TransportError] must never re-classify it as retryable — +// that is the exact conflation §19.3 case 1 forbids. Cause is still exposed +// as a plain field (and folded into Error()'s text) so a caller can log or +// inspect it; it just cannot flow back into the retry classification. +type RetryExhaustedError struct { + Failures int + Cause error +} + +func (e *RetryExhaustedError) Error() string { + return fmt.Sprintf("gave up after %d sequential failures: %v", e.Failures, e.Cause) +} + +// Source is everything the gate reads from Grafana. httpSource is the one +// production implementation; every later phase's tests use a scripted fake +// (source_fake_test.go) instead of real HTTP. +type Source interface { + Version(ctx context.Context) (string, error) + Definitions(ctx context.Context) ([]Definition, error) + RuleState(ctx context.Context, title string) (Observation, error) +} + +// grafanaVersion is a parsed major.minor.patch triple. +type grafanaVersion struct{ major, minor, patch int } + +func (v grafanaVersion) String() string { return fmt.Sprintf("%d.%d.%d", v.major, v.minor, v.patch) } + +// ord encodes the triple as a single comparable integer. Safe as long as +// minor and patch stay under 1000, true of every real Grafana version. +func (v grafanaVersion) ord() int64 { + return int64(v.major)*1_000_000 + int64(v.minor)*1_000 + int64(v.patch) +} + +func parseGrafanaVersion(s string) (grafanaVersion, error) { + s = strings.TrimSpace(s) + if s == "" { + return grafanaVersion{}, errors.New("empty version string") + } + // Grafana's /api/health always reports exactly three components + // (health.json: "13.1.0"). Require all three explicitly rather than + // defaulting missing ones to zero or silently dropping extras — either + // would accept a value ("13", "13.1.0.5") that was never actually seen + // and never verified against. + parts := strings.Split(s, ".") + if len(parts) != 3 { + return grafanaVersion{}, fmt.Errorf("unparseable version %q: want exactly 3 dot-separated components, got %d", s, len(parts)) + } + var v grafanaVersion + fields := [3]*int{&v.major, &v.minor, &v.patch} + for i, field := range fields { + // Trim any trailing non-digit suffix (prerelease/build metadata, e.g. + // "0+security") rather than requiring an exact numeric match. + digits := parts[i] + j := 0 + for j < len(digits) && digits[j] >= '0' && digits[j] <= '9' { + j++ + } + if j == 0 { + return grafanaVersion{}, fmt.Errorf("unparseable version %q", s) + } + n, err := strconv.Atoi(digits[:j]) + if err != nil { + return grafanaVersion{}, fmt.Errorf("unparseable version %q: %w", s, err) + } + *field = n + } + return v, nil +} + +// supportedGrafanaMin and supportedGrafanaMax bound the platform this gate is +// verified against (§2.7 control 2, §21.5): >= 13.0.0, < 14.0.0. +var ( + supportedGrafanaMin = grafanaVersion{13, 0, 0} + supportedGrafanaMax = grafanaVersion{14, 0, 0} // exclusive +) + +// CheckGrafanaVersion enforces the supported range. An unparseable or +// out-of-range version is a hard error naming both what was found and what is +// supported — trusting an unverified schema is exactly the deprecation risk +// §2.7 control 2 exists to catch. +func CheckGrafanaVersion(version string) error { + v, err := parseGrafanaVersion(version) + if err != nil { + return fmt.Errorf("grafana version %q: %w (supported: >=%s, <%s)", + version, err, supportedGrafanaMin, supportedGrafanaMax) + } + if v.ord() < supportedGrafanaMin.ord() || v.ord() >= supportedGrafanaMax.ord() { + return fmt.Errorf("unsupported grafana version %q (supported: >=%s, <%s)", + version, supportedGrafanaMin, supportedGrafanaMax) + } + return nil +} + +// httpSource is the production Source: stdlib net/http only, bearer auth +// from a token supplied at construction (the caller reads it from the +// environment — §20.2 — this type never touches env itself), and manual +// strict decoding via ParseState/ParseDefinitions (H1). The retry limit and +// backoff parameters are struct fields with production defaults set here, +// not package constants, so a test can shrink them without a hook. +type httpSource struct { + baseURL string + token string + client *http.Client + clock Clock + + maxSequentialFailures int + backoffBase time.Duration + backoffCap time.Duration +} + +// NewHTTPSource builds the production Source. token is never logged and +// never enters an error string (§20.2) — it is used only to set the +// Authorization header. +func NewHTTPSource(baseURL, token string, clock Clock) Source { + return &httpSource{ + baseURL: strings.TrimSuffix(baseURL, "/"), + token: token, + client: &http.Client{Timeout: 30 * time.Second}, + clock: clock, + maxSequentialFailures: 5, + backoffBase: time.Second, + backoffCap: 30 * time.Second, + } +} + +func (s *httpSource) Version(ctx context.Context) (string, error) { + return retryTransport(ctx, s.clock, s.maxSequentialFailures, s.backoffBase, s.backoffCap, func() (string, error) { + r, err := s.doRequest(ctx, "/api/health") + if err != nil { + return "", err + } + var health struct { + Version string `json:"version"` + } + if err := json.Unmarshal(r.Body, &health); err != nil { + return "", &TransportError{Err: fmt.Errorf("parse /api/health: %w", err)} + } + if health.Version == "" { + return "", &TransportError{Err: errors.New("/api/health: empty version")} + } + return health.Version, nil + }) +} + +func (s *httpSource) Definitions(ctx context.Context) ([]Definition, error) { + return retryTransport(ctx, s.clock, s.maxSequentialFailures, s.backoffBase, s.backoffCap, func() ([]Definition, error) { + r, err := s.doRequest(ctx, "/api/ruler/grafana/api/v1/rules") + if err != nil { + return nil, err + } + defs, parseErr := ParseDefinitions(r.Body) + if parseErr != nil { + return nil, &TransportError{Err: fmt.Errorf("parse ruler definitions: %w", parseErr)} + } + return defs, nil + }) +} + +func (s *httpSource) RuleState(ctx context.Context, title string) (Observation, error) { + path := "/api/prometheus/grafana/api/v1/rules?rule_name=" + url.QueryEscape(title) + return retryTransport(ctx, s.clock, s.maxSequentialFailures, s.backoffBase, s.backoffCap, func() (Observation, error) { + r, err := s.doRequest(ctx, path) + if err != nil { + return Observation{}, err + } + rules, parseErr := ParseState(r.Body) + if parseErr != nil { + // Treated as transient, not a schema break: an unparseable 2xx + // is far more likely a mid-stream hiccup than a permanent shape + // change, and H1's strict parser already turns a real shape + // change into a loud per-field error the moment it's visible. + return Observation{}, &TransportError{Err: fmt.Errorf("parse rule state: %w", parseErr)} + } + return Observation{ + Rules: rules, + GrafanaNow: r.ServerDate, + Skew: r.Skew, + SkewBound: r.SkewBound, + Latency: r.Latency, + }, nil + }) +} + +// requestResult is the outcome of one successful HTTP attempt in doRequest: +// the raw body plus everything derived from timing the round trip against +// the response's own clock (§16). +type requestResult struct { + Body []byte + ServerDate time.Time // the Date header — H4 + Skew time.Duration // serverDate - (t_send+t_headers)/2, signed + SkewBound time.Duration // (t_headers-t_send)/2 — RTT/2 to the response headers + // Latency spans t_send through the full body read (§5.2's budget check + // needs the whole poll's wall time, or a schedule feasibility check that + // only sees header latency goes optimistic — fail-open). It does not + // include the caller's subsequent JSON parse (ParseState/ParseDefinitions + // run outside doRequest); if P4's budget accounting needs parse time + // folded in too, extend here rather than approximating it at the call + // site. + Latency time.Duration +} + +// doRequest performs one HTTP GET and classifies the outcome (§14.5, §16): +// a network failure, a non-2xx status, or a body-read failure is retryable +// (*TransportError); a missing or unparseable Date header, or a skew beyond +// skewHardLimit, is a hard error — retrying can never fix either, so neither +// may enter the backoff loop (H4). +// +// The Date-header/skew check runs for every endpoint this hits, including +// /api/health — broader than §16's own scope, which only discusses the state +// endpoint. Deliberate: a skewed clock discovered only once RuleState starts +// polling is a skew that has already masked whatever /api/health and the +// ruler read reported; failing closed at the first response catches it +// before any of that is trusted, and every response comes with a Date header +// for free. +func (s *httpSource) doRequest(ctx context.Context, path string) (requestResult, error) { + req, buildErr := http.NewRequestWithContext(ctx, http.MethodGet, s.baseURL+path, nil) + if buildErr != nil { + return requestResult{}, fmt.Errorf("build request for %s: %w", path, buildErr) + } + if s.token != "" { + req.Header.Set("Authorization", "Bearer "+s.token) + } + + tSend := s.clock.Now() + resp, doErr := s.client.Do(req) + tHeaders := s.clock.Now() + if doErr != nil { + return requestResult{}, &TransportError{Err: doErr} + } + defer resp.Body.Close() + + b, readErr := io.ReadAll(resp.Body) + tBodyRead := s.clock.Now() + if readErr != nil { + return requestResult{}, &TransportError{Err: fmt.Errorf("read response body (status %d): %w", resp.StatusCode, readErr)} + } + latency := tBodyRead.Sub(tSend) + + if resp.StatusCode < 200 || resp.StatusCode >= 300 { + return requestResult{}, &TransportError{Err: fmt.Errorf("unexpected status %d", resp.StatusCode), Status: resp.StatusCode} + } + + dateHeader := resp.Header.Get("Date") + if dateHeader == "" { + return requestResult{}, fmt.Errorf("%s: response has no Date header (H4)", path) + } + serverDate, parseErr := http.ParseTime(dateHeader) + if parseErr != nil { + return requestResult{}, fmt.Errorf("%s: unparseable Date header %q: %w", path, dateHeader, parseErr) + } + + bound := tHeaders.Sub(tSend) / 2 + mid := tSend.Add(bound) + signedSkew := serverDate.Sub(mid) + absSkew := signedSkew + if absSkew < 0 { + absSkew = -absSkew + } + if absSkew > skewHardLimit { + return requestResult{}, fmt.Errorf("%s: clock skew %s exceeds hard limit %s (§16)", path, absSkew, skewHardLimit) + } + + return requestResult{Body: b, ServerDate: serverDate, Skew: signedSkew, SkewBound: bound, Latency: latency}, nil +} + +// retryTransport runs fn, retrying with backoff only while it fails with a +// *TransportError — any other error is a hard error and returns immediately, +// never retried. failures counts consecutive *TransportError results; +// exceeding maxFailures gives up with a wrapped hard error (§19.3 case 1). +// The wait between attempts goes through clock.After so a test with a fake +// Clock never sleeps on real time (§22). +func retryTransport[T any](ctx context.Context, clock Clock, maxFailures int, backoffBase, backoffCap time.Duration, fn func() (T, error)) (T, error) { + var zero T + failures := 0 + for { + v, err := fn() + if err == nil { + return v, nil + } + if _, ok := errors.AsType[*TransportError](err); !ok { + return zero, err + } + failures++ + if failures > maxFailures { + return zero, &RetryExhaustedError{Failures: failures, Cause: err} + } + select { + case <-ctx.Done(): + return zero, ctx.Err() + case <-clock.After(backoffDelay(backoffBase, backoffCap, failures)): + } + } +} + +// backoffDelay is 1s base, doubling per failure, capped at maxDelay, with +// ±20% jitter (§5's filled-in value for maxSequentialFailures). +func backoffDelay(base, maxDelay time.Duration, failureCount int) time.Duration { + d := base + for i := 1; i < failureCount && d < maxDelay; i++ { + d *= 2 + } + if d > maxDelay { + d = maxDelay + } + jitter := 0.8 + rand.Float64()*0.4 // [0.8, 1.2] + return time.Duration(float64(d) * jitter) +} diff --git a/grafana-alertcheck/internal/gate/source_fake_test.go b/grafana-alertcheck/internal/gate/source_fake_test.go new file mode 100644 index 000000000..b3ef23cca --- /dev/null +++ b/grafana-alertcheck/internal/gate/source_fake_test.go @@ -0,0 +1,142 @@ +package gate + +import ( + "context" + "fmt" + "sync" + "time" +) + +// fakeClock is a manually-advanced Clock — no test in this package sleeps on +// real time (§22). It is goroutine-safe (a concurrent fleet under -race must +// not trip on the double itself), but After always fires immediately, +// regardless of the requested duration or whether Advance was ever called. +// That is sufficient here: every retry/backoff test in this phase only needs +// to avoid a real sleep. It is NOT sufficient for a test that must prove a +// wait did not fire early — e.g. a P4 scheduler test asserting Due() doesn't +// return a rule before its next-due time. That needs a clock with a real +// waiter list keyed off Advance, which does not exist yet; build it when a +// phase actually needs it rather than guessing its shape now. +type fakeClock struct { + mu sync.Mutex + now time.Time +} + +func newFakeClock(now time.Time) *fakeClock { return &fakeClock{now: now} } + +func (c *fakeClock) Now() time.Time { + c.mu.Lock() + defer c.mu.Unlock() + return c.now +} + +func (c *fakeClock) Advance(d time.Duration) { + c.mu.Lock() + defer c.mu.Unlock() + c.now = c.now.Add(d) +} + +func (c *fakeClock) After(d time.Duration) <-chan time.Time { + c.mu.Lock() + fireAt := c.now.Add(d) + c.mu.Unlock() + ch := make(chan time.Time, 1) + ch <- fireAt + return ch +} + +// steppingClock advances by a fixed step on every Now() call, so a test can +// assert exact latency/skew-bound arithmetic (doRequest's three clock reads +// per attempt) without depending on real wall-clock timing. +type steppingClock struct { + mu sync.Mutex + now time.Time + step time.Duration +} + +func (c *steppingClock) Now() time.Time { + c.mu.Lock() + defer c.mu.Unlock() + t := c.now + c.now = c.now.Add(c.step) + return t +} + +func (c *steppingClock) After(d time.Duration) <-chan time.Time { + c.mu.Lock() + fireAt := c.now.Add(d) + c.mu.Unlock() + ch := make(chan time.Time, 1) + ch <- fireAt + return ch +} + +// scriptedObservation is one canned (Observation, error) pair a fakeSource +// returns from RuleState, in the order scripted. +type scriptedObservation struct { + obs Observation + err error +} + +// fakeSource is a scripted Source with no HTTP, goroutine-safe so a phase +// that polls several rules concurrently (P6) can share one instance across +// goroutines without tripping -race. P3 through at least P5 can construct +// one directly instead of talking to HTTP; a phase that needs it to behave +// like a live server under concurrent load beyond simple locking should +// verify that assumption rather than take this comment's word for it. +type fakeSource struct { + mu sync.Mutex + + version string + versionErr error + + defs []Definition + defsErr error + + // states maps a rule title to a queue of scripted results, popped one + // per call to RuleState. Once the queue is down to its last entry, that + // entry repeats — so a test can script the interesting transitions and + // let a long collection loop settle into steady state without scripting + // every single poll. + states map[string][]scriptedObservation +} + +func newFakeSource() *fakeSource { + return &fakeSource{states: make(map[string][]scriptedObservation)} +} + +func (f *fakeSource) Version(_ context.Context) (string, error) { + f.mu.Lock() + defer f.mu.Unlock() + return f.version, f.versionErr +} + +func (f *fakeSource) Definitions(_ context.Context) ([]Definition, error) { + f.mu.Lock() + defer f.mu.Unlock() + return f.defs, f.defsErr +} + +func (f *fakeSource) RuleState(_ context.Context, title string) (Observation, error) { + f.mu.Lock() + defer f.mu.Unlock() + q := f.states[title] + if len(q) == 0 { + return Observation{}, fmt.Errorf("fakeSource: no scripted response for %q", title) + } + next := q[0] + if len(q) > 1 { + f.states[title] = q[1:] + } + return next.obs, next.err +} + +// script appends one scripted (Observation, error) pair to be returned, in +// order, by RuleState(ctx, title). +func (f *fakeSource) script(title string, obs Observation, err error) { + f.mu.Lock() + defer f.mu.Unlock() + f.states[title] = append(f.states[title], scriptedObservation{obs: obs, err: err}) +} + +var _ Source = (*fakeSource)(nil) diff --git a/grafana-alertcheck/internal/gate/source_test.go b/grafana-alertcheck/internal/gate/source_test.go new file mode 100644 index 000000000..41bae2dfa --- /dev/null +++ b/grafana-alertcheck/internal/gate/source_test.go @@ -0,0 +1,559 @@ +package gate + +import ( + "context" + "errors" + "fmt" + "net/http" + "net/http/httptest" + "net/url" + "strings" + "sync" + "sync/atomic" + "testing" + "time" +) + +func healthBody(version string) string { + return fmt.Sprintf(`{"database":"ok","version":%q,"commit":"abc123"}`, version) +} + +func emptyStateBody() string { + return `{"status":"success","data":{"groups":[]}}` +} + +// rawHTTPServer starts an httptest server whose handler hijacks the +// connection and writes exactly the bytes respond returns, bypassing +// net/http's automatic Date-header insertion — the only way to test a +// response with no Date header at all, or a deliberately garbled one. +func rawHTTPServer(t *testing.T, respond func(r *http.Request) []byte) *httptest.Server { + t.Helper() + srv := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) { + hj, ok := w.(http.Hijacker) + if !ok { + t.Errorf("ResponseWriter does not support Hijacker") + return + } + conn, buf, err := hj.Hijack() + if err != nil { + t.Errorf("hijack: %v", err) + return + } + defer conn.Close() + if _, err := buf.Write(respond(r)); err != nil { + t.Errorf("write raw response: %v", err) + return + } + _ = buf.Flush() + })) + t.Cleanup(srv.Close) + return srv +} + +// rawResponse builds a minimal, fully-controlled HTTP/1.1 response: no +// header net/http would add unasked, in particular no automatic Date. +func rawResponse(status int, statusText string, headers map[string]string, body string) []byte { + out := fmt.Sprintf("HTTP/1.1 %d %s\r\n", status, statusText) + for k, v := range headers { + out += fmt.Sprintf("%s: %s\r\n", k, v) + } + out += fmt.Sprintf("Content-Length: %d\r\n", len(body)) + out += "Connection: close\r\n\r\n" + out += body + return []byte(out) +} + +func TestCheckGrafanaVersion(t *testing.T) { + cases := []struct { + version string + wantErr bool + wantContains []string // required substrings of the error message, per case, when wantErr + }{ + {"13.1.0", false, nil}, + {"13.0.0", false, nil}, + {"13.99.99", false, nil}, + {"12.9.9", true, []string{`"12.9.9"`, "13.0.0", "14.0.0"}}, + {"14.0.0", true, []string{`"14.0.0"`, "13.0.0", "14.0.0"}}, + {"14.1.0", true, []string{`"14.1.0"`, "13.0.0", "14.0.0"}}, + {"not-a-version", true, []string{`"not-a-version"`}}, + {"", true, nil}, + } + for _, c := range cases { + err := CheckGrafanaVersion(c.version) + if c.wantErr && err == nil { + t.Errorf("CheckGrafanaVersion(%q): want error, got nil", c.version) + continue + } + if !c.wantErr && err != nil { + t.Errorf("CheckGrafanaVersion(%q): unexpected error: %v", c.version, err) + continue + } + for _, want := range c.wantContains { + if !strings.Contains(err.Error(), want) { + t.Errorf("CheckGrafanaVersion(%q): error %q does not mention %q (the plan requires naming both what was found and what is supported)", c.version, err.Error(), want) + } + } + } +} + +func TestBackoffDelay(t *testing.T) { + base := time.Second + maxDelay := 30 * time.Second + maxWithJitter := maxDelay + maxDelay/5 + time.Millisecond + for n := 1; n <= 10; n++ { + d := backoffDelay(base, maxDelay, n) + if d <= 0 { + t.Fatalf("backoffDelay(_, _, %d) = %v, want > 0", n, d) + } + if d > maxWithJitter { + t.Fatalf("backoffDelay(_, _, %d) = %v, want <= ~%v", n, d, maxWithJitter) + } + } +} + +func TestHTTPSource_Version_HappyPath(t *testing.T) { + srv := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) { + if r.URL.Path != "/api/health" { + t.Errorf("path = %q, want /api/health", r.URL.Path) + } + w.Header().Set("Content-Type", "application/json") + _, _ = w.Write([]byte(healthBody("13.1.0"))) + })) + defer srv.Close() + + clock := newFakeClock(time.Now()) + src := NewHTTPSource(srv.URL, "", clock) + v, err := src.Version(context.Background()) + if err != nil { + t.Fatalf("Version(): unexpected error: %v", err) + } + if v != "13.1.0" { + t.Fatalf("Version() = %q, want 13.1.0", v) + } +} + +func TestHTTPSource_Version_NeverLogsToken(t *testing.T) { + const secret = "super-secret-token" + srv := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) { + if got := r.Header.Get("Authorization"); got != "Bearer "+secret { + t.Errorf("Authorization = %q, want Bearer %s", got, secret) + } + w.WriteHeader(http.StatusInternalServerError) + })) + defer srv.Close() + + clock := newFakeClock(time.Now()) + src := NewHTTPSource(srv.URL, secret, clock) + _, err := src.Version(context.Background()) + if err == nil { + t.Fatalf("Version(): want error, got nil") + } + if strings.Contains(err.Error(), secret) { + t.Fatalf("error %q leaks the token", err.Error()) + } +} + +func TestHTTPSource_RuleState_EmptyIsNotAnError(t *testing.T) { + srv := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) { + w.Header().Set("Content-Type", "application/json") + _, _ = w.Write([]byte(emptyStateBody())) + })) + defer srv.Close() + + clock := newFakeClock(time.Now()) + src := NewHTTPSource(srv.URL, "", clock) + obs, err := src.RuleState(context.Background(), "Anything") + if err != nil { + t.Fatalf("RuleState(): unexpected error: %v", err) + } + if len(obs.Rules) != 0 { + t.Fatalf("Rules = %+v, want empty (an authoritative 2xx is not a transport error)", obs.Rules) + } + if obs.GrafanaNow.IsZero() { + t.Fatalf("GrafanaNow is zero, want the response's Date header value (H4)") + } +} + +func TestHTTPSource_RuleState_EscapesRuleName(t *testing.T) { + var gotQuery string + srv := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) { + gotQuery = r.URL.RawQuery + if r.URL.Path != "/api/prometheus/grafana/api/v1/rules" { + t.Errorf("path = %q, want /api/prometheus/grafana/api/v1/rules", r.URL.Path) + } + w.Header().Set("Content-Type", "application/json") + _, _ = w.Write([]byte(emptyStateBody())) + })) + defer srv.Close() + + clock := newFakeClock(time.Now()) + src := NewHTTPSource(srv.URL, "", clock) + title := "[JD] No Job Proposals & More" + if _, err := src.RuleState(context.Background(), title); err != nil { + t.Fatalf("RuleState(): unexpected error: %v", err) + } + want := "rule_name=" + url.QueryEscape(title) + if gotQuery != want { + t.Fatalf("query = %q, want %q", gotQuery, want) + } +} + +func TestHTTPSource_Definitions_HappyPath(t *testing.T) { + body := readFixture(t, "ruler_rules.json") + srv := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) { + if r.URL.Path != "/api/ruler/grafana/api/v1/rules" { + t.Errorf("path = %q, want /api/ruler/grafana/api/v1/rules", r.URL.Path) + } + if r.URL.RawQuery != "" { + t.Errorf("query = %q, want none — Definitions reads the ruler API unfiltered", r.URL.RawQuery) + } + w.Header().Set("Content-Type", "application/json") + _, _ = w.Write(body) + })) + defer srv.Close() + + clock := newFakeClock(time.Now()) + src := NewHTTPSource(srv.URL, "", clock) + defs, err := src.Definitions(context.Background()) + if err != nil { + t.Fatalf("Definitions(): unexpected error: %v", err) + } + if len(defs) == 0 { + t.Fatalf("Definitions(): got 0 definitions from a fixture known to have some") + } +} + +func TestHTTPSource_Skew(t *testing.T) { + cases := []struct { + name string + drift time.Duration + wantErr bool + }{ + {"30s skew is fine", 30 * time.Second, false}, + {"120s skew is a hard error", 120 * time.Second, true}, + } + for _, c := range cases { + t.Run(c.name, func(t *testing.T) { + anchor := time.Now() + clock := newFakeClock(anchor) + var calls atomic.Int32 + srv := rawHTTPServer(t, func(r *http.Request) []byte { + calls.Add(1) + date := anchor.Add(c.drift).UTC().Format(http.TimeFormat) + return rawResponse(200, "OK", map[string]string{ + "Content-Type": "application/json", + "Date": date, + }, healthBody("13.1.0")) + }) + src := NewHTTPSource(srv.URL, "", clock) + _, err := src.Version(context.Background()) + if c.wantErr && err == nil { + t.Fatalf("Version(): want error, got nil") + } + if !c.wantErr && err != nil { + t.Fatalf("Version(): unexpected error: %v", err) + } + if c.wantErr && calls.Load() != 1 { + t.Fatalf("calls = %d, want 1 — a skew hard error must never be retried", calls.Load()) + } + }) + } +} + +func TestHTTPSource_MissingDateHeader(t *testing.T) { + var calls atomic.Int32 + srv := rawHTTPServer(t, func(r *http.Request) []byte { + calls.Add(1) + return rawResponse(200, "OK", map[string]string{ + "Content-Type": "application/json", + }, healthBody("13.1.0")) + }) + clock := newFakeClock(time.Now()) + src := NewHTTPSource(srv.URL, "", clock) + _, err := src.Version(context.Background()) + if err == nil { + t.Fatalf("Version(): want error, got nil (H4: a missing Date header is a hard error)") + } + if calls.Load() != 1 { + t.Fatalf("calls = %d, want 1 — a missing Date header must never be retried", calls.Load()) + } +} + +func TestHTTPSource_UnparseableDateHeader(t *testing.T) { + var calls atomic.Int32 + srv := rawHTTPServer(t, func(r *http.Request) []byte { + calls.Add(1) + return rawResponse(200, "OK", map[string]string{ + "Content-Type": "application/json", + "Date": "definitely not a date", + }, healthBody("13.1.0")) + }) + clock := newFakeClock(time.Now()) + src := NewHTTPSource(srv.URL, "", clock) + _, err := src.Version(context.Background()) + if err == nil { + t.Fatalf("Version(): want error, got nil (H4: an unparseable Date header is a hard error)") + } + if calls.Load() != 1 { + t.Fatalf("calls = %d, want 1 — an unparseable Date header must never be retried", calls.Load()) + } +} + +// TestHTTPSource_ObservationTiming pins the arithmetic behind Observation's +// three derived fields, not just the hard-limit behavior TestHTTPSource_Skew +// already covers: the sign of Skew, SkewBound as exactly RTT/2 to the +// response headers, and Latency as the full send-through-body-read span +// (not just the header round trip). steppingClock advances by a fixed 2s on +// every clock.Now() call, and doRequest calls Now() exactly three times per +// attempt (before send, after headers, after the body read), so the +// arithmetic is exact rather than a real-time approximation. +func TestHTTPSource_ObservationTiming(t *testing.T) { + cases := []struct { + name string + drift time.Duration + }{ + {"positive skew", 5 * time.Second}, + {"negative skew", -5 * time.Second}, + } + for _, c := range cases { + t.Run(c.name, func(t *testing.T) { + anchor := time.Now().Truncate(time.Second) + clock := &steppingClock{now: anchor, step: 2 * time.Second} + // With step=2s: t_send=anchor, t_headers=anchor+2s, t_bodyRead=anchor+4s. + // mid = t_send + (t_headers-t_send)/2 = anchor+1s, so serverDate = + // anchor+1s+drift makes Skew land on exactly `drift`. + srv := rawHTTPServer(t, func(r *http.Request) []byte { + date := anchor.Add(time.Second + c.drift).UTC().Format(http.TimeFormat) + return rawResponse(200, "OK", map[string]string{ + "Content-Type": "application/json", + "Date": date, + }, emptyStateBody()) + }) + src := NewHTTPSource(srv.URL, "", clock) + obs, err := src.RuleState(context.Background(), "Anything") + if err != nil { + t.Fatalf("RuleState(): unexpected error: %v", err) + } + if obs.Skew != c.drift { + t.Errorf("Skew = %v, want %v", obs.Skew, c.drift) + } + if obs.SkewBound != time.Second { + t.Errorf("SkewBound = %v, want 1s (RTT/2 with a 2s round trip to headers)", obs.SkewBound) + } + if obs.Latency != 4*time.Second { + t.Errorf("Latency = %v, want 4s (send through full body read, §5.2) — not just the 2s header round trip", obs.Latency) + } + }) + } +} + +func TestHTTPSource_Retry_TransientRecovers(t *testing.T) { + var mu sync.Mutex + calls := 0 + srv := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) { + mu.Lock() + calls++ + n := calls + mu.Unlock() + if n <= 2 { + w.WriteHeader(http.StatusInternalServerError) + return + } + w.Header().Set("Content-Type", "application/json") + _, _ = w.Write([]byte(healthBody("13.1.0"))) + })) + defer srv.Close() + + clock := newFakeClock(time.Now()) + src := NewHTTPSource(srv.URL, "", clock) + v, err := src.Version(context.Background()) + if err != nil { + t.Fatalf("Version(): unexpected error after a transient failure: %v", err) + } + if v != "13.1.0" { + t.Fatalf("Version() = %q, want 13.1.0", v) + } + mu.Lock() + n := calls + mu.Unlock() + if n != 3 { + t.Fatalf("calls = %d, want 3 (2 failures + 1 success)", n) + } +} + +func TestHTTPSource_Retry_ExceedsLimit(t *testing.T) { + var calls atomic.Int32 + srv := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) { + calls.Add(1) + w.WriteHeader(http.StatusInternalServerError) + })) + defer srv.Close() + + clock := newFakeClock(time.Now()) + src := NewHTTPSource(srv.URL, "", clock) + _, err := src.Version(context.Background()) + if err == nil { + t.Fatalf("Version(): want error, got nil") + } + if n := calls.Load(); n != 6 { + t.Fatalf("calls = %d, want 6 (maxSequentialFailures=5 tolerates 5, gives up on the 6th)", n) + } + assertRetryExhausted(t, err, 6) +} + +func TestHTTPSource_RuleState_GarbageBodyRetries(t *testing.T) { + var mu sync.Mutex + calls := 0 + srv := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) { + mu.Lock() + calls++ + n := calls + mu.Unlock() + w.Header().Set("Content-Type", "application/json") + if n <= 2 { + // A 2xx with a body that fails ParseState — classified as a + // transient *TransportError (source.go), not a hard schema + // break, so it must retry rather than fail immediately. + _, _ = w.Write([]byte("{not valid json")) + return + } + _, _ = w.Write([]byte(emptyStateBody())) + })) + defer srv.Close() + + clock := newFakeClock(time.Now()) + src := NewHTTPSource(srv.URL, "", clock) + obs, err := src.RuleState(context.Background(), "Anything") + if err != nil { + t.Fatalf("RuleState(): unexpected error after a transient garbage body: %v", err) + } + if len(obs.Rules) != 0 { + t.Fatalf("Rules = %+v, want empty", obs.Rules) + } + mu.Lock() + n := calls + mu.Unlock() + if n != 3 { + t.Fatalf("calls = %d, want 3 (2 unparseable bodies + 1 valid one)", n) + } +} + +func TestHTTPSource_Definitions_GarbageBodyGivesUp(t *testing.T) { + var calls atomic.Int32 + srv := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) { + calls.Add(1) + w.Header().Set("Content-Type", "application/json") + _, _ = w.Write([]byte("{not valid json")) + })) + defer srv.Close() + + clock := newFakeClock(time.Now()) + src := NewHTTPSource(srv.URL, "", clock) + _, err := src.Definitions(context.Background()) + if err == nil { + t.Fatalf("Definitions(): want error, got nil") + } + if n := calls.Load(); n != 6 { + t.Fatalf("calls = %d, want 6 — a persistently unparseable 2xx body retries like any other transport failure", n) + } + assertRetryExhausted(t, err, 6) +} + +func TestHTTPSource_NetworkFailureRetries(t *testing.T) { + // A server that is never listening: every attempt is a network failure, + // classified as *TransportError, so this exercises the same retry path + // as a 5xx without needing a real listening socket per failure. + clock := newFakeClock(time.Now()) + src := NewHTTPSource("http://127.0.0.1:1", "", clock) + _, err := src.Version(context.Background()) + if err == nil { + t.Fatalf("Version(): want error, got nil") + } + assertRetryExhausted(t, err, 6) +} + +// assertRetryExhausted checks the two properties a retry give-up must have: +// it names how many failures it gave up after, and — the regression this +// pins — it is never itself classified as a *TransportError. If it were, +// something one layer up that also retries on *TransportError would treat an +// already-exhausted give-up as retryable again, the exact conflation §19.3 +// case 1 forbids. +func assertRetryExhausted(t *testing.T, err error, wantFailures int) { + t.Helper() + var reErr *RetryExhaustedError + if !errors.As(err, &reErr) { + t.Fatalf("error %v (%T): want a *RetryExhaustedError", err, err) + } + if reErr.Failures != wantFailures { + t.Errorf("RetryExhaustedError.Failures = %d, want %d", reErr.Failures, wantFailures) + } + if !strings.Contains(err.Error(), fmt.Sprintf("gave up after %d", wantFailures)) { + t.Errorf("error %q does not name the failure count", err.Error()) + } + if _, ok := errors.AsType[*TransportError](err); ok { + t.Fatalf("error %v (%T) is classified as *TransportError — an exhausted retry must be a terminal, non-retryable error", err, err) + } +} + +func TestFakeClock(t *testing.T) { + start := time.Date(2026, 1, 1, 0, 0, 0, 0, time.UTC) + c := newFakeClock(start) + if !c.Now().Equal(start) { + t.Fatalf("Now() = %v, want %v", c.Now(), start) + } + c.Advance(5 * time.Minute) + want := start.Add(5 * time.Minute) + if !c.Now().Equal(want) { + t.Fatalf("Now() after Advance = %v, want %v", c.Now(), want) + } + + select { + case fired := <-c.After(time.Hour): + if !fired.Equal(want.Add(time.Hour)) { + t.Fatalf("After fired with %v, want %v", fired, want.Add(time.Hour)) + } + default: + t.Fatalf("After(1h) did not fire immediately") + } +} + +func TestFakeSource(t *testing.T) { + f := newFakeSource() + f.version = "13.1.0" + f.defs = []Definition{{UID: "u1", Title: "Rule One"}} + + ctx := context.Background() + if v, err := f.Version(ctx); err != nil || v != "13.1.0" { + t.Fatalf("Version() = (%q, %v), want (13.1.0, nil)", v, err) + } + if defs, err := f.Definitions(ctx); err != nil || len(defs) != 1 { + t.Fatalf("Definitions() = (%v, %v), want one definition", defs, err) + } + + f.script("Rule One", Observation{Rules: []StateRule{{UID: "u1"}}}, nil) + f.script("Rule One", Observation{}, fmt.Errorf("boom")) + f.script("Rule One", Observation{Rules: nil}, nil) + + obs, err := f.RuleState(ctx, "Rule One") + if err != nil || len(obs.Rules) != 1 { + t.Fatalf("RuleState() call 1 = (%v, %v), want one rule, no error", obs, err) + } + if _, err := f.RuleState(ctx, "Rule One"); err == nil { + t.Fatalf("RuleState() call 2: want the scripted error, got nil") + } + obs, err = f.RuleState(ctx, "Rule One") + if err != nil { + t.Fatalf("RuleState() call 3: unexpected error: %v", err) + } + if obs.Rules != nil { + t.Fatalf("RuleState() call 3: Rules = %v, want nil (last script entry, then repeats)", obs.Rules) + } + obs, err = f.RuleState(ctx, "Rule One") + if err != nil || obs.Rules != nil { + t.Fatalf("RuleState() call 4: want the last scripted entry to repeat, got (%v, %v)", obs, err) + } + + if _, err := f.RuleState(ctx, "Unscripted Rule"); err == nil { + t.Fatalf("RuleState() for an unscripted title: want an error, got nil") + } +} From 8e8a430ef00fb94654e2962449d17eea0aa1eb0b Mon Sep 17 00:00:00 2001 From: Bartek Tofel Date: Tue, 1 Sep 2026 16:17:29 +0200 Subject: [PATCH 06/43] chore: enhance unit tests --- .../internal/gate/duration_test.go | 20 ++++++++++--------- .../internal/gate/source_test.go | 4 +++- 2 files changed, 14 insertions(+), 10 deletions(-) diff --git a/grafana-alertcheck/internal/gate/duration_test.go b/grafana-alertcheck/internal/gate/duration_test.go index 73a0c00e1..ba3a1629a 100644 --- a/grafana-alertcheck/internal/gate/duration_test.go +++ b/grafana-alertcheck/internal/gate/duration_test.go @@ -47,15 +47,17 @@ func TestParsePromDuration(t *testing.T) { func TestParsePromDuration_Errors(t *testing.T) { cases := []string{ - "5", // bare number, no unit - "-5m", // negative - "5x", // unknown unit - "30m1h", // ascending order (must be descending) - "1h1h", // duplicate unit - "m", // unit with no number - "1.5h", // fractional number not supported by this grammar - "1 h", // whitespace - "300y", // overflows time.Duration (int64 nanoseconds) — must error, not wrap negative + "5", // bare number, no unit + "-5m", // negative + "5x", // unknown unit + "30m1h", // ascending order (must be descending) + "1h1h", // duplicate unit + "m", // unit with no number + "1.5h", // fractional number not supported by this grammar + "1 h", // whitespace + "300y", // overflows time.Duration (int64 nanoseconds) — must error, not wrap negative + "1w2d3h4m5s6ms7us8ns", // too many units + "carrot", // completely invalid } for _, in := range cases { if _, err := ParsePromDuration(in); err == nil { diff --git a/grafana-alertcheck/internal/gate/source_test.go b/grafana-alertcheck/internal/gate/source_test.go index 41bae2dfa..d25f1f2ec 100644 --- a/grafana-alertcheck/internal/gate/source_test.go +++ b/grafana-alertcheck/internal/gate/source_test.go @@ -229,8 +229,10 @@ func TestHTTPSource_Skew(t *testing.T) { drift time.Duration wantErr bool }{ + {"0s skew is fine", 0 * time.Second, false}, {"30s skew is fine", 30 * time.Second, false}, - {"120s skew is a hard error", 120 * time.Second, true}, + {"60s skew is fine", 60 * time.Second, false}, + {"61s skew is a hard error", 61 * time.Second, true}, } for _, c := range cases { t.Run(c.name, func(t *testing.T) { From ba93502745852f699df17ca28910a4f86aa9decf Mon Sep 17 00:00:00 2001 From: Bartek Tofel Date: Mon, 7 Sep 2026 12:28:06 +0200 Subject: [PATCH 07/43] chore: address code review comments --- grafana-alertcheck/internal/gate/source.go | 16 ++++++++- .../internal/gate/source_fake_test.go | 9 ++++- .../internal/gate/source_test.go | 33 +++++++++++++++++++ 3 files changed, 56 insertions(+), 2 deletions(-) diff --git a/grafana-alertcheck/internal/gate/source.go b/grafana-alertcheck/internal/gate/source.go index 46de818bb..63cce7eb6 100644 --- a/grafana-alertcheck/internal/gate/source.go +++ b/grafana-alertcheck/internal/gate/source.go @@ -19,6 +19,14 @@ import ( // block once P4 exists; defined here because P2 needs it first. const skewHardLimit = 60 * time.Second +// maxResponseBytes caps how much doRequest will read from a response body. +// It is far above the largest real payload the gate retrieves (the ~600 KB +// high-cardinality state fetch documented in parse_state_test.go), so a +// legitimate response never trips it, but a misbehaving server or proxy +// streaming an unbounded body is cut off loudly instead of OOMing the process +// across retries. +const maxResponseBytes = 25 << 20 // 25 MiB + // Clock is the seam that lets tests advance time without sleeping (§22) — the // only two operations the gate ever needs from a clock. type Clock interface { @@ -295,11 +303,17 @@ func (s *httpSource) doRequest(ctx context.Context, path string) (requestResult, } defer resp.Body.Close() - b, readErr := io.ReadAll(resp.Body) + b, readErr := io.ReadAll(io.LimitReader(resp.Body, maxResponseBytes)) tBodyRead := s.clock.Now() if readErr != nil { return requestResult{}, &TransportError{Err: fmt.Errorf("read response body (status %d): %w", resp.StatusCode, readErr)} } + if len(b) >= maxResponseBytes { + // Hard error, never retried: a response this large is a stable + // property of the server's reply, not a transient network hiccup, so + // retrying would just reallocate the same bounded-but-pointless body. + return requestResult{}, fmt.Errorf("response body exceeded %d bytes", maxResponseBytes) + } latency := tBodyRead.Sub(tSend) if resp.StatusCode < 200 || resp.StatusCode >= 300 { diff --git a/grafana-alertcheck/internal/gate/source_fake_test.go b/grafana-alertcheck/internal/gate/source_fake_test.go index b3ef23cca..197788a24 100644 --- a/grafana-alertcheck/internal/gate/source_fake_test.go +++ b/grafana-alertcheck/internal/gate/source_fake_test.go @@ -3,6 +3,7 @@ package gate import ( "context" "fmt" + "slices" "sync" "time" ) @@ -114,7 +115,10 @@ func (f *fakeSource) Version(_ context.Context) (string, error) { func (f *fakeSource) Definitions(_ context.Context) ([]Definition, error) { f.mu.Lock() defer f.mu.Unlock() - return f.defs, f.defsErr + // Defensive copy: Definition is a value type, so Clone copies the + // full slice contents, not just the header — callers are free to mutate + // what they got back without racing or corrupting later reads. + return slices.Clone(f.defs), f.defsErr } func (f *fakeSource) RuleState(_ context.Context, title string) (Observation, error) { @@ -128,6 +132,9 @@ func (f *fakeSource) RuleState(_ context.Context, title string) (Observation, er if len(q) > 1 { f.states[title] = q[1:] } + // Defensive copy of the shared Rules slice so a caller mutating the + // returned Observation can't corrupt the scripted state other calls read. + next.obs.Rules = slices.Clone(next.obs.Rules) return next.obs, next.err } diff --git a/grafana-alertcheck/internal/gate/source_test.go b/grafana-alertcheck/internal/gate/source_test.go index d25f1f2ec..ba864a1a8 100644 --- a/grafana-alertcheck/internal/gate/source_test.go +++ b/grafana-alertcheck/internal/gate/source_test.go @@ -461,6 +461,39 @@ func TestHTTPSource_Definitions_GarbageBodyGivesUp(t *testing.T) { assertRetryExhausted(t, err, 6) } +func TestHTTPSource_ResponseBodyTooLarge(t *testing.T) { + var calls atomic.Int32 + srv := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) { + calls.Add(1) + w.Header().Set("Content-Type", "application/json") + // Stream more than maxResponseBytes in 1 MiB chunks so the test never + // materializes the whole oversized body in its own memory — doRequest + // must cut it off, not buffer it. + chunk := strings.Repeat("x", 1<<20) + for i := 0; i < (maxResponseBytes/(1<<20))+2; i++ { + if _, err := w.Write([]byte(chunk)); err != nil { + return + } + } + })) + defer srv.Close() + + clock := newFakeClock(time.Now()) + src := NewHTTPSource(srv.URL, "", clock) + _, err := src.Version(context.Background()) + if err == nil { + t.Fatalf("Version(): want error, got nil (an oversized body must fail loudly)") + } + if !strings.Contains(err.Error(), "exceeded") { + t.Fatalf("error %q does not name the size limit", err.Error()) + } + // An oversized body is a stable condition, not a transient one: it must + // fail hard on the first attempt, never burning retries re-reading it. + if n := calls.Load(); n != 1 { + t.Fatalf("calls = %d, want 1 — an oversized body must never be retried", n) + } +} + func TestHTTPSource_NetworkFailureRetries(t *testing.T) { // A server that is never listening: every attempt is a network failure, // classified as *TransportError, so this exercises the same retry path From d89173ddb665968fd032ea6aade494473bf84df2 Mon Sep 17 00:00:00 2001 From: Bartek Tofel Date: Mon, 31 Aug 2026 14:05:48 +0200 Subject: [PATCH 08/43] chore: implement phase 3 Add Resolve() for alert name resolution (uid:/Title/Folder/Title/ Folder/Group/Title forms, UID collapse, no-match suggestions) and the grafana-alertcheck CLI's list subcommand, the first runnable piece of the gate. Incorporates review fixes: reject empty path segments in classifyForm, guard uid: against an empty suffix, scope the no-match rule count and suggestions to supported rule kinds only, and exit 0 on -h/--help. --- .../cmd/grafana-alertcheck/env.go | 22 ++ .../cmd/grafana-alertcheck/list.go | 93 ++++++ .../cmd/grafana-alertcheck/list_test.go | 102 +++++++ .../cmd/grafana-alertcheck/main.go | 43 ++- .../cmd/grafana-alertcheck/main_test.go | 60 ++++ grafana-alertcheck/internal/gate/resolve.go | 209 ++++++++++++++ .../internal/gate/resolve_test.go | 273 ++++++++++++++++++ 7 files changed, 800 insertions(+), 2 deletions(-) create mode 100644 grafana-alertcheck/cmd/grafana-alertcheck/env.go create mode 100644 grafana-alertcheck/cmd/grafana-alertcheck/list.go create mode 100644 grafana-alertcheck/cmd/grafana-alertcheck/list_test.go create mode 100644 grafana-alertcheck/cmd/grafana-alertcheck/main_test.go create mode 100644 grafana-alertcheck/internal/gate/resolve.go create mode 100644 grafana-alertcheck/internal/gate/resolve_test.go diff --git a/grafana-alertcheck/cmd/grafana-alertcheck/env.go b/grafana-alertcheck/cmd/grafana-alertcheck/env.go new file mode 100644 index 000000000..e5702a6d1 --- /dev/null +++ b/grafana-alertcheck/cmd/grafana-alertcheck/env.go @@ -0,0 +1,22 @@ +package main + +import ( + "fmt" + "os" +) + +// grafanaEnv reads the connection details from the environment only, never +// from a flag — a flag value lands in the process argv and in CI logs, and +// the token must never be logged or otherwise surface in an error string +// (§20.2). +func grafanaEnv() (url, token string, err error) { + url = os.Getenv("GRAFANA_URL") + if url == "" { + return "", "", fmt.Errorf("GRAFANA_URL is not set") + } + token = os.Getenv("GRAFANA_TOKEN") + if token == "" { + return "", "", fmt.Errorf("GRAFANA_TOKEN is not set") + } + return url, token, nil +} diff --git a/grafana-alertcheck/cmd/grafana-alertcheck/list.go b/grafana-alertcheck/cmd/grafana-alertcheck/list.go new file mode 100644 index 000000000..687e5dd9f --- /dev/null +++ b/grafana-alertcheck/cmd/grafana-alertcheck/list.go @@ -0,0 +1,93 @@ +package main + +import ( + "context" + "fmt" + "io" + "sort" + "text/tabwriter" + + "github.com/smartcontractkit/chainlink-testing-framework/grafana-alertcheck/internal/gate" +) + +// runList reads every rule definition from the ruler endpoint and prints one +// line per rule: its kind, its Folder/Group/Title, and its uid. This is what +// makes the gate runnable end to end before any coverage logic exists (§9 +// rule 4) — it validates auth, the ruler parse, and the shapes Resolve +// matches against, all against a real Grafana. It is also the "did you mean" +// surface §17.2's no-match error points operators at. +func runList(args []string, stdout, stderr io.Writer) int { + if len(args) != 0 { + fmt.Fprintf(stderr, "list takes no arguments, got %v\n", args) + return 2 + } + + url, token, err := grafanaEnv() + if err != nil { + fmt.Fprintln(stderr, err) + return 2 + } + + // context.Background(), no outer deadline: httpSource bounds every single + // attempt with its http.Client's 30s Timeout (source.go) and gives up + // after maxSequentialFailures consecutive transport errors, so this call + // always terminates. It can still take minutes end-to-end under repeated + // transient failures (5 retries * up to 30s backoff each, per call) — an + // acceptable wait for an interactive `list`, not for `watch`/`check`, + // which get their own deadlines from `--until`/`to` in P10. + src := gate.NewHTTPSource(url, token, gate.SystemClock{}) + version, err := src.Version(context.Background()) + if err != nil { + fmt.Fprintf(stderr, "checking grafana version: %v\n", err) + return 2 + } + if err := gate.CheckGrafanaVersion(version); err != nil { + fmt.Fprintln(stderr, err) + return 2 + } + + defs, err := src.Definitions(context.Background()) + if err != nil { + fmt.Fprintf(stderr, "reading rule definitions: %v\n", err) + return 2 + } + + sort.Slice(defs, func(i, j int) bool { + if defs[i].Folder != defs[j].Folder { + return defs[i].Folder < defs[j].Folder + } + if defs[i].Group != defs[j].Group { + return defs[i].Group < defs[j].Group + } + return defs[i].Title < defs[j].Title + }) + + tw := tabwriter.NewWriter(stdout, 0, 4, 2, ' ', 0) + fmt.Fprintln(tw, "KIND\tFOLDER\tGROUP\tTITLE\tUID") + for _, d := range defs { + fmt.Fprintf(tw, "%s\t%s\t%s\t%s\t%s\n", kindLabel(d.Kind), d.Folder, d.Group, d.Title, uidOrDash(d.UID)) + } + if err := tw.Flush(); err != nil { + fmt.Fprintf(stderr, "writing output: %v\n", err) + return 2 + } + return 0 +} + +func kindLabel(k gate.RuleKind) string { + switch k { + case gate.KindDatasourceManaged: + return "datasource-managed" + case gate.KindRecording: + return "recording" + default: + return "grafana-managed" + } +} + +func uidOrDash(uid string) string { + if uid == "" { + return "-" + } + return uid +} diff --git a/grafana-alertcheck/cmd/grafana-alertcheck/list_test.go b/grafana-alertcheck/cmd/grafana-alertcheck/list_test.go new file mode 100644 index 000000000..119326c2c --- /dev/null +++ b/grafana-alertcheck/cmd/grafana-alertcheck/list_test.go @@ -0,0 +1,102 @@ +package main + +import ( + "bytes" + "fmt" + "net/http" + "net/http/httptest" + "strings" + "testing" +) + +const rulerBody = `{ + "Example-Zone-A": [ + { + "name": "Gateway", + "rules": [ + { + "for": "5m", + "grafana_alert": { + "title": "Example No Gateways Available", + "uid": "rule0000006a", + "namespace_uid": "folder0000006", + "intervalSeconds": 60, + "no_data_state": "OK", + "exec_err_state": "OK", + "is_paused": false + } + } + ] + } + ] +}` + +func healthBody(version string) string { + return fmt.Sprintf(`{"database":"ok","version":%q,"commit":"abc123"}`, version) +} + +func grafanaTestServer(t *testing.T, version string) *httptest.Server { + t.Helper() + srv := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) { + w.Header().Set("Content-Type", "application/json") + switch r.URL.Path { + case "/api/health": + _, _ = w.Write([]byte(healthBody(version))) + case "/api/ruler/grafana/api/v1/rules": + _, _ = w.Write([]byte(rulerBody)) + default: + t.Errorf("unexpected path %q", r.URL.Path) + w.WriteHeader(http.StatusNotFound) + } + })) + t.Cleanup(srv.Close) + return srv +} + +func TestRunList_HappyPath(t *testing.T) { + srv := grafanaTestServer(t, "13.1.0") + t.Setenv("GRAFANA_URL", srv.URL) + t.Setenv("GRAFANA_TOKEN", "test-token") + + var stdout, stderr bytes.Buffer + code := run([]string{"list"}, &stdout, &stderr) + if code != 0 { + t.Fatalf("code = %d, want 0; stderr = %q", code, stderr.String()) + } + out := stdout.String() + if !strings.Contains(out, "rule0000006a") { + t.Errorf("stdout = %q, want it to list rule0000006a", out) + } + if !strings.Contains(out, "Example No Gateways Available") { + t.Errorf("stdout = %q, want it to list the rule title", out) + } + if !strings.Contains(out, "grafana-managed") { + t.Errorf("stdout = %q, want it to name the rule kind", out) + } +} + +func TestRunList_UnsupportedVersion(t *testing.T) { + srv := grafanaTestServer(t, "12.5.0") + t.Setenv("GRAFANA_URL", srv.URL) + t.Setenv("GRAFANA_TOKEN", "test-token") + + var stdout, stderr bytes.Buffer + code := run([]string{"list"}, &stdout, &stderr) + if code != 2 { + t.Fatalf("code = %d, want 2", code) + } + if !strings.Contains(stderr.String(), "12.5.0") { + t.Fatalf("stderr = %q, want it to name the unsupported version", stderr.String()) + } +} + +func TestRunList_RejectsArgs(t *testing.T) { + t.Setenv("GRAFANA_URL", "http://example.invalid") + t.Setenv("GRAFANA_TOKEN", "test-token") + + var stdout, stderr bytes.Buffer + code := run([]string{"list", "extra"}, &stdout, &stderr) + if code != 2 { + t.Fatalf("code = %d, want 2", code) + } +} diff --git a/grafana-alertcheck/cmd/grafana-alertcheck/main.go b/grafana-alertcheck/cmd/grafana-alertcheck/main.go index 150d617d4..d7968f5a8 100644 --- a/grafana-alertcheck/cmd/grafana-alertcheck/main.go +++ b/grafana-alertcheck/cmd/grafana-alertcheck/main.go @@ -1,7 +1,46 @@ +// Command grafana-alertcheck is the CLI entry point for the gate. P3 wires +// only the `list` subcommand — enough to validate auth, the ruler parse, and +// resolution against a real Grafana before any coverage logic exists (§9 rule +// 4, "reach runnable at PR 5"). P10 extends this file with `watch` and +// `check`. package main -import "os" +import ( + "fmt" + "io" + "os" +) func main() { - os.Exit(2) + os.Exit(run(os.Args[1:], os.Stdout, os.Stderr)) +} + +const usage = "usage: grafana-alertcheck " + +// run is the whole of main's testable surface: parse the subcommand, dispatch, +// return the process exit code. Exit codes below 2 (pass/violations) belong to +// `check` alone (§20.3, P10); every failure reachable from here — a missing +// subcommand, a bad flag, a transport or auth failure — is a could-not-check +// condition and maps to 2, never to 0 or 1 (H7). +// +// Requested help (-h/--help) is not a failure — it is the one exception to +// that rule. Convention (and every stdlib flag.FlagSet default) is exit 0 to +// stdout for help the caller asked for, reserving 2/stderr for help printed +// *because* something else went wrong (no subcommand, an unknown one). +func run(args []string, stdout, stderr io.Writer) int { + if len(args) == 0 { + fmt.Fprintln(stderr, usage) + return 2 + } + + switch args[0] { + case "list": + return runList(args[1:], stdout, stderr) + case "-h", "-help", "--help": + fmt.Fprintln(stdout, usage) + return 0 + default: + fmt.Fprintf(stderr, "unknown subcommand %q\n", args[0]) + return 2 + } } diff --git a/grafana-alertcheck/cmd/grafana-alertcheck/main_test.go b/grafana-alertcheck/cmd/grafana-alertcheck/main_test.go new file mode 100644 index 000000000..1d7906674 --- /dev/null +++ b/grafana-alertcheck/cmd/grafana-alertcheck/main_test.go @@ -0,0 +1,60 @@ +package main + +import ( + "bytes" + "strings" + "testing" +) + +func TestRun_NoArgs(t *testing.T) { + var stdout, stderr bytes.Buffer + code := run(nil, &stdout, &stderr) + if code != 2 { + t.Fatalf("code = %d, want 2", code) + } + if !strings.Contains(stderr.String(), "usage") { + t.Fatalf("stderr = %q, want a usage message", stderr.String()) + } +} + +func TestRun_Help(t *testing.T) { + for _, flag := range []string{"-h", "-help", "--help"} { + t.Run(flag, func(t *testing.T) { + var stdout, stderr bytes.Buffer + code := run([]string{flag}, &stdout, &stderr) + if code != 0 { + t.Fatalf("code = %d, want 0 (requested help is not a could-not-check condition)", code) + } + if !strings.Contains(stdout.String(), "usage") { + t.Fatalf("stdout = %q, want a usage message", stdout.String()) + } + if stderr.String() != "" { + t.Fatalf("stderr = %q, want empty — help goes to stdout", stderr.String()) + } + }) + } +} + +func TestRun_UnknownSubcommand(t *testing.T) { + var stdout, stderr bytes.Buffer + code := run([]string{"bogus"}, &stdout, &stderr) + if code != 2 { + t.Fatalf("code = %d, want 2", code) + } + if !strings.Contains(stderr.String(), `"bogus"`) { + t.Fatalf("stderr = %q, want it to name the unknown subcommand", stderr.String()) + } +} + +func TestRun_List_MissingEnv(t *testing.T) { + t.Setenv("GRAFANA_URL", "") + t.Setenv("GRAFANA_TOKEN", "") + var stdout, stderr bytes.Buffer + code := run([]string{"list"}, &stdout, &stderr) + if code != 2 { + t.Fatalf("code = %d, want 2", code) + } + if !strings.Contains(stderr.String(), "GRAFANA_URL") { + t.Fatalf("stderr = %q, want it to name the missing env var", stderr.String()) + } +} diff --git a/grafana-alertcheck/internal/gate/resolve.go b/grafana-alertcheck/internal/gate/resolve.go new file mode 100644 index 000000000..3c3710038 --- /dev/null +++ b/grafana-alertcheck/internal/gate/resolve.go @@ -0,0 +1,209 @@ +package gate + +import ( + "fmt" + "slices" + "sort" + "strings" +) + +// Resolve turns the operator-supplied alert names into resolved Definitions +// (§17). Order is load-bearing (§17.3): +// +// 1. Trim each name. +// 2. Discard empty lines. +// 3. Resolve each name to a UID (this is what resolveOne does). +// 4. Collapse the result by UID — two names hitting the same rule is a note, +// never an error (almost always a copy mistake, and a message costs the +// user less than a failure). +// +// The caller-visible consequence: len(resolved) is the count *after* the +// collapse. A later phase's MinObserved must default from that length, never +// from len(names) — using the input line count would make one rule named +// twice turn an achievable default into an unsatisfiable one (§17.3). +func Resolve(defs []Definition, names []string, folder string) (resolved []Definition, notes []string, err error) { + seenUID := map[string]string{} // uid -> the first input name that resolved to it + for _, raw := range names { + name := strings.TrimSpace(raw) + if name == "" { + continue + } + + def, rerr := resolveOne(defs, name, folder) + if rerr != nil { + return nil, nil, rerr + } + + if firstName, ok := seenUID[def.UID]; ok { + notes = append(notes, fmt.Sprintf( + "%q and %q both resolve to %s (uid:%s); counted once", firstName, name, def.Title, def.UID)) + continue + } + seenUID[def.UID] = name + resolved = append(resolved, def) + } + return resolved, notes, nil +} + +// resolveOne resolves a single trimmed, non-empty name against defs (§17.1): +// one match wins outright, zero is an error with suggestions, two or more is +// an error listing every candidate. folder scopes a bare title (no "/" in the +// name) to one folder; it is ignored for the "Folder/Title" and +// "Folder/Group/Title" forms, which already name their own folder. +// +// Policy on unsupported kinds (datasource-managed, recording) — decided here +// because §17.1 only says to refuse them, not how they interact with the +// no-match/ambiguous surfaces: a name can still match an unsupported rule (so +// naming one by title still gets the specific, named refusal, not a bare "no +// match"), but only *supported* candidates count for ambiguity — an +// unsupported rule sharing a title with a supported one is resolved silently +// in the supported rule's favor rather than reported as ambiguous — and the +// "%d rules available" count and substring suggestions in a genuine no-match +// are scoped to supported rules only, so an unsupported rule never inflates +// or pollutes either. uid: is always exact regardless of kind (typically +// copy-pasted from `list`, which already shows Kind). +func resolveOne(defs []Definition, name, folder string) (Definition, error) { + if uid, ok := strings.CutPrefix(name, "uid:"); ok { + if uid != "" { + for _, d := range defs { + if d.UID == uid { + return refuseUnsupportedKind(name, d) + } + } + } + // uid == "" falls through to the same message as "not found": several + // Definition kinds legitimately carry UID == "" (datasource-managed + // rules have no uid at all, P1.3), so matching on an empty suffix + // would silently hit one of those and report a misleading + // kind-specific refusal for what is really an empty/typo'd uid. This + // deliberately does not go through noMatchError: that function's + // substring suggestion would degenerate to an empty needle, which + // strings.Contains matches against every title — printing the whole + // fleet instead of a real suggestion. + return Definition{}, fmt.Errorf("no rule matched %q: no rule has this uid (run 'grafana-alertcheck list' to see uids)", name) + } + + wantFolder, wantGroup, wantTitle, err := classifyForm(name, folder) + if err != nil { + return Definition{}, err + } + + var supportedCandidates, unsupportedCandidates []Definition + for _, d := range defs { + if wantFolder != "" && d.Folder != wantFolder { + continue + } + if wantGroup != "" && d.Group != wantGroup { + continue + } + if d.Title != wantTitle { + continue + } + if d.Kind == KindGrafanaManaged { + supportedCandidates = append(supportedCandidates, d) + } else { + unsupportedCandidates = append(unsupportedCandidates, d) + } + } + + switch { + case len(supportedCandidates) == 1: + return supportedCandidates[0], nil + case len(supportedCandidates) > 1: + return Definition{}, ambiguousError(name, supportedCandidates) + case len(unsupportedCandidates) > 0: + return refuseUnsupportedKind(name, unsupportedCandidates[0]) + default: + return Definition{}, noMatchError(supportedDefs(defs), name, wantTitle) + } +} + +// supportedDefs filters out the two kinds §17.1 refuses. Only these +// participate in name-based matching, the no-match rule count, and substring +// suggestions (see the policy note on resolveOne). +func supportedDefs(defs []Definition) []Definition { + out := make([]Definition, 0, len(defs)) + for _, d := range defs { + if d.Kind == KindGrafanaManaged { + out = append(out, d) + } + } + return out +} + +// classifyForm splits name into the Title | Folder/Title | Folder/Group/Title +// forms (§17). A bare title is scoped by folder when the caller supplied one; +// the two- and three-segment forms already carry their own folder and ignore +// it. +// +// Every segment must be non-empty. Without this, "/Title" would parse as an +// empty wantFolder — silently dropping the folder filter and matching +// unscoped, a fail-open — and "Folder/" would parse as an empty wantTitle, +// which would then feed noMatchError's substring search an empty needle that +// matches every title. +func classifyForm(name, folder string) (wantFolder, wantGroup, wantTitle string, err error) { + parts := strings.Split(name, "/") + if slices.Contains(parts, "") { + return "", "", "", fmt.Errorf("no rule matched %q: empty /-separated segment (want Title, Folder/Title, or Folder/Group/Title)", name) + } + switch len(parts) { + case 1: + return folder, "", parts[0], nil + case 2: + return parts[0], "", parts[1], nil + case 3: + return parts[0], parts[1], parts[2], nil + default: + return "", "", "", fmt.Errorf("no rule matched %q: too many /-separated segments (want Title, Folder/Title, or Folder/Group/Title)", name) + } +} + +// refuseUnsupportedKind rejects the two kinds §17.1 names explicitly with a +// clear, specific error — distinct from "no match" and from "ambiguous" — so +// an operator who names a recording or datasource-managed rule learns why, +// not just that nothing matched. +func refuseUnsupportedKind(name string, d Definition) (Definition, error) { + switch d.Kind { + case KindDatasourceManaged: + return Definition{}, fmt.Errorf("%q resolves to %s, a datasource-managed rule, which is not supported", name, d.Title) + case KindRecording: + return Definition{}, fmt.Errorf("%q resolves to %s, a recording rule, which is not supported", name, d.Title) + default: + return d, nil + } +} + +// noMatchError reports a no-match with the count of rules the gate could see +// and, per Context decision 4, case-insensitive substring matches in place of +// the source plan's cut Levenshtein suggestions (§17.2). +func noMatchError(defs []Definition, name, wantTitle string) error { + msg := fmt.Sprintf("no rule matched %q (%d rules available; run 'grafana-alertcheck list' to see titles)", name, len(defs)) + + needle := strings.ToLower(wantTitle) + var subs []string + for _, d := range defs { + if strings.Contains(strings.ToLower(d.Title), needle) { + subs = append(subs, fmt.Sprintf("%s/%s/%s", d.Folder, d.Group, d.Title)) + } + } + if len(subs) > 0 { + sort.Strings(subs) + msg += fmt.Sprintf("; did you mean: %s", strings.Join(subs, ", ")) + } + return fmt.Errorf("%s", msg) +} + +// ambiguousError lists every candidate with its folder, its group, and the +// full copyable Folder/Group/Title (§17.1) — including the uid: form, which +// resolves unambiguously on the next attempt. +func ambiguousError(name string, candidates []Definition) error { + sorted := append([]Definition(nil), candidates...) + sort.Slice(sorted, func(i, j int) bool { return sorted[i].UID < sorted[j].UID }) + + var b strings.Builder + fmt.Fprintf(&b, "%q matches %d rules; use uid: or the full Folder/Group/Title:", name, len(sorted)) + for _, d := range sorted { + fmt.Fprintf(&b, "\n %s/%s/%s (uid:%s)", d.Folder, d.Group, d.Title, d.UID) + } + return fmt.Errorf("%s", b.String()) +} diff --git a/grafana-alertcheck/internal/gate/resolve_test.go b/grafana-alertcheck/internal/gate/resolve_test.go new file mode 100644 index 000000000..11c69d6fb --- /dev/null +++ b/grafana-alertcheck/internal/gate/resolve_test.go @@ -0,0 +1,273 @@ +package gate + +import ( + "fmt" + "strings" + "testing" +) + +func rulerDefs(t *testing.T) []Definition { + t.Helper() + defs, err := ParseDefinitions(readFixture(t, "ruler_rules.json")) + if err != nil { + t.Fatalf("ParseDefinitions: unexpected error: %v", err) + } + return defs +} + +func TestResolve_SingleMatch(t *testing.T) { + defs := rulerDefs(t) + resolved, notes, err := Resolve(defs, []string{"example_workflow_paused_rule"}, "") + if err != nil { + t.Fatalf("Resolve: unexpected error: %v", err) + } + if len(notes) != 0 { + t.Errorf("notes = %v, want none", notes) + } + if len(resolved) != 1 || resolved[0].UID != "rule0000007" { + t.Fatalf("resolved = %+v, want [rule0000007]", resolved) + } +} + +func TestResolve_UIDForm(t *testing.T) { + defs := rulerDefs(t) + resolved, _, err := Resolve(defs, []string{"uid:rule0000006a"}, "") + if err != nil { + t.Fatalf("Resolve: unexpected error: %v", err) + } + if len(resolved) != 1 || resolved[0].UID != "rule0000006a" { + t.Fatalf("resolved = %+v, want [rule0000006a]", resolved) + } +} + +func TestResolve_FolderGroupTitleForm(t *testing.T) { + defs := rulerDefs(t) + resolved, _, err := Resolve(defs, []string{"Example-Zone-A/Gateway/Example No Gateways Available"}, "") + if err == nil { + t.Fatalf("Resolve: want ambiguous error (real 2-way collision), got resolved=%+v", resolved) + } + if !strings.Contains(err.Error(), "matches 2 rules") { + t.Fatalf("Resolve: error = %q, want it to report 2 matches", err) + } + if !strings.Contains(err.Error(), "uid:rule0000006a") || !strings.Contains(err.Error(), "uid:rule0000006b") { + t.Fatalf("Resolve: error = %q, want both candidate uids listed", err) + } +} + +func TestResolve_TrueCollisionResolvesByUID(t *testing.T) { + defs := rulerDefs(t) + resolved, _, err := Resolve(defs, []string{"uid:rule0000006a", "uid:rule0000006b"}, "") + if err != nil { + t.Fatalf("Resolve: unexpected error: %v", err) + } + if len(resolved) != 2 { + t.Fatalf("resolved = %+v, want 2 distinct rules", resolved) + } +} + +func TestResolve_NoMatch(t *testing.T) { + defs := rulerDefs(t) + _, _, err := Resolve(defs, []string{"Does Not Exist"}, "") + if err == nil { + t.Fatal("Resolve: want error for unknown name") + } + if !strings.Contains(err.Error(), "no rule matched") || !strings.Contains(err.Error(), "list") { + t.Errorf("Resolve: error = %q, want it to name 'no rule matched' and point at 'list'", err) + } +} + +func TestResolve_NoMatchSubstringSuggestion(t *testing.T) { + defs := rulerDefs(t) + _, _, err := Resolve(defs, []string{"paused rule"}, "") + if err == nil { + t.Fatal("Resolve: want error for unknown name") + } + if !strings.Contains(err.Error(), "did you mean") || !strings.Contains(err.Error(), "Example Paused Rule") { + t.Errorf("Resolve: error = %q, want a case-insensitive substring suggestion", err) + } +} + +func TestResolve_RefusesDatasourceManaged(t *testing.T) { + defs, err := ParseDefinitions(readFixture(t, "ruler_datasource_managed.json")) + if err != nil { + t.Fatalf("ParseDefinitions: unexpected error: %v", err) + } + _, _, err = Resolve(defs, []string{"ExampleTargetDown"}, "") + if err == nil { + t.Fatal("Resolve: want refusal for a datasource-managed rule") + } + if !strings.Contains(err.Error(), "datasource-managed") { + t.Errorf("Resolve: error = %q, want it to name the datasource-managed kind", err) + } +} + +func TestResolve_RefusesRecording(t *testing.T) { + defs, err := ParseDefinitions(readFixture(t, "ruler_recording.json")) + if err != nil { + t.Fatalf("ParseDefinitions: unexpected error: %v", err) + } + _, _, err = Resolve(defs, []string{"uid:rule0000011"}, "") + if err == nil { + t.Fatal("Resolve: want refusal for a recording rule") + } + if !strings.Contains(err.Error(), "recording rule") { + t.Errorf("Resolve: error = %q, want it to name the recording kind", err) + } +} + +func TestResolve_RejectsEmptySegments(t *testing.T) { + defs := rulerDefs(t) + cases := []string{"/Title", "Folder/", "a//b"} + for _, name := range cases { + t.Run(name, func(t *testing.T) { + _, _, err := Resolve(defs, []string{name}, "") + if err == nil { + t.Fatalf("Resolve(%q): want error for an empty /-separated segment", name) + } + if !strings.Contains(err.Error(), "empty") { + t.Errorf("Resolve(%q): error = %q, want it to name the empty segment", name, err) + } + }) + } +} + +func TestResolve_UIDEmptySuffix(t *testing.T) { + // ruler_datasource_managed.json's only rule has UID == "" (P1.3: this + // shape has no uid at all). "uid:" with an empty suffix must not match it + // — that would report the misleading "datasource-managed rule, not + // supported" for what is really a typo'd/empty uid. + defs, err := ParseDefinitions(readFixture(t, "ruler_datasource_managed.json")) + if err != nil { + t.Fatalf("ParseDefinitions: unexpected error: %v", err) + } + _, _, err = Resolve(defs, []string{"uid:"}, "") + if err == nil { + t.Fatal("Resolve: want error for an empty uid: suffix") + } + if !strings.Contains(err.Error(), "no rule has this uid") { + t.Errorf("Resolve: error = %q, want it to say no rule has this uid", err) + } + if strings.Contains(err.Error(), "datasource-managed") { + t.Errorf("Resolve: error = %q, must not misreport this as a datasource-managed refusal", err) + } +} + +func TestResolve_UnsupportedKindsExcludedFromNoMatchSurfaces(t *testing.T) { + dsDefs, err := ParseDefinitions(readFixture(t, "ruler_datasource_managed.json")) + if err != nil { + t.Fatalf("ParseDefinitions(datasource_managed): unexpected error: %v", err) + } + recDefs, err := ParseDefinitions(readFixture(t, "ruler_recording.json")) + if err != nil { + t.Fatalf("ParseDefinitions(recording): unexpected error: %v", err) + } + supported := rulerDefs(t) + combined := append(append(append([]Definition{}, supported...), dsDefs...), recDefs...) + + _, _, err = Resolve(combined, []string{"Example"}, "") + if err == nil { + t.Fatal("Resolve: want a no-match error for a name matching no title exactly") + } + + wantCount := fmt.Sprintf("(%d rules available", len(supported)) + if !strings.Contains(err.Error(), wantCount) { + t.Errorf("Resolve: error = %q, want the available count scoped to the %d supported rules, not the %d combined", err, len(supported), len(combined)) + } + if strings.Contains(err.Error(), "ExampleTargetDown") { + t.Errorf("Resolve: error = %q, must not suggest the datasource-managed rule", err) + } + if strings.Contains(err.Error(), "example:recorded_metric:rate5m") { + t.Errorf("Resolve: error = %q, must not suggest the recording rule", err) + } + if !strings.Contains(err.Error(), "Example Paused Rule") { + t.Errorf("Resolve: error = %q, want it to still suggest a matching supported rule", err) + } +} + +func TestResolve_UnsupportedHomonymResolvesSupportedSilently(t *testing.T) { + // Synthetic: a supported and an unsupported rule sharing an identical + // Folder/Group/Title. Real Grafana data has no such case in the capture, + // but the policy must not treat this as ambiguous — the unsupported rule + // is invisible next to a same-named supported one. + defs := []Definition{ + {UID: "supported-1", Folder: "F", Group: "G", Title: "Shared Title", Kind: KindGrafanaManaged}, + {UID: "", Folder: "F", Group: "G", Title: "Shared Title", Kind: KindDatasourceManaged}, + } + resolved, _, err := Resolve(defs, []string{"F/G/Shared Title"}, "") + if err != nil { + t.Fatalf("Resolve: unexpected error: %v", err) + } + if len(resolved) != 1 || resolved[0].UID != "supported-1" { + t.Fatalf("resolved = %+v, want the supported rule alone, no ambiguity", resolved) + } +} + +func TestResolve_CollapseByUIDGivesNoteNotError(t *testing.T) { + defs := rulerDefs(t) + // The bare title and its Folder/Group/Title spelling both name the same + // rule (rule0000007) — a duplicate-name copy mistake, not an error + // (§17.3). + resolved, notes, err := Resolve(defs, []string{ + "example_workflow_paused_rule", + "ExampleObservability/Example Auth Production/example_workflow_paused_rule", + }, "") + if err != nil { + t.Fatalf("Resolve: unexpected error: %v", err) + } + if len(resolved) != 1 || resolved[0].UID != "rule0000007" { + t.Fatalf("resolved = %+v, want exactly one rule0000007", resolved) + } + if len(notes) != 1 { + t.Fatalf("notes = %v, want exactly one collapse note", notes) + } +} + +func TestResolve_MinObservedCountIsPostCollapse(t *testing.T) { + defs := rulerDefs(t) + names := []string{ + "example_workflow_paused_rule", + "ExampleObservability/Example Auth Production/example_workflow_paused_rule", // duplicate of the same rule + "Example Paused Rule", + } + resolved, notes, err := Resolve(defs, names, "") + if err != nil { + t.Fatalf("Resolve: unexpected error: %v", err) + } + // §17.3: the default MinObserved must come from len(resolved) (2 distinct + // rules) — never len(names) (3 input lines), which would be unsatisfiable. + if len(resolved) != 2 { + t.Fatalf("resolved = %+v, want 2 distinct rules after collapse", resolved) + } + if len(notes) != 1 { + t.Fatalf("notes = %v, want exactly one collapse note", notes) + } +} + +func TestResolve_EmptyAndBlankLinesDiscarded(t *testing.T) { + defs := rulerDefs(t) + resolved, _, err := Resolve(defs, []string{"", " ", "example_workflow_paused_rule", " \t "}, "") + if err != nil { + t.Fatalf("Resolve: unexpected error: %v", err) + } + if len(resolved) != 1 || resolved[0].UID != "rule0000007" { + t.Fatalf("resolved = %+v, want [rule0000007]", resolved) + } +} + +func TestResolve_FolderScopesBareTitle(t *testing.T) { + defs := rulerDefs(t) + // Bare title, scoped to the wrong folder — must not match. + _, _, err := Resolve(defs, []string{"example_workflow_paused_rule"}, "Example-Zone-A") + if err == nil { + t.Fatal("Resolve: want no-match when folder scope excludes the only candidate") + } + + // Scoped to the right folder — must match. + resolved, _, err := Resolve(defs, []string{"example_workflow_paused_rule"}, "ExampleObservability") + if err != nil { + t.Fatalf("Resolve: unexpected error: %v", err) + } + if len(resolved) != 1 || resolved[0].UID != "rule0000007" { + t.Fatalf("resolved = %+v, want [rule0000007]", resolved) + } +} From f674ee0a875b62c4b0c4e71ac68b9df55254295b Mon Sep 17 00:00:00 2001 From: Bartek Tofel Date: Tue, 1 Sep 2026 16:43:56 +0200 Subject: [PATCH 09/43] chore: enhance unit tests --- grafana-alertcheck/internal/gate/resolve_test.go | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/grafana-alertcheck/internal/gate/resolve_test.go b/grafana-alertcheck/internal/gate/resolve_test.go index 11c69d6fb..209b84bd5 100644 --- a/grafana-alertcheck/internal/gate/resolve_test.go +++ b/grafana-alertcheck/internal/gate/resolve_test.go @@ -117,7 +117,7 @@ func TestResolve_RefusesRecording(t *testing.T) { func TestResolve_RejectsEmptySegments(t *testing.T) { defs := rulerDefs(t) - cases := []string{"/Title", "Folder/", "a//b"} + cases := []string{"/Title", "Folder/", "a//b", "//", "/"} for _, name := range cases { t.Run(name, func(t *testing.T) { _, _, err := Resolve(defs, []string{name}, "") From 52011d01b5fed0bdf319a9e83eb241368504ab45 Mon Sep 17 00:00:00 2001 From: Bartek Tofel Date: Mon, 31 Aug 2026 14:25:20 +0200 Subject: [PATCH 10/43] chore: implement phase 4 MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Add per-rule poll timings, scheduler, and budget check (P4). - schedule.go: DeriveTimings, Scheduler, CheckBudget (§5) - Address review: add Folder/Title resolve test, rename CheckBudget's minPollEvery to tightestUID --- .../internal/gate/resolve_test.go | 11 + grafana-alertcheck/internal/gate/schedule.go | 293 ++++++++++++++++ .../internal/gate/schedule_test.go | 329 ++++++++++++++++++ grafana-alertcheck/internal/gate/source.go | 5 - 4 files changed, 633 insertions(+), 5 deletions(-) create mode 100644 grafana-alertcheck/internal/gate/schedule.go create mode 100644 grafana-alertcheck/internal/gate/schedule_test.go diff --git a/grafana-alertcheck/internal/gate/resolve_test.go b/grafana-alertcheck/internal/gate/resolve_test.go index 209b84bd5..77fd3bb5d 100644 --- a/grafana-alertcheck/internal/gate/resolve_test.go +++ b/grafana-alertcheck/internal/gate/resolve_test.go @@ -40,6 +40,17 @@ func TestResolve_UIDForm(t *testing.T) { } } +func TestResolve_FolderTitleForm(t *testing.T) { + defs := rulerDefs(t) + resolved, _, err := Resolve(defs, []string{"ExampleFeeds/TEMP - Example depeg alert"}, "") + if err != nil { + t.Fatalf("Resolve: unexpected error: %v", err) + } + if len(resolved) != 1 || resolved[0].UID != "rule0000008" { + t.Fatalf("resolved = %+v, want [rule0000008]", resolved) + } +} + func TestResolve_FolderGroupTitleForm(t *testing.T) { defs := rulerDefs(t) resolved, _, err := Resolve(defs, []string{"Example-Zone-A/Gateway/Example No Gateways Available"}, "") diff --git a/grafana-alertcheck/internal/gate/schedule.go b/grafana-alertcheck/internal/gate/schedule.go new file mode 100644 index 000000000..bf860ddef --- /dev/null +++ b/grafana-alertcheck/internal/gate/schedule.go @@ -0,0 +1,293 @@ +package gate + +import ( + "fmt" + "math/rand/v2" + "sort" + "strings" + "time" +) + +// skewHardLimit is one of §5's filled-in values (basis: §16; §22.11 asserts +// 120s errors, 30s does not). Defined here, in schedule.go's named-constants +// block, per §5's instruction — it moved out of source.go now that P4 exists; +// P2 needed it before this file did, so it started there. +const skewHardLimit = 60 * time.Second + +// minDrainTimeout is §5's floor on drainTimeout: max(2 x max(intervalSeconds), +// 2m). Without the floor, a fleet of very tight rules would derive a +// drainTimeout too short to let a healthy in-flight poll land. +const minDrainTimeout = 2 * time.Minute + +// graceWarnFraction is §13.2's threshold for warning that transitionGrace eats +// too much of the requested window: "approximately one quarter of the window". +const graceWarnFraction = 0.25 + +// ruleTimings groups the per-rule threshold values §5/§10.1/§14.1 derive from +// a rule's poll cadence and its own evaluation interval. +type ruleTimings struct { + pollEvery time.Duration + maxGap time.Duration + healthGrace time.Duration + evalStaleAfter time.Duration +} + +// globalTimings groups the values that apply to the whole run rather than to +// one rule: §13.1's transitionGrace and §19's drainTimeout are each derived +// once, across every non-skipped watched rule, not per rule. +type globalTimings struct { + transitionGrace time.Duration + // graceSource names, and already carries the `for` value of, the rule + // that set transitionGrace (§13.2 requires printing both) — one string + // field rather than a second (rule, duration) pair, matching this + // struct's fixed shape. "none" when no rule contributed (transitionGrace + // is then 0). + graceSource string + drainTimeout time.Duration +} + +// newRuleTimings derives one rule's thresholds from its fully-resolved poll +// cadence and its evaluation interval (§5, §10.1, §14.1). pollEvery arrives +// already resolved for the caller's mode — the §5 default, the operator's +// --poll-interval override, or (in log mode, a later phase) the cadence +// recorded in the log header. Deriving pollEvery inline here, instead of +// accepting it as an input, would let a caller in the wrong mode compute +// maxGap against the wrong authority — see the "Two authorities" note in P5. +func newRuleTimings(pollEvery time.Duration, intervalSeconds int) ruleTimings { + interval := time.Duration(intervalSeconds) * time.Second + maxGap := 2 * pollEvery + healthGrace := max(maxGap, interval) + return ruleTimings{ + pollEvery: pollEvery, + maxGap: maxGap, + healthGrace: healthGrace, + evalStaleAfter: 2 * interval, + } +} + +// defaultPollEvery is §5's default per-rule cadence: half the rule's own +// evaluation interval. +func defaultPollEvery(intervalSeconds int) time.Duration { + return time.Duration(intervalSeconds) * time.Second / 2 +} + +// DeriveTimings computes every resolved rule's ruleTimings, keyed by UID, +// plus the shared globalTimings, from resolved definitions and watch's +// optional --poll-interval override (0 = no override: use each rule's §5 +// default of half its own interval). Per §5.1, a supplied override is used +// verbatim for every rule and is never clamped down to the default even when +// it exceeds intervalSeconds/2 — that case is reported back as a note, not +// silently corrected or refused, because clamping would defeat the one knob +// §5.1 gives an operator for making a tight schedule fit. +func DeriveTimings(defs []Definition, override time.Duration) (rules map[string]ruleTimings, global globalTimings, notes []string) { + rules = make(map[string]ruleTimings, len(defs)) + for _, d := range defs { + def := defaultPollEvery(d.IntervalSeconds) + pollEvery := def + if override > 0 { + pollEvery = override + if override > def { + notes = append(notes, fmt.Sprintf( + "rule %s: --poll-interval %s exceeds half its %ds evaluation interval (%s); maxGap widens accordingly", + d.Title, override, d.IntervalSeconds, def)) + } + } + rules[d.UID] = newRuleTimings(pollEvery, d.IntervalSeconds) + } + return rules, deriveGlobalTimings(defs), notes +} + +// deriveGlobalTimings computes transitionGrace and drainTimeout over defs +// (§5, §13.1, §19). A rule paused before the window opened — skipped, §12 — +// is excluded from the transitionGrace max: its `for` value can never fire +// during the window, so counting it would only inflate the wait past what any +// watched rule actually needs (a judgment call the v2 plan makes explicitly +// for this formula; §19's drainTimeout carries no such exclusion, so it still +// runs over every resolved rule). +func deriveGlobalTimings(defs []Definition) globalTimings { + var g globalTimings + var maxInterval time.Duration + for _, d := range defs { + interval := time.Duration(d.IntervalSeconds) * time.Second + if interval > maxInterval { + maxInterval = interval + } + if d.IsPaused { + continue + } + if candidate := d.For + interval; candidate > g.transitionGrace { + g.transitionGrace = candidate + g.graceSource = fmt.Sprintf("%s (for=%s, interval=%s)", d.Title, d.For, interval) + } + } + g.drainTimeout = max(2*maxInterval, minDrainTimeout) + return g +} + +// Scheduler drives one per-rule schedule, never a global cycle (§5): a rule +// at intervalSeconds=10 alongside twenty at 300 keeps its own 5s cadence +// without forcing the same cadence onto the other twenty. +type Scheduler struct { + next map[string]time.Time + every map[string]time.Duration +} + +// NewScheduler builds a Scheduler over rules (keyed by UID), staggering each +// rule's initial next-due time across [0, pollEvery) so the fleet does not +// start phase-aligned (§5's burst-bound proof depends on this: an +// already-staggered fleet only re-aligns by chance, briefly, not by +// construction). +func NewScheduler(rules map[string]ruleTimings, now time.Time) *Scheduler { + s := &Scheduler{ + next: make(map[string]time.Time, len(rules)), + every: make(map[string]time.Duration, len(rules)), + } + for uid, rt := range rules { + s.every[uid] = rt.pollEvery + var offset time.Duration + if rt.pollEvery > 0 { + offset = rand.N(rt.pollEvery) + } + s.next[uid] = now.Add(offset) + } + return s +} + +// Due returns the UIDs whose next-due time has arrived, earliest-due-first. +// Ties (equal next-due time) break by tightest cadence first: the burst-bound +// proof in §5 assumes a newly-due tight rule waits at most for one in-flight +// request, which only holds if a simultaneous batch serves the tightest rule +// ahead of slacker ones. A tie-break that instead followed map iteration +// order would silently void that proof — nothing else would fail until a +// phase-aligned fleet opened a mid-run gap in production. +func (s *Scheduler) Due(now time.Time) []string { + var due []string + for uid, t := range s.next { + if !t.After(now) { + due = append(due, uid) + } + } + sort.Slice(due, func(i, j int) bool { + a, b := due[i], due[j] + if !s.next[a].Equal(s.next[b]) { + return s.next[a].Before(s.next[b]) + } + if s.every[a] != s.every[b] { + return s.every[a] < s.every[b] + } + return a < b // stable, deterministic fallback for an exact tie + }) + return due +} + +// Mark records that uid was just polled at now, scheduling its next poll one +// cadence later. +func (s *Scheduler) Mark(uid string, now time.Time) { + s.next[uid] = now.Add(s.every[uid]) +} + +// CheckBudget applies §5's error-at-start check to a fully resolved schedule. +// t and measured are both keyed by rule UID; measured must carry every UID in +// t; a rule this run never measured can't have its budget proved, and a +// silent zero-duration default would be exactly the kind of pass-on-an- +// unproven-window bug §5 exists to catch. CheckBudget fails when any of three +// conditions holds (sanity-checked against §22.3's mixed-interval regression +// in this phase's tests): +// +// - utilization: the long-run request rate exceeds what concurrency serves; +// - a single rule's own request cannot fit inside its own cadence; +// - the burst bound: the slowest measured request is slower than the +// fleet's tightest cadence, which — even under earliest-due-first +// ordering — can open a mid-run gap bigger than that rule's maxGap. +// +// The message never suggests a single interval (§5.1) — only the three +// controls an operator actually has: concurrency, poll-interval, and the +// alert list. +func CheckBudget(t map[string]ruleTimings, measured map[string]time.Duration, concurrency int) error { + if len(t) == 0 { + return nil + } + + uids := make([]string, 0, len(t)) + for uid := range t { + uids = append(uids, uid) + } + sort.Strings(uids) // deterministic message order + + for _, uid := range uids { + if _, ok := measured[uid]; !ok { + return fmt.Errorf("schedule budget: rule %s was never measured", uid) + } + } + + var utilization float64 + tightestUID := uids[0] // the rule with the smallest pollEvery seen so far — a UID, not a duration + var maxMeasuredUID string + var maxMeasured time.Duration + var overCadence []string + for _, uid := range uids { + rt, m := t[uid], measured[uid] + utilization += float64(m) / float64(rt.pollEvery) + if t[tightestUID].pollEvery > rt.pollEvery { + tightestUID = uid + } + if m > maxMeasured { + maxMeasured, maxMeasuredUID = m, uid + } + if m > rt.pollEvery { + overCadence = append(overCadence, uid) + } + } + + var problems []string + if utilization > float64(concurrency) { + problems = append(problems, fmt.Sprintf("utilization %.2f exceeds concurrency %d", utilization, concurrency)) + } + for _, uid := range overCadence { + problems = append(problems, fmt.Sprintf( + "rule %s: measured %s exceeds its own poll-interval %s", uid, measured[uid], t[uid].pollEvery)) + } + if maxMeasured > t[tightestUID].pollEvery { + problems = append(problems, fmt.Sprintf( + "burst bound: rule %s's measured %s exceeds the fleet's tightest poll-interval %s (rule %s)", + maxMeasuredUID, maxMeasured, t[tightestUID].pollEvery, tightestUID)) + } + + if len(problems) == 0 { + return nil + } + + var b strings.Builder + fmt.Fprintf(&b, "schedule does not fit at concurrency %d:\n", concurrency) + for _, uid := range uids { + fmt.Fprintf(&b, " rule %s: measured %s, poll-interval %s\n", uid, measured[uid], t[uid].pollEvery) + } + for _, p := range problems { + fmt.Fprintf(&b, " - %s\n", p) + } + b.WriteString("fix by: raising concurrency, raising poll-interval, or watching fewer alerts") + return fmt.Errorf("%s", b.String()) +} + +// StartupSummary formats §13.2's required pre-run print: the total planned +// run time and the rule (with its `for` value) that set transitionGrace, plus +// a warning when the grace eats more than graceWarnFraction of the requested +// window. from/to are the requested classification window. +func StartupSummary(from, to time.Time, global globalTimings) (summary, warning string) { + window := to.Sub(from) + total := window + global.transitionGrace + global.drainTimeout + source := global.graceSource + if source == "" { + source = "none" + } + summary = fmt.Sprintf( + "planned run time: %s (window %s + transitionGrace %s [source: %s] + drainTimeout %s)", + total, window, global.transitionGrace, source, global.drainTimeout) + + if window > 0 && float64(global.transitionGrace) > float64(window)*graceWarnFraction { + warning = fmt.Sprintf( + "transitionGrace %s is more than %.0f%% of the window %s (source: %s) — the window may be too short for this alert's `for`", + global.transitionGrace, graceWarnFraction*100, window, source) + } + return summary, warning +} diff --git a/grafana-alertcheck/internal/gate/schedule_test.go b/grafana-alertcheck/internal/gate/schedule_test.go new file mode 100644 index 000000000..02c77e8b1 --- /dev/null +++ b/grafana-alertcheck/internal/gate/schedule_test.go @@ -0,0 +1,329 @@ +package gate + +import ( + "strings" + "testing" + "time" +) + +func TestDeriveTimings_Default(t *testing.T) { + defs := []Definition{ + {UID: "r1", Title: "R1", IntervalSeconds: 60}, + } + rules, _, notes := DeriveTimings(defs, 0) + if len(notes) != 0 { + t.Fatalf("notes = %v, want none", notes) + } + rt := rules["r1"] + if rt.pollEvery != 30*time.Second { + t.Errorf("pollEvery = %s, want 30s", rt.pollEvery) + } + if rt.maxGap != 60*time.Second { + t.Errorf("maxGap = %s, want 60s", rt.maxGap) + } + if rt.healthGrace != 60*time.Second { + t.Errorf("healthGrace = %s, want 60s", rt.healthGrace) + } + if rt.evalStaleAfter != 120*time.Second { + t.Errorf("evalStaleAfter = %s, want 120s", rt.evalStaleAfter) + } +} + +func TestDeriveTimings_OverrideVerbatimNoClamp(t *testing.T) { + defs := []Definition{ + {UID: "r1", Title: "R1", IntervalSeconds: 10}, // default pollEvery = 5s + } + rules, _, notes := DeriveTimings(defs, 20*time.Second) + rt := rules["r1"] + if rt.pollEvery != 20*time.Second { + t.Fatalf("pollEvery = %s, want the override verbatim (20s), never clamped down to the 5s default", rt.pollEvery) + } + if rt.maxGap != 40*time.Second { + t.Errorf("maxGap = %s, want 2x the override (40s)", rt.maxGap) + } + if len(notes) != 1 || !strings.Contains(notes[0], "R1") { + t.Fatalf("notes = %v, want one note naming R1's exceeded default", notes) + } +} + +func TestDeriveTimings_OverrideBelowDefaultNoNote(t *testing.T) { + defs := []Definition{{UID: "r1", Title: "R1", IntervalSeconds: 60}} // default pollEvery = 30s + _, _, notes := DeriveTimings(defs, 5*time.Second) + if len(notes) != 0 { + t.Fatalf("notes = %v, want none when the override tightens rather than exceeds the default", notes) + } +} + +func TestDeriveTimings_TransitionGraceExcludesSkippedRule(t *testing.T) { + defs := []Definition{ + {UID: "r1", Title: "Tight", IntervalSeconds: 60, For: time.Minute}, + {UID: "r2", Title: "PausedLongFor", IntervalSeconds: 60, For: time.Hour, IsPaused: true}, + } + _, global, _ := DeriveTimings(defs, 0) + want := time.Minute + 60*time.Second // r1's for+interval; r2 (skipped) must not win despite its huge `for` + if global.transitionGrace != want { + t.Fatalf("transitionGrace = %s, want %s (paused rule r2 must be excluded from the max)", global.transitionGrace, want) + } + if !strings.Contains(global.graceSource, "Tight") { + t.Errorf("graceSource = %q, want it to name the contributing rule Tight", global.graceSource) + } +} + +func TestDeriveTimings_TransitionGraceZeroWhenAllSkipped(t *testing.T) { + defs := []Definition{{UID: "r1", Title: "R1", IntervalSeconds: 60, For: time.Hour, IsPaused: true}} + _, global, _ := DeriveTimings(defs, 0) + if global.transitionGrace != 0 { + t.Fatalf("transitionGrace = %s, want 0 when every rule is skipped", global.transitionGrace) + } +} + +func TestDeriveTimings_DrainTimeoutFloor(t *testing.T) { + // Tiny intervals: 2 x max(intervalSeconds) would be far under the 2m floor. + defs := []Definition{{UID: "r1", Title: "R1", IntervalSeconds: 10}} + _, global, _ := DeriveTimings(defs, 0) + if global.drainTimeout != minDrainTimeout { + t.Fatalf("drainTimeout = %s, want the %s floor", global.drainTimeout, minDrainTimeout) + } +} + +func TestDeriveTimings_DrainTimeoutAboveFloor(t *testing.T) { + defs := []Definition{{UID: "r1", Title: "R1", IntervalSeconds: 300}} // 2x300s = 600s > 2m floor + _, global, _ := DeriveTimings(defs, 0) + if global.drainTimeout != 600*time.Second { + t.Fatalf("drainTimeout = %s, want 600s", global.drainTimeout) + } +} + +// TestScheduler_DueOrderingTiesBreakByTightestCadence pins the ordering +// invariant the burst bound depends on (§5): when several rules become due at +// the exact same instant, Due must serve the tightest cadence first, not +// whatever order the underlying map happens to iterate in. A refactor that +// loses this ordering must fail here, not in a production phase-aligned gap. +func TestScheduler_DueOrderingTiesBreakByTightestCadence(t *testing.T) { + now := time.Date(2026, 1, 1, 0, 0, 0, 0, time.UTC) + s := &Scheduler{ + next: map[string]time.Time{ + "slack1": now, "slack2": now, "tight": now, "slack3": now, + }, + every: map[string]time.Duration{ + "slack1": 300 * time.Second, + "slack2": 300 * time.Second, + "tight": 10 * time.Second, + "slack3": 300 * time.Second, + }, + } + due := s.Due(now) + if len(due) != 4 || due[0] != "tight" { + t.Fatalf("Due = %v, want the tightest-cadence rule (tight) first when all are simultaneously due", due) + } +} + +func TestScheduler_DueExcludesNotYetDue(t *testing.T) { + now := time.Date(2026, 1, 1, 0, 0, 0, 0, time.UTC) + s := &Scheduler{ + next: map[string]time.Time{"soon": now.Add(-time.Second), "later": now.Add(time.Minute)}, + every: map[string]time.Duration{"soon": 10 * time.Second, "later": 10 * time.Second}, + } + due := s.Due(now) + if len(due) != 1 || due[0] != "soon" { + t.Fatalf("Due = %v, want only [soon]", due) + } +} + +func TestScheduler_MarkAdvancesNextDue(t *testing.T) { + now := time.Date(2026, 1, 1, 0, 0, 0, 0, time.UTC) + s := &Scheduler{ + next: map[string]time.Time{"r1": now}, + every: map[string]time.Duration{"r1": 30 * time.Second}, + } + s.Mark("r1", now) + if got := s.Due(now); len(got) != 0 { + t.Fatalf("Due right after Mark = %v, want none (next due is 30s out)", got) + } + if got := s.Due(now.Add(30 * time.Second)); len(got) != 1 { + t.Fatalf("Due at next-due time = %v, want [r1]", got) + } +} + +// TestScheduler_PerRuleCadenceOverTime simulates a run and counts how often +// each rule comes due, pinning §5's core claim: schedules are per rule, never +// a global cycle. A tight rule must be polled at its own cadence regardless +// of what slower rules in the same fleet need, and a slack rule must never be +// forced onto the tight rule's cadence. +func TestScheduler_PerRuleCadenceOverTime(t *testing.T) { + start := time.Date(2026, 1, 1, 0, 0, 0, 0, time.UTC) + rules := map[string]ruleTimings{ + "tight": {pollEvery: 10 * time.Second}, + "slack": {pollEvery: 300 * time.Second}, + } + s := NewScheduler(rules, start) + + const runFor = 900 * time.Second + const step = time.Second + counts := map[string]int{} + for elapsed := time.Duration(0); elapsed <= runFor; elapsed += step { + now := start.Add(elapsed) + for _, uid := range s.Due(now) { + counts[uid]++ + s.Mark(uid, now) + } + } + + // 900s of runtime: "tight" (10s cadence) polls ~90 times, "slack" (300s + // cadence) ~3 times. Assert the ratio holds rather than an exact count, + // since the staggered initial offset shifts each by up to one cadence. + if counts["tight"] < 85 || counts["tight"] > 91 { + t.Errorf("tight polled %d times over 900s, want ~90 (its own 10s cadence)", counts["tight"]) + } + if counts["slack"] < 2 || counts["slack"] > 4 { + t.Errorf("slack polled %d times over 900s, want ~3 (its own 300s cadence, not tight's)", counts["slack"]) + } + if counts["slack"] >= counts["tight"] { + t.Fatalf("slack polled as often as tight (%d vs %d) — schedules must be per rule, not a shared global cycle", counts["slack"], counts["tight"]) + } +} + +func TestNewScheduler_StaggersWithinPollEvery(t *testing.T) { + now := time.Date(2026, 1, 1, 0, 0, 0, 0, time.UTC) + rules := map[string]ruleTimings{"r1": {pollEvery: 100 * time.Second}} + s := NewScheduler(rules, now) + offset := s.next["r1"].Sub(now) + if offset < 0 || offset >= 100*time.Second { + t.Fatalf("initial offset = %s, want within [0, 100s)", offset) + } +} + +// TestCheckBudget_MixedIntervalRegression is §22.3's sanity check from the +// plan: one rule at 10s beside twenty at 300s, all measured ~1.8s, must not +// error at any reasonable concurrency — the exact case a naive worst-case-slot +// simulation would wrongly fail. +func TestCheckBudget_MixedIntervalRegression(t *testing.T) { + timings := map[string]ruleTimings{"tight": {pollEvery: 5 * time.Second}} + measured := map[string]time.Duration{"tight": 1800 * time.Millisecond} + for i := range 20 { + uid := uidN(i) + timings[uid] = ruleTimings{pollEvery: 150 * time.Second} + measured[uid] = 1800 * time.Millisecond + } + if err := CheckBudget(timings, measured, 1); err != nil { + t.Fatalf("CheckBudget = %v, want nil (utilization 0.6, burst bound 1.8s <= 5s)", err) + } +} + +func TestCheckBudget_UtilizationExceeded(t *testing.T) { + timings := map[string]ruleTimings{ + "a": {pollEvery: 10 * time.Second}, + "b": {pollEvery: 10 * time.Second}, + } + measured := map[string]time.Duration{"a": 9 * time.Second, "b": 9 * time.Second} + err := CheckBudget(timings, measured, 1) + if err == nil { + t.Fatal("CheckBudget = nil, want an error: utilization 1.8 > concurrency 1") + } + assertBudgetMessage(t, err.Error()) +} + +func TestCheckBudget_SingleRuleExceedsOwnCadence(t *testing.T) { + timings := map[string]ruleTimings{"slow": {pollEvery: 5 * time.Second}} + measured := map[string]time.Duration{"slow": 6 * time.Second} + err := CheckBudget(timings, measured, 10) + if err == nil { + t.Fatal("CheckBudget = nil, want an error: measured 6s exceeds its own 5s poll-interval") + } + assertBudgetMessage(t, err.Error()) +} + +func TestCheckBudget_BurstBoundViolation(t *testing.T) { + // Utilization is trivially fine, but the slower rule's request time (3s) + // exceeds the tighter rule's cadence (2s) — a mid-run gap risk even + // though no single rule breaches its own cadence and utilization is low. + timings := map[string]ruleTimings{ + "tight": {pollEvery: 2 * time.Second}, + "slow": {pollEvery: 100 * time.Second}, + } + measured := map[string]time.Duration{"tight": 100 * time.Millisecond, "slow": 3 * time.Second} + err := CheckBudget(timings, measured, 10) + if err == nil { + t.Fatal("CheckBudget = nil, want a burst-bound error: slow's 3s measured exceeds tight's 2s cadence") + } + if !strings.Contains(err.Error(), "burst bound") { + t.Errorf("error = %q, want it to name the burst bound", err.Error()) + } + assertBudgetMessage(t, err.Error()) +} + +func TestCheckBudget_BurstBoundOKWhenNotExceeded(t *testing.T) { + timings := map[string]ruleTimings{ + "tight": {pollEvery: 5 * time.Second}, + "slow": {pollEvery: 100 * time.Second}, + } + measured := map[string]time.Duration{"tight": 100 * time.Millisecond, "slow": 1800 * time.Millisecond} + if err := CheckBudget(timings, measured, 10); err != nil { + t.Fatalf("CheckBudget = %v, want nil (1.8s <= 5s tightest cadence)", err) + } +} + +func TestCheckBudget_MissingMeasurementIsAnError(t *testing.T) { + timings := map[string]ruleTimings{"r1": {pollEvery: 30 * time.Second}} + err := CheckBudget(timings, map[string]time.Duration{}, 10) + if err == nil { + t.Fatal("CheckBudget = nil, want an error: r1 was never measured (fail closed, not a silent zero)") + } +} + +func TestCheckBudget_EmptyScheduleIsFine(t *testing.T) { + if err := CheckBudget(nil, nil, 1); err != nil { + t.Fatalf("CheckBudget = %v, want nil for an empty schedule", err) + } +} + +// assertBudgetMessage checks §5.1's required message contents: a measured +// duration is present, and all three controls are named — never a single +// suggested interval. +func assertBudgetMessage(t *testing.T, msg string) { + t.Helper() + for _, want := range []string{"measured", "concurrency", "poll-interval", "fewer"} { + if !strings.Contains(msg, want) { + t.Errorf("message %q missing %q", msg, want) + } + } +} + +func uidN(i int) string { + return "slack" + string(rune('a'+i)) +} + +func TestStartupSummary_WarningWhenGraceTooLarge(t *testing.T) { + from := time.Date(2026, 1, 1, 0, 0, 0, 0, time.UTC) + to := from.Add(10 * time.Minute) + global := globalTimings{transitionGrace: 5 * time.Minute, graceSource: "R (for=4m30s, interval=30s)", drainTimeout: time.Minute} + summary, warning := StartupSummary(from, to, global) + if !strings.Contains(summary, "planned run time") { + t.Errorf("summary = %q, want it to name the planned run time", summary) + } + if warning == "" { + t.Fatal("warning = \"\", want one: transitionGrace (5m) > 1/4 of the 10m window") + } + if !strings.Contains(warning, "R (for=4m30s, interval=30s)") { + t.Errorf("warning = %q, want it to name the grace source", warning) + } +} + +func TestStartupSummary_NoWarningWhenGraceSmall(t *testing.T) { + from := time.Date(2026, 1, 1, 0, 0, 0, 0, time.UTC) + to := from.Add(time.Hour) + global := globalTimings{transitionGrace: time.Minute, graceSource: "R (for=30s, interval=30s)", drainTimeout: time.Minute} + _, warning := StartupSummary(from, to, global) + if warning != "" { + t.Fatalf("warning = %q, want none: 1m grace is well under 1/4 of a 1h window", warning) + } +} + +func TestStartupSummary_NoGraceSourceReadsNone(t *testing.T) { + from := time.Date(2026, 1, 1, 0, 0, 0, 0, time.UTC) + to := from.Add(time.Hour) + summary, _ := StartupSummary(from, to, globalTimings{}) + if !strings.Contains(summary, "none") { + t.Fatalf("summary = %q, want it to read \"none\" when no rule set the grace", summary) + } +} diff --git a/grafana-alertcheck/internal/gate/source.go b/grafana-alertcheck/internal/gate/source.go index 63cce7eb6..6e587b580 100644 --- a/grafana-alertcheck/internal/gate/source.go +++ b/grafana-alertcheck/internal/gate/source.go @@ -14,11 +14,6 @@ import ( "time" ) -// skewHardLimit is one of §5's filled-in values (basis: §16; §22.11 asserts -// 120s errors, 30s does not). It belongs in schedule.go's named-constants -// block once P4 exists; defined here because P2 needs it first. -const skewHardLimit = 60 * time.Second - // maxResponseBytes caps how much doRequest will read from a response body. // It is far above the largest real payload the gate retrieves (the ~600 KB // high-cardinality state fetch documented in parse_state_test.go), so a From 4d74f941d22f7d33af99c9ebfdb53e9a2d5b247b Mon Sep 17 00:00:00 2001 From: Bartek Tofel Date: Tue, 1 Sep 2026 17:20:35 +0200 Subject: [PATCH 11/43] chore: enhance unit tests --- .../internal/gate/schedule_test.go | 26 +++++++++++++++++-- 1 file changed, 24 insertions(+), 2 deletions(-) diff --git a/grafana-alertcheck/internal/gate/schedule_test.go b/grafana-alertcheck/internal/gate/schedule_test.go index 02c77e8b1..972300f77 100644 --- a/grafana-alertcheck/internal/gate/schedule_test.go +++ b/grafana-alertcheck/internal/gate/schedule_test.go @@ -77,8 +77,7 @@ func TestDeriveTimings_TransitionGraceZeroWhenAllSkipped(t *testing.T) { } } -func TestDeriveTimings_DrainTimeoutFloor(t *testing.T) { - // Tiny intervals: 2 x max(intervalSeconds) would be far under the 2m floor. +func TestDeriveTimings_DrainTimeoutIncludesPaused(t *testing.T) { defs := []Definition{{UID: "r1", Title: "R1", IntervalSeconds: 10}} _, global, _ := DeriveTimings(defs, 0) if global.drainTimeout != minDrainTimeout { @@ -86,6 +85,18 @@ func TestDeriveTimings_DrainTimeoutFloor(t *testing.T) { } } +func TestDeriveTimings_DrainTimeoutFloor(t *testing.T) { + defs := []Definition{ + {UID: "r1", Title: "Tight", IntervalSeconds: 60, For: time.Minute}, + {UID: "r2", Title: "PausedLongFor", IntervalSeconds: 180, For: time.Hour, IsPaused: true}, + } + _, global, _ := DeriveTimings(defs, 0) + // double the longest interval (2 * 180s) should be the drain timeout + if global.drainTimeout != 2*180*time.Second { + t.Fatalf("drainTimeout = %s, want %s", global.drainTimeout, 180*time.Second) + } +} + func TestDeriveTimings_DrainTimeoutAboveFloor(t *testing.T) { defs := []Definition{{UID: "r1", Title: "R1", IntervalSeconds: 300}} // 2x300s = 600s > 2m floor _, global, _ := DeriveTimings(defs, 0) @@ -271,6 +282,17 @@ func TestCheckBudget_MissingMeasurementIsAnError(t *testing.T) { } } +func TestCheckBudget_MissingMixedMeasurementIsAnError(t *testing.T) { + timings := map[string]ruleTimings{ + "tight": {pollEvery: 5 * time.Second}, + "slow": {pollEvery: 100 * time.Second}, + } + measured := map[string]time.Duration{"tight": 100 * time.Millisecond} + if err := CheckBudget(timings, measured, 10); err == nil { + t.Fatal("CheckBudget = nil, want an error: slow was never measured (fail closed, not a silent zero)") + } +} + func TestCheckBudget_EmptyScheduleIsFine(t *testing.T) { if err := CheckBudget(nil, nil, 1); err != nil { t.Fatalf("CheckBudget = %v, want nil for an empty schedule", err) From cc2dd22da02b00ec242c1afc68ac5e9e74d112d8 Mon Sep 17 00:00:00 2001 From: Bartek Tofel Date: Mon, 7 Sep 2026 16:08:59 +0200 Subject: [PATCH 12/43] chore: address code review comments --- grafana-alertcheck/internal/gate/schedule.go | 17 ++++++-- .../internal/gate/schedule_test.go | 39 +++++++++++++++++-- 2 files changed, 50 insertions(+), 6 deletions(-) diff --git a/grafana-alertcheck/internal/gate/schedule.go b/grafana-alertcheck/internal/gate/schedule.go index bf860ddef..023bf10ba 100644 --- a/grafana-alertcheck/internal/gate/schedule.go +++ b/grafana-alertcheck/internal/gate/schedule.go @@ -181,9 +181,17 @@ func (s *Scheduler) Due(now time.Time) []string { } // Mark records that uid was just polled at now, scheduling its next poll one -// cadence later. -func (s *Scheduler) Mark(uid string, now time.Time) { - s.next[uid] = now.Add(s.every[uid]) +// cadence later. It fails on an unknown uid rather than silently treating the +// missing cadence as zero: a zero cadence would schedule an immediate re-due +// (next = now.Add(0)) and insert a bogus next-due entry, hiding a caller that +// is polling a rule the scheduler never owns. +func (s *Scheduler) Mark(uid string, now time.Time) error { + every, ok := s.every[uid] + if !ok { + return fmt.Errorf("Mark: unknown rule uid %q", uid) + } + s.next[uid] = now.Add(every) + return nil } // CheckBudget applies §5's error-at-start check to a fully resolved schedule. @@ -218,6 +226,9 @@ func CheckBudget(t map[string]ruleTimings, measured map[string]time.Duration, co if _, ok := measured[uid]; !ok { return fmt.Errorf("schedule budget: rule %s was never measured", uid) } + if t[uid].pollEvery <= 0 { + return fmt.Errorf("schedule budget: rule %s has a non-positive poll-interval %s", uid, t[uid].pollEvery) + } } var utilization float64 diff --git a/grafana-alertcheck/internal/gate/schedule_test.go b/grafana-alertcheck/internal/gate/schedule_test.go index 972300f77..f0eb0351d 100644 --- a/grafana-alertcheck/internal/gate/schedule_test.go +++ b/grafana-alertcheck/internal/gate/schedule_test.go @@ -93,7 +93,7 @@ func TestDeriveTimings_DrainTimeoutFloor(t *testing.T) { _, global, _ := DeriveTimings(defs, 0) // double the longest interval (2 * 180s) should be the drain timeout if global.drainTimeout != 2*180*time.Second { - t.Fatalf("drainTimeout = %s, want %s", global.drainTimeout, 180*time.Second) + t.Fatalf("drainTimeout = %s, want %s", global.drainTimeout, 2*180*time.Second) } } @@ -147,7 +147,9 @@ func TestScheduler_MarkAdvancesNextDue(t *testing.T) { next: map[string]time.Time{"r1": now}, every: map[string]time.Duration{"r1": 30 * time.Second}, } - s.Mark("r1", now) + if err := s.Mark("r1", now); err != nil { + t.Fatalf("Mark: unexpected error: %v", err) + } if got := s.Due(now); len(got) != 0 { t.Fatalf("Due right after Mark = %v, want none (next due is 30s out)", got) } @@ -156,6 +158,21 @@ func TestScheduler_MarkAdvancesNextDue(t *testing.T) { } } +func TestScheduler_MarkUnknownUIDFails(t *testing.T) { + now := time.Date(2026, 1, 1, 0, 0, 0, 0, time.UTC) + s := &Scheduler{ + next: map[string]time.Time{"r1": now}, + every: map[string]time.Duration{"r1": 30 * time.Second}, + } + if err := s.Mark("not-a-rule", now); err == nil { + t.Fatalf("Mark of an unknown uid: want error, got nil (a missing cadence must not read as zero and loop)") + } + // The failed Mark must not have inserted a bogus next-due entry. + if _, ok := s.next["not-a-rule"]; ok { + t.Errorf("Mark of an unknown uid inserted a next-due entry") + } +} + // TestScheduler_PerRuleCadenceOverTime simulates a run and counts how often // each rule comes due, pinning §5's core claim: schedules are per rule, never // a global cycle. A tight rule must be polled at its own cadence regardless @@ -176,7 +193,9 @@ func TestScheduler_PerRuleCadenceOverTime(t *testing.T) { now := start.Add(elapsed) for _, uid := range s.Due(now) { counts[uid]++ - s.Mark(uid, now) + if err := s.Mark(uid, now); err != nil { + t.Fatalf("Mark(%q): unexpected error: %v", uid, err) + } } } @@ -299,6 +318,20 @@ func TestCheckBudget_EmptyScheduleIsFine(t *testing.T) { } } +func TestCheckBudget_NonPositivePollIntervalIsAnError(t *testing.T) { + for _, pe := range []time.Duration{0, -time.Second} { + timings := map[string]ruleTimings{"r1": {pollEvery: pe}} + measured := map[string]time.Duration{"r1": time.Second} + err := CheckBudget(timings, measured, 1) + if err == nil { + t.Fatalf("CheckBudget(pollEvery=%s) = nil, want error (non-positive poll-interval would divide by zero)", pe) + } + if !strings.Contains(err.Error(), "non-positive") { + t.Errorf("error %q does not name the non-positive poll-interval", err.Error()) + } + } +} + // assertBudgetMessage checks §5.1's required message contents: a measured // duration is present, and all three controls are named — never a single // suggested interval. From dd01b7d2734c5467d7e205f5d9fd50509c5bc439 Mon Sep 17 00:00:00 2001 From: Bartek Tofel Date: Mon, 31 Aug 2026 15:08:24 +0200 Subject: [PATCH 13/43] chore: implement phase 5 MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Add the JSONL evidence log (P5). - log.go: Header/Poll records, reduction, H2 transition markers, §3.2 verification, append-only Writer with flock, ReadLog - flock_unix.go: non-blocking exclusive lock, unix only - schedule.go: DeriveTimingsFromLog — log-mode cadence comes from the header, never from the definitions --- .../internal/gate/flock_unix.go | 24 + grafana-alertcheck/internal/gate/log.go | 521 ++++++++++ grafana-alertcheck/internal/gate/log_test.go | 894 ++++++++++++++++++ .../internal/gate/parse_state.go | 14 +- grafana-alertcheck/internal/gate/schedule.go | 61 ++ 5 files changed, 1509 insertions(+), 5 deletions(-) create mode 100644 grafana-alertcheck/internal/gate/flock_unix.go create mode 100644 grafana-alertcheck/internal/gate/log.go create mode 100644 grafana-alertcheck/internal/gate/log_test.go diff --git a/grafana-alertcheck/internal/gate/flock_unix.go b/grafana-alertcheck/internal/gate/flock_unix.go new file mode 100644 index 000000000..7b927f315 --- /dev/null +++ b/grafana-alertcheck/internal/gate/flock_unix.go @@ -0,0 +1,24 @@ +//go:build unix + +package gate + +import ( + "fmt" + "os" + "syscall" +) + +// lockExclusive takes a non-blocking exclusive lock on f. Non-blocking is the +// point (§8): a second writer must fail immediately with an error the operator +// sees, not queue behind the first and start appending to a log somebody else +// already finished. +// +// There is deliberately no Windows implementation — runners are Linux and +// goreleaser builds linux+darwin only (P6, P12) — so the package does not +// build there at all rather than silently skipping the lock. +func lockExclusive(f *os.File) error { + if err := syscall.Flock(int(f.Fd()), syscall.LOCK_EX|syscall.LOCK_NB); err != nil { + return fmt.Errorf("flock: %w", err) + } + return nil +} diff --git a/grafana-alertcheck/internal/gate/log.go b/grafana-alertcheck/internal/gate/log.go new file mode 100644 index 000000000..3f940f7d0 --- /dev/null +++ b/grafana-alertcheck/internal/gate/log.go @@ -0,0 +1,521 @@ +package gate + +import ( + "encoding/json" + "fmt" + "os" + "sort" + "strings" + "sync" + "time" +) + +// LogSchemaVersion is the version stamped into every log header. A log with +// any other value is a read error, never a best-effort read: the log is the +// gate's only evidence, and misreading a stale shape is a fail-open (§5). +const LogSchemaVersion = 1 + +// RecordType tags each JSONL line. There are exactly three, and a poll record +// IS the heartbeat — there is deliberately no separate heartbeat type (§4.6). +type RecordType string + +const ( + RecordHeader RecordType = "header" + RecordPoll RecordType = "poll" + RecordStopped RecordType = "stopped" +) + +// missingSeriesReason is the reason Grafana parks a disappearing series at +// ("Normal (MissingSeries)") for a couple of evaluations before deleting the +// instance. Reading that as a recovery is H2's named bug, so the markers below +// route it to Vanished (P1.2a). +const missingSeriesReason = "MissingSeries" + +// LoggedRule is the per-rule identity written into the header. Together with +// the header URL it IS the log's identity, which check validates (§19.1 step +// 3), and it supplies the alert set in check mode. +type LoggedRule struct { + UID string `json:"uid"` + Title string `json:"title"` + Folder string `json:"folder"` + Group string `json:"group"` + // ForSeconds, IntervalSeconds, IsPaused, NoDataState and ExecErrState are + // purely forensic: a resolve-time snapshot that makes the uploaded + // artifact self-describing to a human reading it after the runner is gone + // (§21.3). check never converts them back into a Definition — it always + // re-resolves definitions from the ruler API (§19.1 step 2). + ForSeconds float64 `json:"for_seconds"` + IntervalSeconds int `json:"interval_seconds"` + IsPaused bool `json:"is_paused"` + NoDataState string `json:"no_data_state"` + ExecErrState string `json:"exec_err_state"` + // PollEverySeconds is the cadence this recording ACTUALLY used, after any + // --poll-interval override. Load-bearing, not forensic: check derives + // maxGap from it and never re-derives it from the definitions. Getting + // that wrong is fail-open in the faster-override direction — a real + // recorder gap would pass silently (see "Two authorities", P5). + PollEverySeconds float64 `json:"poll_every_seconds"` +} + +// Header is the log's first line: what was recorded, from where, and when the +// recording started. It carries no States field — recording is deliberately +// unfiltered, so the same log can be re-classified under different --states +// without re-recording (P6). +type Header struct { + SchemaVersion int `json:"schema_version"` + URL string `json:"url"` // the log's identity (§19.1 step 3) + GrafanaVersion string `json:"grafana_version"` + StartedAt time.Time `json:"started_at"` // the record start (§7 validation) + Rules []LoggedRule `json:"rules"` // THE alert set (§19.1 step 3) +} + +// Poll is one reduced observation of one rule — the log's heartbeat and the +// only input the pure coverage and classification layers ever see. +type Poll struct { + RuleUID string `json:"rule_uid"` + GrafanaNow time.Time `json:"grafana_now"` // the Date header — H4 + // SkewMS, SkewBoundMS and LatencyMS are milliseconds for JSONL + // compactness ONLY. The pure layer never touches raw ms: it reads + // Skew(), SkewBound() and Latency() below, which convert at the + // (de)serialization boundary. + SkewMS int64 `json:"skew_ms"` + SkewBoundMS int64 `json:"skew_bound_ms"` + LatencyMS int64 `json:"latency_ms"` + // Found false means an authoritative 2xx in which this rule was absent + // (§14.5) — never a transport failure, which P2 retried and never turns + // into a Poll. P7 check 8 turns it into unobservable. + Found bool `json:"found"` + // State, Health and LastError are the raw rule-level strings, reporting + // only and never classified (P1.2a). + State string `json:"state,omitempty"` + Health string `json:"health,omitempty"` + LastError string `json:"last_error,omitempty"` + // omitzero, not omitempty: a not-found poll (and a paused rule, §2.3) has + // no evaluation time, and writing "0001-01-01T00:00:00Z" into an artifact + // humans and jq read (§21.3) invites reading it as a real timestamp. + LastEvaluation time.Time `json:"last_evaluation,omitzero"` + IsPaused bool `json:"is_paused"` + Histogram map[string]int `json:"histogram,omitempty"` // §4.9 — written, never analysed + // Reasons counts this poll's non-empty instance reasons, e.g. + // {"NoData":1091,"Error":14}; nil when none. Reporting-only, and the ONLY + // place composite states stay visible: they are canonical normal (so they + // are dropped from Abnormal) and `totals` never carries composite keys. + // + // The KEYS are raw reason strings and can be comma-joined composites + // ("KeepLast, MissingSeries") — newer Grafana versions join several + // reasons into one. So any consumer, P7 check 9's KeepLast note included, + // must test membership across the keys with reasonNames and must NEVER + // index a literal key: reasons["KeepLast"] misses every composite. + Reasons map[string]int `json:"reasons,omitempty"` + // Abnormal holds the instances whose CANONICAL state is not normal + // (§4.6). "Normal (NoData)" and "Normal (Error)" are canonical normal and + // are deliberately not retained here (P1.2a). + Abnormal []Instance `json:"abnormal,omitempty"` + // Cleared and Vanished are instance keys (§4.7): keys that left the + // abnormal set, resolved against the SAME response — a clear and a + // discontinuity are not the same fact (H2). + Cleared []string `json:"cleared,omitempty"` + Vanished []string `json:"vanished,omitempty"` +} + +// Skew is the signed clock skew of this poll (§16). +func (p Poll) Skew() time.Duration { return time.Duration(p.SkewMS) * time.Millisecond } + +// SkewBound is the uncertainty on Skew — the tolerance every cross-domain +// comparison in P7 applies alongside it. +func (p Poll) SkewBound() time.Duration { return time.Duration(p.SkewBoundMS) * time.Millisecond } + +// Latency is the wall time this poll's request took, feeding §5.2's budget check. +func (p Poll) Latency() time.Duration { return time.Duration(p.LatencyMS) * time.Millisecond } + +// Reducer turns each Observation into the single Poll record that goes into +// the log. It holds the previous poll's abnormal instance keys per rule, which +// is all the state the transition markers need (§4.7). +// +// A Reducer is safe for concurrent use: watch polls a fleet of rules +// concurrently (P6) and every one of those goroutines reduces through the same +// instance, because the per-rule marker state has to live in one place. The +// lock is per-Reducer rather than per-rule — Reduce only touches maps and +// slices, so it never blocks on I/O while holding it. +type Reducer struct { + mu sync.Mutex + prevAbnormal map[string]map[string]struct{} +} + +func NewReducer() *Reducer { + return &Reducer{prevAbnormal: make(map[string]map[string]struct{})} +} + +// Reduce selects the rule identified by uid out of obs and reduces it to a +// Poll. Selection is BY UID, never by title: a filtered response can carry +// several rules sharing one title (the known 2-way collision, §14.5), and +// picking the first would silently watch the wrong rule. +// +// The reduction (§4.6) keeps the rule-level fields, the raw totals histogram, +// the reason counts, and only the instances whose canonical state is not +// normal. That makes per-poll size independent of NORMAL cardinality — not of +// cardinality outright: a rule with 449 firing instances still stores all 449. +func (r *Reducer) Reduce(uid string, obs Observation) Poll { + r.mu.Lock() + defer r.mu.Unlock() + + p := Poll{ + RuleUID: uid, + GrafanaNow: obs.GrafanaNow, + SkewMS: obs.Skew.Milliseconds(), + SkewBoundMS: obs.SkewBound.Milliseconds(), + LatencyMS: obs.Latency.Milliseconds(), + } + + var rule *StateRule + for i := range obs.Rules { + if obs.Rules[i].UID == uid { + rule = &obs.Rules[i] + break + } + } + if rule == nil { + // An authoritative "the rule is absent". No markers are computed and + // the previous abnormal set is kept untouched: if the rule comes back + // with an instance missing, the next poll still reports that instance + // as vanished rather than losing the transition entirely. + return p + } + + p.Found = true + p.State = rule.State + p.Health = rule.Health + p.LastError = rule.LastError + p.LastEvaluation = rule.LastEvaluation + p.IsPaused = rule.IsPaused + p.Histogram = rule.Totals + + // present indexes every instance in THIS response, normal ones included — + // the markers below must resolve a departed key against the same response + // (H2), which is impossible from the abnormal subset alone. + present := make(map[string]Instance, len(rule.Instances)) + curAbnormal := make(map[string]struct{}) + for _, inst := range rule.Instances { + key := instanceKey(inst.Labels) + present[key] = inst + if inst.Reason != "" { + if p.Reasons == nil { + p.Reasons = make(map[string]int) + } + p.Reasons[inst.Reason]++ + } + if inst.State != StateNormal { + p.Abnormal = append(p.Abnormal, inst) + curAbnormal[key] = struct{}{} + } + } + + for key := range r.prevAbnormal[uid] { + if _, still := curAbnormal[key]; still { + continue + } + inst, found := present[key] + switch { + case !found: + // Fully absent from the response: a discontinuity, not a recovery. + p.Vanished = append(p.Vanished, key) + case reasonNames(inst.Reason, missingSeriesReason): + // The vanish in disguise, caught one poll earlier than the fully + // absent case — H2's named bug. + p.Vanished = append(p.Vanished, key) + default: + // Present as canonical normal without a MissingSeries reason. + p.Cleared = append(p.Cleared, key) + } + } + // Map iteration is unordered; sort so a log line is byte-stable for a + // given poll and a golden fixture stays meaningful. + sort.Strings(p.Cleared) + sort.Strings(p.Vanished) + + r.prevAbnormal[uid] = curAbnormal + return p +} + +// reasonNames reports whether reason names want. Newer Grafana versions +// comma-join several reasons into one string, so this tests membership rather +// than equality (P7 check 9 needs the same test for KeepLast). +func reasonNames(reason, want string) bool { + for part := range strings.SplitSeq(reason, ",") { + if strings.TrimSpace(part) == want { + return true + } + } + return false +} + +// VerifyNormalInstancesVisible checks §3.2's assumption on a first +// observation: that the state endpoint really does return normal instances, +// not only the abnormal ones. If it ever stops doing so, the reduction's +// "keep the non-normal instances" becomes "keep everything the API happened to +// send" and the transition markers lose their ground truth — a silent +// fail-open. So this is verified at start, never assumed. +// +// The counts are summed over every totals key whose LOWERCASED name is +// "normal" or "inactive". Never index one literal key: the captured +// vocabulary is mixed across rules ({"alerting":445,"normal":2004} on one, +// {"firing":2,"inactive":363} on another) and its case has already drifted +// from the original recon. Composite states never appear in totals — Grafana +// counts a "Normal (NoData)" instance under normal. +func VerifyNormalInstancesVisible(rules []StateRule) error { + for _, r := range rules { + var claimed int + for k, v := range r.Totals { + switch strings.ToLower(k) { + case "normal", "inactive": + claimed += v + } + } + if claimed == 0 { + continue + } + if hasNormalInstance(r.Instances) { + continue + } + return fmt.Errorf( + "rule %q (%s): totals claim %d normal instances but the response returned none — "+ + "the state endpoint no longer returns normal instances, which the §3.2 reduction depends on", + r.Title, r.UID, claimed) + } + return nil +} + +func hasNormalInstance(instances []Instance) bool { + for _, inst := range instances { + if inst.State == StateNormal { + return true + } + } + return false +} + +// headerRecord, pollRecord and stoppedRecord are the three wire shapes. The +// type tag is a real field on each line rather than an envelope, so a human +// (or jq) reading an uploaded log sees flat records. +type headerRecord struct { + Type RecordType `json:"type"` + Header +} + +type pollRecord struct { + Type RecordType `json:"type"` + Poll +} + +type stoppedRecord struct { + Type RecordType `json:"type"` + At time.Time `json:"at"` +} + +// Writer appends records to the JSONL log. It is append-only by construction +// (§8): O_APPEND|O_CREATE|O_WRONLY, never O_TRUNC, so no writer can ever +// destroy evidence a previous one recorded. An exclusive non-blocking flock +// makes a second writer fail immediately rather than interleave. +type Writer struct { + mu sync.Mutex + f *os.File + enc *json.Encoder + clock Clock + stopped bool +} + +// NewWriter opens path for appending and takes the exclusive lock. A second +// writer on the same path fails here, immediately — it never blocks and never +// waits, because two recorders on one log means one of them is recording a +// window nobody will classify. +func NewWriter(path string, clock Clock) (*Writer, error) { + f, err := os.OpenFile(path, os.O_APPEND|os.O_CREATE|os.O_WRONLY, 0o644) + if err != nil { + return nil, fmt.Errorf("open log %s: %w", path, err) + } + if err := lockExclusive(f); err != nil { + f.Close() + return nil, fmt.Errorf("lock log %s: %w (another writer holds it)", path, err) + } + return &Writer{f: f, enc: json.NewEncoder(f), clock: clock}, nil +} + +// WriteHeader writes line 1 and stamps the current schema version, so no +// caller can leave it at zero. It refuses a non-empty file: the log already +// has a header, and a second one would make ReadLog's "header is line 1" +// contract a lie. In the P6 handoff the parent writes the header and the child +// only appends polls. +func (w *Writer) WriteHeader(h Header) error { + w.mu.Lock() + defer w.mu.Unlock() + if w.stopped { + return fmt.Errorf("log writer already stopped") + } + info, err := w.f.Stat() + if err != nil { + return fmt.Errorf("stat log: %w", err) + } + if info.Size() != 0 { + return fmt.Errorf("log %s is not empty: it already has a header", w.f.Name()) + } + h.SchemaVersion = LogSchemaVersion + if err := w.enc.Encode(headerRecord{Type: RecordHeader, Header: h}); err != nil { + return fmt.Errorf("write log header: %w", err) + } + return nil +} + +// WritePoll appends one poll record — the heartbeat. +func (w *Writer) WritePoll(p Poll) error { + w.mu.Lock() + defer w.mu.Unlock() + if w.stopped { + return fmt.Errorf("log writer already stopped") + } + if err := w.enc.Encode(pollRecord{Type: RecordPoll, Poll: p}); err != nil { + return fmt.Errorf("write poll for rule %s: %w", p.RuleUID, err) + } + return nil +} + +// Stop finishes recording in the fixed §4.4 order, which must not be +// reordered: let the in-flight write finish (the mutex), append the stopped +// sentinel, fsync, then release. Any other order can leave a log whose last +// durable byte is a sentinel that was never actually preceded by the polls it +// vouches for. +// +// Stop writes the sentinel with the recorder's OWN stop time and makes no +// comparison against `to` — watch never knows `to` or the transition grace. +// check does that comparison, after this writer has exited (§4.5). +// +// Calling Stop twice is a no-op: watch reaches it from both a signal handler +// and a defer, and a second sentinel would be indistinguishable from a second +// writer. +func (w *Writer) Stop() error { + w.mu.Lock() + defer w.mu.Unlock() + if w.stopped { + return nil + } + w.stopped = true + + encErr := w.enc.Encode(stoppedRecord{Type: RecordStopped, At: w.clock.Now()}) + syncErr := w.f.Sync() + closeErr := w.f.Close() + + switch { + case encErr != nil: + return fmt.Errorf("write stopped sentinel: %w", encErr) + case syncErr != nil: + return fmt.Errorf("fsync log: %w", syncErr) + case closeErr != nil: + return fmt.Errorf("close log: %w", closeErr) + } + return nil +} + +// Close releases the file and the lock WITHOUT writing a sentinel. It exists +// for exactly one caller: watch's parent, which writes the header and then +// hands the log to the detached child that will finish it (P6). A sentinel +// here would tell check the recording ended before the child had even +// started. +func (w *Writer) Close() error { + w.mu.Lock() + defer w.mu.Unlock() + if w.stopped { + return nil + } + w.stopped = true + if err := w.f.Close(); err != nil { + return fmt.Errorf("close log: %w", err) + } + return nil +} + +// ReadLog reads the whole log once and returns its header, its polls in +// recorded order, and the sentinel time when one is present (nil when the +// recording never finished — check turns that into unobservable, never a +// pass). +// +// Call this only after the writer has exited (§4.4 step 4). Reading a log a +// writer can still append to can only produce a shorter window than the one +// that was recorded. +// +// The parse rules are deliberately the crudest possible (§24.2): the header +// must be line 1 with a matching schema version, and ANY unparseable line — +// including the last one, and including a last line that follows a sentinel — +// is an error, full stop. No heuristics, no discarding an untidy tail: a +// truncated log is evidence that something killed the recorder, which is +// exactly what must not pass. +func ReadLog(path string) (Header, []Poll, *time.Time, error) { + b, err := os.ReadFile(path) + if err != nil { + return Header{}, nil, nil, fmt.Errorf("read log %s: %w", path, err) + } + + lines := strings.Split(string(b), "\n") + // A complete record always ends with the encoder's newline, so the split + // leaves one trailing empty element. Drop exactly that one; any other + // empty line stays and fails below as unparseable. + if len(lines) > 0 && lines[len(lines)-1] == "" { + lines = lines[:len(lines)-1] + } + if len(lines) == 0 { + return Header{}, nil, nil, fmt.Errorf("log %s is empty: the header must be line 1", path) + } + + var ( + header Header + polls []Poll + sentinel *time.Time + ) + for i, line := range lines { + var probe struct { + Type RecordType `json:"type"` + } + if err := json.Unmarshal([]byte(line), &probe); err != nil { + return Header{}, nil, nil, fmt.Errorf("log %s line %d: unparseable record: %w", path, i+1, err) + } + if sentinel != nil { + return Header{}, nil, nil, fmt.Errorf( + "log %s line %d: a %q record follows the stopped sentinel — the log had a second writer", + path, i+1, probe.Type) + } + if (i == 0) != (probe.Type == RecordHeader) { + return Header{}, nil, nil, fmt.Errorf( + "log %s line %d: got record type %q; the header must be line 1 and appear only once", + path, i+1, probe.Type) + } + + switch probe.Type { + case RecordHeader: + var rec headerRecord + if err := json.Unmarshal([]byte(line), &rec); err != nil { + return Header{}, nil, nil, fmt.Errorf("log %s line 1: unparseable header: %w", path, err) + } + if rec.SchemaVersion != LogSchemaVersion { + return Header{}, nil, nil, fmt.Errorf( + "log %s: schema version %d is not %d — this log was written by a different version of the gate", + path, rec.SchemaVersion, LogSchemaVersion) + } + header = rec.Header + case RecordPoll: + var rec pollRecord + if err := json.Unmarshal([]byte(line), &rec); err != nil { + return Header{}, nil, nil, fmt.Errorf("log %s line %d: unparseable poll: %w", path, i+1, err) + } + polls = append(polls, rec.Poll) + case RecordStopped: + var rec stoppedRecord + if err := json.Unmarshal([]byte(line), &rec); err != nil { + return Header{}, nil, nil, fmt.Errorf("log %s line %d: unparseable sentinel: %w", path, i+1, err) + } + at := rec.At + sentinel = &at + default: + return Header{}, nil, nil, fmt.Errorf("log %s line %d: unknown record type %q", path, i+1, probe.Type) + } + } + + return header, polls, sentinel, nil +} diff --git a/grafana-alertcheck/internal/gate/log_test.go b/grafana-alertcheck/internal/gate/log_test.go new file mode 100644 index 000000000..494c9535c --- /dev/null +++ b/grafana-alertcheck/internal/gate/log_test.go @@ -0,0 +1,894 @@ +package gate + +import ( + "encoding/json" + "fmt" + "os" + "path/filepath" + "reflect" + "strings" + "sync" + "testing" + "time" +) + +// Every time literal in this file is UTC and built with time.Date, so it +// carries no monotonic reading and survives a JSON round trip byte-identical — +// which is what lets the round-trip tests below use reflect.DeepEqual on whole +// Poll values instead of comparing field by field. +var testNow = time.Date(2026, 8, 31, 9, 0, 0, 0, time.UTC) + +func testInstance(state State, reason, instanceLabel string) Instance { + return Instance{ + Labels: map[string]string{"alertname": "Example", "instance": instanceLabel}, + State: state, + Reason: reason, + ActiveAt: testNow, + } +} + +// observation wraps rules into an Observation with plausible timing numbers, +// deliberately not round millisecond values so a lost conversion at the ms +// boundary shows up as a wrong number rather than a coincidentally equal one. +func observation(grafanaNow time.Time, rules ...StateRule) Observation { + return Observation{ + Rules: rules, + GrafanaNow: grafanaNow, + Skew: 1500 * time.Millisecond, + SkewBound: 40 * time.Millisecond, + Latency: 1800 * time.Millisecond, + } +} + +func TestLogReduceKeepsOnlyAbnormalInstances(t *testing.T) { + rule := StateRule{ + UID: "rule1", Title: "Example", Folder: "F", Group: "G", + Interval: time.Minute, State: "firing", Health: "ok", + LastEvaluation: testNow, Totals: map[string]int{"alerting": 1, "normal": 2}, + Instances: []Instance{ + testInstance(StateNormal, "", "a"), + testInstance(StateFiring, "", "b"), + // Both composites are canonical normal (P1.2a): they must NOT be + // retained as abnormal, and their reasons must still be counted. + testInstance(StateNormal, "NoData", "c"), + testInstance(StateNormal, "Error", "d"), + }, + } + + p := NewReducer().Reduce("rule1", observation(testNow, rule)) + + if !p.Found { + t.Fatalf("Found = false, want true") + } + if len(p.Abnormal) != 1 || p.Abnormal[0].Labels["instance"] != "b" { + t.Errorf("Abnormal = %+v, want only the firing instance b", p.Abnormal) + } + if want := map[string]int{"NoData": 1, "Error": 1}; !reflect.DeepEqual(p.Reasons, want) { + t.Errorf("Reasons = %v, want %v", p.Reasons, want) + } + // The histogram is a verbatim copy of the response totals — raw keys, no + // normalization (§4.9). + if want := map[string]int{"alerting": 1, "normal": 2}; !reflect.DeepEqual(p.Histogram, want) { + t.Errorf("Histogram = %v, want %v", p.Histogram, want) + } + // Rule-level state and health stay raw and unnormalized (P1.2a). + if p.State != "firing" || p.Health != "ok" { + t.Errorf("State/Health = %q/%q, want firing/ok", p.State, p.Health) + } + if p.Skew() != 1500*time.Millisecond || p.SkewBound() != 40*time.Millisecond || p.Latency() != 1800*time.Millisecond { + t.Errorf("durations = %s/%s/%s, want 1.5s/40ms/1.8s", p.Skew(), p.SkewBound(), p.Latency()) + } + if p.Reasons["MissingSeries"] != 0 { + t.Errorf("unexpected MissingSeries count") + } +} + +// A filtered response can hold several rules sharing one title (the known +// 2-way collision, §14.5), so the reducer must select by UID. +func TestLogReduceSelectsRuleByUID(t *testing.T) { + first := StateRule{UID: "ruleA", Title: "Same Title", Health: "ok", State: "inactive", LastEvaluation: testNow} + second := StateRule{ + UID: "ruleB", Title: "Same Title", Health: "error", State: "firing", LastEvaluation: testNow, + Instances: []Instance{testInstance(StateFiring, "", "x")}, + } + + p := NewReducer().Reduce("ruleB", observation(testNow, first, second)) + + if p.Health != "error" || len(p.Abnormal) != 1 { + t.Errorf("reduced the wrong rule: %+v", p) + } +} + +func TestLogReduceRuleAbsentIsAuthoritative(t *testing.T) { + other := StateRule{UID: "other", Title: "Other", Health: "ok", State: "inactive", LastEvaluation: testNow} + + p := NewReducer().Reduce("rule1", observation(testNow, other)) + + if p.Found { + t.Errorf("Found = true, want false for a rule absent from an authoritative 2xx") + } + if p.RuleUID != "rule1" { + t.Errorf("RuleUID = %q, want rule1 — an absent rule is still attributed", p.RuleUID) + } + // The heartbeat still exists: a not-found poll is evidence that Grafana + // answered at this time, which the coverage proof reads. + if !p.GrafanaNow.Equal(testNow) || p.Latency() == 0 { + t.Errorf("absent-rule poll lost its timing evidence: %+v", p) + } + if p.Health != "" || p.Abnormal != nil { + t.Errorf("absent-rule poll carries rule fields: %+v", p) + } +} + +// H2: an instance that leaves the abnormal set is resolved against the SAME +// response, and MissingSeries is a vanish, never a recovery. +func TestTransitionMarkersClearedVersusVanished(t *testing.T) { + badKey := instanceKey(testInstance(StateFiring, "", "b").Labels) + + cases := []struct { + name string + second []Instance + wantCleared []string + wantVanished []string + }{ + { + name: "present as canonical normal is a clear", + second: []Instance{testInstance(StateNormal, "", "b")}, + wantCleared: []string{badKey}, + }, + { + name: "fully absent is a discontinuity", + second: nil, + wantVanished: []string{badKey}, + }, + { + name: "Normal (MissingSeries) is the vanish in disguise", + second: []Instance{testInstance(StateNormal, "MissingSeries", "b")}, + wantVanished: []string{badKey}, + }, + { + name: "a comma-joined reason naming MissingSeries still vanishes", + second: []Instance{testInstance(StateNormal, "KeepLast, MissingSeries", "b")}, + wantVanished: []string{badKey}, + }, + { + name: "an unrelated comma-joined reason still clears", + second: []Instance{testInstance(StateNormal, "KeepLast, Updated", "b")}, + wantCleared: []string{badKey}, + }, + { + name: "still abnormal is neither", + second: []Instance{testInstance(StateFiring, "", "b")}, + }, + { + name: "abnormal under a different state, then gone, still vanishes", + second: []Instance{testInstance(StateNormal, "", "unrelated")}, + wantVanished: []string{badKey}, + }, + } + + for _, c := range cases { + t.Run(c.name, func(t *testing.T) { + r := NewReducer() + firing := StateRule{ + UID: "rule1", Health: "ok", State: "firing", LastEvaluation: testNow, + Instances: []Instance{testInstance(StateFiring, "", "b")}, + } + first := r.Reduce("rule1", observation(testNow, firing)) + if first.Cleared != nil || first.Vanished != nil { + t.Fatalf("first poll produced markers with no previous poll: %+v", first) + } + + next := StateRule{UID: "rule1", Health: "ok", State: "inactive", LastEvaluation: testNow, Instances: c.second} + p := r.Reduce("rule1", observation(testNow.Add(30*time.Second), next)) + + if !reflect.DeepEqual(p.Cleared, c.wantCleared) { + t.Errorf("Cleared = %q, want %q", p.Cleared, c.wantCleared) + } + if !reflect.DeepEqual(p.Vanished, c.wantVanished) { + t.Errorf("Vanished = %q, want %q", p.Vanished, c.wantVanished) + } + }) + } +} + +// A departed key must be resolved against the response the reducer is holding, +// not against a later one — so a rule that goes absent and comes back with the +// instance missing still reports the vanish rather than losing it. +func TestTransitionMarkersSurviveAnAbsentPoll(t *testing.T) { + r := NewReducer() + firing := StateRule{ + UID: "rule1", Health: "ok", State: "firing", LastEvaluation: testNow, + Instances: []Instance{testInstance(StateFiring, "", "b")}, + } + r.Reduce("rule1", observation(testNow, firing)) + + absent := r.Reduce("rule1", observation(testNow.Add(30*time.Second))) + if absent.Vanished != nil || absent.Cleared != nil { + t.Fatalf("an absent rule produced markers: %+v", absent) + } + + back := StateRule{UID: "rule1", Health: "ok", State: "inactive", LastEvaluation: testNow} + p := r.Reduce("rule1", observation(testNow.Add(60*time.Second), back)) + if len(p.Vanished) != 1 { + t.Errorf("Vanished = %q, want the instance that disappeared across the absent poll", p.Vanished) + } +} + +func TestTransitionMarkersAreSortedAndPerRule(t *testing.T) { + r := NewReducer() + ruleOne := StateRule{ + UID: "rule1", Health: "ok", State: "firing", LastEvaluation: testNow, + Instances: []Instance{ + testInstance(StateFiring, "", "z"), + testInstance(StateFiring, "", "a"), + testInstance(StateFiring, "", "m"), + }, + } + ruleTwo := StateRule{ + UID: "rule2", Health: "ok", State: "firing", LastEvaluation: testNow, + Instances: []Instance{testInstance(StateFiring, "", "q")}, + } + r.Reduce("rule1", observation(testNow, ruleOne, ruleTwo)) + r.Reduce("rule2", observation(testNow, ruleOne, ruleTwo)) + + clearedOne := StateRule{UID: "rule1", Health: "ok", State: "inactive", LastEvaluation: testNow} + p := r.Reduce("rule1", observation(testNow.Add(time.Minute), clearedOne, ruleTwo)) + if len(p.Vanished) != 3 { + t.Fatalf("Vanished = %q, want 3 keys", p.Vanished) + } + for i := 1; i < len(p.Vanished); i++ { + if p.Vanished[i-1] > p.Vanished[i] { + t.Errorf("Vanished is not sorted: %q", p.Vanished) + } + } + // rule2's own abnormal set is untouched by rule1's transitions. + q := r.Reduce("rule2", observation(testNow.Add(time.Minute), clearedOne, ruleTwo)) + if q.Cleared != nil || q.Vanished != nil { + t.Errorf("rule2 picked up rule1's transitions: %+v", q) + } +} + +// §3.2: the reduction depends on the state endpoint returning normal instances. +// If it ever stops, that must fail loudly at start, never be assumed. +func TestLogVerifyNormalInstancesVisible(t *testing.T) { + cases := []struct { + fixture string + wantError bool + }{ + {"state_one_instance.json", false}, + {"state_reason_composite.json", false}, // composites plus one plain Normal + {"state_paused.json", false}, // no totals, no instances + {"state_missing_optional.json", false}, // no totals key at all + {"state_health_error.json", false}, // totals {"error":1} claims no normal + {"state_only_active_instances.json", true}, + } + + for _, c := range cases { + t.Run(c.fixture, func(t *testing.T) { + rules, err := ParseState(readFixture(t, c.fixture)) + if err != nil { + t.Fatalf("ParseState: %v", err) + } + err = VerifyNormalInstancesVisible(rules) + if c.wantError { + if err == nil { + t.Fatalf("VerifyNormalInstancesVisible: want an error, got nil") + } + if !strings.Contains(err.Error(), "§3.2") { + t.Errorf("error does not name §3.2: %v", err) + } + return + } + if err != nil { + t.Fatalf("VerifyNormalInstancesVisible: unexpected error: %v", err) + } + }) + } +} + +// The totals vocabulary is mixed and its case has drifted, so the check sums +// every key that lowercases to normal or inactive rather than indexing one +// literal key. +func TestLogVerifyNormalInstancesVisibleVocabularies(t *testing.T) { + cases := []struct { + name string + totals map[string]int + wantError bool + }{ + {"lowercase normal", map[string]int{"alerting": 1, "normal": 4}, true}, + {"capitalized Normal", map[string]int{"Alerting": 1, "Normal": 4}, true}, + {"rule vocabulary inactive", map[string]int{"firing": 2, "inactive": 363}, true}, + {"no normal claimed", map[string]int{"alerting": 1}, false}, + {"nil totals", nil, false}, + } + + for _, c := range cases { + t.Run(c.name, func(t *testing.T) { + rules := []StateRule{{ + UID: "rule1", Title: "Example", Totals: c.totals, + Instances: []Instance{testInstance(StateFiring, "", "b")}, + }} + err := VerifyNormalInstancesVisible(rules) + if (err != nil) != c.wantError { + t.Errorf("VerifyNormalInstancesVisible: error = %v, want error = %v", err, c.wantError) + } + }) + } +} + +// The two authorities: the header owns the recording facts (the cadence +// actually used), the ruler API owns the rule facts. Mixing them up is +// fail-open in the faster-override direction, so this pins both. +func TestLogModeCadenceComesFromTheHeader(t *testing.T) { + defs := []Definition{{UID: "rule1", Title: "Example", IntervalSeconds: 300, For: time.Minute}} + + h := testHeader() + h.Rules[0].IntervalSeconds = 300 + h.Rules[0].PollEverySeconds = 5 // an operator override far tighter than the default 150s + + rt, _, err := DeriveTimingsFromLog(h, defs) + if err != nil { + t.Fatalf("DeriveTimingsFromLog: %v", err) + } + got := rt["rule1"] + if got.pollEvery != 5*time.Second { + t.Errorf("pollEvery = %s, want the header's 5s, not the default 150s", got.pollEvery) + } + // maxGap and healthGrace follow the recorded cadence; without this a 250s + // hole in a log recorded at 5s would pass silently. + if got.maxGap != 10*time.Second { + t.Errorf("maxGap = %s, want 10s (2 x the recorded cadence)", got.maxGap) + } + if got.healthGrace != 300*time.Second { + t.Errorf("healthGrace = %s, want 300s (max(maxGap, interval))", got.healthGrace) + } + // evalStaleAfter is a rule fact, so it stays 2 x intervalSeconds from the + // definitions regardless of how often the gate polled. + if got.evalStaleAfter != 600*time.Second { + t.Errorf("evalStaleAfter = %s, want 600s from the definition's interval", got.evalStaleAfter) + } + + // A log that cannot say how often it was written cannot have its coverage + // proved, and neither can one naming a rule that no longer resolves. + missingCadence := testHeader() + missingCadence.Rules[0].PollEverySeconds = 0 + if _, _, err := DeriveTimingsFromLog(missingCadence, defs); err == nil { + t.Errorf("a header with no recorded cadence was accepted") + } + if _, _, err := DeriveTimingsFromLog(testHeader(), nil); err == nil { + t.Errorf("a header naming an unresolvable rule was accepted") + } + + // A duplicated UID must not resolve last-one-wins: the slower duplicate + // would widen maxGap, which is fail-open through log corruption alone. + duplicated := testHeader() + slower := duplicated.Rules[0] + slower.PollEverySeconds = 600 + duplicated.Rules = append(duplicated.Rules, slower) + if _, _, err := DeriveTimingsFromLog(duplicated, defs); err == nil { + t.Errorf("a header naming one rule twice was accepted") + } +} + +// watch polls a fleet concurrently through one Reducer (P6), so the marker +// state it holds per rule must be safe under -race — a latent data race here +// surfaces as a wrong transition, which is the one thing markers exist to get +// right. +func TestLogReduceIsSafeForConcurrentUse(t *testing.T) { + r := NewReducer() + rules := make([]StateRule, 0, 8) + for i := range 8 { + rules = append(rules, StateRule{ + UID: fmt.Sprintf("rule%d", i), Health: "ok", State: "firing", LastEvaluation: testNow, + Instances: []Instance{testInstance(StateFiring, "", "b")}, + }) + } + obs := observation(testNow, rules...) + + var wg sync.WaitGroup + for range 3 { + for _, rule := range rules { + wg.Go(func() { r.Reduce(rule.UID, obs) }) + } + } + wg.Wait() + + // Each rule's abnormal instance never left, so no round may invent a + // transition — the concurrency must not corrupt the per-rule state either. + for _, rule := range rules { + p := r.Reduce(rule.UID, obs) + if p.Cleared != nil || p.Vanished != nil { + t.Errorf("rule %s: markers after concurrent reduction: %+v", rule.UID, p) + } + } +} + +// A not-found poll has no evaluation time, and the artifact is read by humans +// and jq (§21.3) — the zero time must not appear as though it were real. +func TestLogPollOmitsTheZeroEvaluationTime(t *testing.T) { + absent := NewReducer().Reduce("rule1", observation(testNow)) + b, err := json.Marshal(pollRecord{Type: RecordPoll, Poll: absent}) + if err != nil { + t.Fatalf("marshal: %v", err) + } + if strings.Contains(string(b), "0001-01-01") { + t.Errorf("a not-found poll wrote the zero time: %s", b) + } + if strings.Contains(string(b), "last_evaluation") { + t.Errorf("a not-found poll wrote last_evaluation at all: %s", b) + } + + // A real evaluation time still round-trips. + found := StateRule{UID: "rule1", Health: "ok", State: "inactive", LastEvaluation: testNow} + p := NewReducer().Reduce("rule1", observation(testNow, found)) + b, err = json.Marshal(pollRecord{Type: RecordPoll, Poll: p}) + if err != nil { + t.Fatalf("marshal: %v", err) + } + var back pollRecord + if err := json.Unmarshal(b, &back); err != nil { + t.Fatalf("unmarshal: %v", err) + } + if !back.LastEvaluation.Equal(testNow) { + t.Errorf("last_evaluation = %s, want %s", back.LastEvaluation, testNow) + } +} + +func testHeader() Header { + return Header{ + URL: "https://grafana.example.com", + GrafanaVersion: "13.1.0", + StartedAt: testNow, + Rules: []LoggedRule{{ + UID: "rule1", Title: "Example", Folder: "F", Group: "G", + ForSeconds: 300, IntervalSeconds: 60, NoDataState: "OK", ExecErrState: "OK", + PollEverySeconds: 30, + }}, + } +} + +func newTestWriter(t *testing.T, path string) (*Writer, *fakeClock) { + t.Helper() + clock := newFakeClock(testNow) + w, err := NewWriter(path, clock) + if err != nil { + t.Fatalf("NewWriter: %v", err) + } + return w, clock +} + +func TestWriterReadLogRoundTrip(t *testing.T) { + path := filepath.Join(t.TempDir(), "log.jsonl") + w, clock := newTestWriter(t, path) + + h := testHeader() + if err := w.WriteHeader(h); err != nil { + t.Fatalf("WriteHeader: %v", err) + } + + r := NewReducer() + firing := StateRule{ + UID: "rule1", Health: "ok", State: "firing", LastError: "", LastEvaluation: testNow, + Totals: map[string]int{"alerting": 1}, + Instances: []Instance{testInstance(StateFiring, "", "b")}, + } + cleared := StateRule{UID: "rule1", Health: "ok", State: "inactive", LastEvaluation: testNow.Add(time.Minute)} + want := []Poll{ + r.Reduce("rule1", observation(testNow, firing)), + r.Reduce("rule1", observation(testNow.Add(time.Minute), cleared)), + } + for _, p := range want { + if err := w.WritePoll(p); err != nil { + t.Fatalf("WritePoll: %v", err) + } + } + + clock.Advance(2 * time.Minute) + if err := w.Stop(); err != nil { + t.Fatalf("Stop: %v", err) + } + + gotHeader, gotPolls, sentinel, err := ReadLog(path) + if err != nil { + t.Fatalf("ReadLog: %v", err) + } + h.SchemaVersion = LogSchemaVersion // WriteHeader stamps it + if !reflect.DeepEqual(gotHeader, h) { + t.Errorf("header round trip:\n got %+v\nwant %+v", gotHeader, h) + } + if !reflect.DeepEqual(gotPolls, want) { + t.Errorf("poll round trip:\n got %+v\nwant %+v", gotPolls, want) + } + if sentinel == nil { + t.Fatalf("sentinel is nil after Stop") + } + // Stop stamps the recorder's own stop time and makes no comparison + // against `to` — watch never knows it (§4.5). + if !sentinel.Equal(testNow.Add(2 * time.Minute)) { + t.Errorf("sentinel = %s, want the writer's stop time %s", sentinel, testNow.Add(2*time.Minute)) + } +} + +// §8: the log is append-only. A second run against the same path must never +// destroy the evidence the first one recorded. +func TestWriterAppendsAndNeverTruncates(t *testing.T) { + path := filepath.Join(t.TempDir(), "log.jsonl") + w, _ := newTestWriter(t, path) + if err := w.WriteHeader(testHeader()); err != nil { + t.Fatalf("WriteHeader: %v", err) + } + if err := w.WritePoll(Poll{RuleUID: "rule1", Found: true, GrafanaNow: testNow}); err != nil { + t.Fatalf("WritePoll: %v", err) + } + if err := w.Close(); err != nil { + t.Fatalf("Close: %v", err) + } + before, err := os.ReadFile(path) + if err != nil { + t.Fatalf("read: %v", err) + } + + // The P6 handoff: the parent wrote the header and closed; the child + // reopens the same path and appends without a second header. + child, _ := newTestWriter(t, path) + if err := child.WritePoll(Poll{RuleUID: "rule1", Found: true, GrafanaNow: testNow.Add(time.Minute)}); err != nil { + t.Fatalf("child WritePoll: %v", err) + } + if err := child.Stop(); err != nil { + t.Fatalf("child Stop: %v", err) + } + + after, err := os.ReadFile(path) + if err != nil { + t.Fatalf("read: %v", err) + } + if !strings.HasPrefix(string(after), string(before)) { + t.Fatalf("reopening the log rewrote earlier records:\n%s", after) + } + _, polls, sentinel, err := ReadLog(path) + if err != nil { + t.Fatalf("ReadLog: %v", err) + } + if len(polls) != 2 || sentinel == nil { + t.Errorf("got %d polls, sentinel %v; want 2 polls and a sentinel", len(polls), sentinel) + } +} + +// Two recorders on one log means one of them is recording a window nobody +// will classify, so the second writer fails immediately — it never blocks. +func TestWriterSecondWriterFails(t *testing.T) { + path := filepath.Join(t.TempDir(), "log.jsonl") + first, _ := newTestWriter(t, path) + defer first.Close() + + done := make(chan error, 1) + go func() { + _, err := NewWriter(path, newFakeClock(testNow)) + done <- err + }() + + select { + case err := <-done: + if err == nil { + t.Fatalf("a second writer took the lock") + } + if !strings.Contains(err.Error(), "another writer") { + t.Errorf("error does not name the conflict: %v", err) + } + case <-time.After(5 * time.Second): + t.Fatalf("the second NewWriter blocked instead of failing immediately") + } +} + +func TestWriterHeaderRefusesANonEmptyLog(t *testing.T) { + path := filepath.Join(t.TempDir(), "log.jsonl") + w, _ := newTestWriter(t, path) + if err := w.WriteHeader(testHeader()); err != nil { + t.Fatalf("WriteHeader: %v", err) + } + if err := w.WriteHeader(testHeader()); err == nil { + t.Fatalf("a second header was accepted") + } + if err := w.Close(); err != nil { + t.Fatalf("Close: %v", err) + } + + reopened, _ := newTestWriter(t, path) + defer reopened.Close() + if err := reopened.WriteHeader(testHeader()); err == nil { + t.Fatalf("a header was accepted on a non-empty log") + } +} + +func TestSentinelStopIsIdempotentAndLast(t *testing.T) { + path := filepath.Join(t.TempDir(), "log.jsonl") + w, _ := newTestWriter(t, path) + if err := w.WriteHeader(testHeader()); err != nil { + t.Fatalf("WriteHeader: %v", err) + } + if err := w.Stop(); err != nil { + t.Fatalf("Stop: %v", err) + } + // watch reaches Stop from both a signal handler and a defer; a second + // sentinel would be indistinguishable from a second writer. + if err := w.Stop(); err != nil { + t.Errorf("second Stop: %v", err) + } + // Nothing may be appended after the sentinel — not even by the same writer. + if err := w.WritePoll(Poll{RuleUID: "rule1"}); err == nil { + t.Errorf("WritePoll after Stop was accepted") + } + + b, err := os.ReadFile(path) + if err != nil { + t.Fatalf("read: %v", err) + } + lines := strings.Split(strings.TrimSuffix(string(b), "\n"), "\n") + if len(lines) != 2 { + t.Fatalf("got %d lines, want header + one sentinel:\n%s", len(lines), b) + } + if !strings.Contains(lines[1], `"type":"stopped"`) { + t.Errorf("last line is not the sentinel: %s", lines[1]) + } +} + +// Close is the parent's handoff path in P6: a sentinel there would tell check +// the recording ended before the child had even started. +func TestSentinelCloseWritesNone(t *testing.T) { + path := filepath.Join(t.TempDir(), "log.jsonl") + w, _ := newTestWriter(t, path) + if err := w.WriteHeader(testHeader()); err != nil { + t.Fatalf("WriteHeader: %v", err) + } + if err := w.Close(); err != nil { + t.Fatalf("Close: %v", err) + } + + _, polls, sentinel, err := ReadLog(path) + if err != nil { + t.Fatalf("ReadLog: %v", err) + } + if sentinel != nil { + t.Errorf("Close wrote a sentinel: %s", sentinel) + } + if polls != nil { + t.Errorf("polls = %+v, want none", polls) + } +} + +// An unfinished recording reads cleanly with a nil sentinel — ReadLog reports +// the absence and P7 turns it into unobservable. It is never ReadLog's job to +// call that a failure, and never anyone's job to call it a pass. +func TestReadLogWithoutASentinel(t *testing.T) { + path := filepath.Join(t.TempDir(), "log.jsonl") + w, _ := newTestWriter(t, path) + if err := w.WriteHeader(testHeader()); err != nil { + t.Fatalf("WriteHeader: %v", err) + } + if err := w.WritePoll(Poll{RuleUID: "rule1", Found: true, GrafanaNow: testNow}); err != nil { + t.Fatalf("WritePoll: %v", err) + } + if err := w.Close(); err != nil { + t.Fatalf("Close: %v", err) + } + + _, polls, sentinel, err := ReadLog(path) + if err != nil { + t.Fatalf("ReadLog: %v", err) + } + if sentinel != nil || len(polls) != 1 { + t.Errorf("got %d polls, sentinel %v; want 1 poll and no sentinel", len(polls), sentinel) + } +} + +// The read rules are deliberately the crudest possible (§24.2): any unparseable +// line is an error, full stop — including the last one, and including a last +// line that follows a sentinel. +func TestReadLogRejectsBadLogs(t *testing.T) { + header := func(version int) string { + h := testHeader() + h.SchemaVersion = version + b, err := json.Marshal(headerRecord{Type: RecordHeader, Header: h}) + if err != nil { + t.Fatalf("marshal header: %v", err) + } + return string(b) + } + poll := func() string { + b, err := json.Marshal(pollRecord{Type: RecordPoll, Poll: Poll{RuleUID: "rule1", Found: true, GrafanaNow: testNow}}) + if err != nil { + t.Fatalf("marshal poll: %v", err) + } + return string(b) + } + sentinel := func() string { + b, err := json.Marshal(stoppedRecord{Type: RecordStopped, At: testNow}) + if err != nil { + t.Fatalf("marshal sentinel: %v", err) + } + return string(b) + } + + cases := []struct { + name string + content string + wantIn string + }{ + {"empty file", "", "empty"}, + {"no header", poll() + "\n", "header must be line 1"}, + {"header not first", poll() + "\n" + header(LogSchemaVersion) + "\n", "header must be line 1"}, + {"second header", header(LogSchemaVersion) + "\n" + header(LogSchemaVersion) + "\n", "appear only once"}, + {"wrong schema version", header(2) + "\n" + poll() + "\n", "schema version"}, + { + name: "unparseable last line", + content: header(LogSchemaVersion) + "\n" + poll() + "\n" + `{"type":"poll","rule_ui`, + wantIn: "unparseable", + }, + { + // A preceding sentinel makes no difference: a truncated tail is + // evidence that something killed the recorder. + name: "unparseable line after the sentinel", + content: header(LogSchemaVersion) + "\n" + sentinel() + "\n" + `{"type":"pol`, + wantIn: "unparseable", + }, + { + name: "unparseable middle line", + content: header(LogSchemaVersion) + "\n" + `{"type":` + "\n" + poll() + "\n", + wantIn: "unparseable", + }, + { + name: "empty middle line", + content: header(LogSchemaVersion) + "\n\n" + poll() + "\n", + wantIn: "unparseable", + }, + { + name: "a record after the sentinel", + content: header(LogSchemaVersion) + "\n" + sentinel() + "\n" + poll() + "\n", + wantIn: "second writer", + }, + { + name: "unknown record type", + content: header(LogSchemaVersion) + "\n" + `{"type":"heartbeat"}` + "\n", + wantIn: "unknown record type", + }, + } + + for _, c := range cases { + t.Run(c.name, func(t *testing.T) { + path := filepath.Join(t.TempDir(), "log.jsonl") + if err := os.WriteFile(path, []byte(c.content), 0o600); err != nil { + t.Fatalf("write: %v", err) + } + _, _, _, err := ReadLog(path) + if err == nil { + t.Fatalf("ReadLog: want an error, got nil") + } + if !strings.Contains(err.Error(), c.wantIn) { + t.Errorf("error %q does not contain %q", err, c.wantIn) + } + }) + } +} + +func TestReadLogMissingFile(t *testing.T) { + _, _, _, err := ReadLog(filepath.Join(t.TempDir(), "absent.jsonl")) + if err == nil { + t.Fatalf("ReadLog on a missing log: want an error, got nil") + } +} + +// §22.3: per-poll log size must not grow across polls on a high-cardinality +// rule, and the one firing instance among 2446 must still be attributed by its +// labels. The reduction makes size independent of NORMAL cardinality — the +// firing instances are still stored, which is why a clear shrinks the record. +func TestLogSizeIsFlatAcrossPollsOnAHighCardinalityRule(t *testing.T) { + body := synthesizeHighCardinalityState(t, 1, 2445) + rules, err := ParseState(body) + if err != nil { + t.Fatalf("ParseState: %v", err) + } + if len(rules[0].Instances) != 2446 { + t.Fatalf("got %d instances, want 2446", len(rules[0].Instances)) + } + uid := rules[0].UID + + path := filepath.Join(t.TempDir(), "log.jsonl") + w, _ := newTestWriter(t, path) + if err := w.WriteHeader(testHeader()); err != nil { + t.Fatalf("WriteHeader: %v", err) + } + + r := NewReducer() + var sizes []int64 + // Measure from the end of the header line, so sizes[0] is the first poll + // record alone rather than the header plus it. + info, err := os.Stat(path) + if err != nil { + t.Fatalf("stat: %v", err) + } + previous := info.Size() + for i := range 5 { + p := r.Reduce(uid, observation(testNow.Add(time.Duration(i)*30*time.Second), rules[0])) + if len(p.Abnormal) != 1 { + t.Fatalf("poll %d: Abnormal = %d instances, want the single firing one", i, len(p.Abnormal)) + } + if got := p.Abnormal[0].Labels["instance"]; got != "alerting-0" { + t.Fatalf("poll %d: the firing instance lost its identity: %q", i, got) + } + if err := w.WritePoll(p); err != nil { + t.Fatalf("WritePoll: %v", err) + } + info, err := os.Stat(path) + if err != nil { + t.Fatalf("stat: %v", err) + } + sizes = append(sizes, info.Size()-previous) + previous = info.Size() + } + + for i := 1; i < len(sizes); i++ { + if sizes[i] != sizes[0] { + t.Errorf("per-poll size grew across polls: %v", sizes) + } + } + // One firing instance among 2446 costs a few hundred bytes, against the + // ~600 KB the unreduced response carries. + if sizes[0] > 2048 { + t.Errorf("per-poll size %d bytes is not a reduction of a %d-byte response", sizes[0], len(body)) + } + + // When the firing instance clears, the record collapses further and the + // transition is still attributed. + rules[0].Instances[0].State = StateNormal + p := r.Reduce(uid, observation(testNow.Add(5*30*time.Second), rules[0])) + if len(p.Cleared) != 1 || len(p.Abnormal) != 0 { + t.Errorf("cleared poll = %+v, want exactly one cleared key and no abnormal instances", p) + } + if err := w.Stop(); err != nil { + t.Fatalf("Stop: %v", err) + } +} + +// The log must stay readable by anything that reads JSONL, one flat object per +// line with its type tag — an uploaded artifact (§21.3) is read by humans and +// by jq, not only by ReadLog. +func TestLogRecordsAreFlatOneLineObjects(t *testing.T) { + path := filepath.Join(t.TempDir(), "log.jsonl") + w, _ := newTestWriter(t, path) + if err := w.WriteHeader(testHeader()); err != nil { + t.Fatalf("WriteHeader: %v", err) + } + if err := w.WritePoll(Poll{RuleUID: "rule1", Found: true, GrafanaNow: testNow}); err != nil { + t.Fatalf("WritePoll: %v", err) + } + if err := w.Stop(); err != nil { + t.Fatalf("Stop: %v", err) + } + + b, err := os.ReadFile(path) + if err != nil { + t.Fatalf("read: %v", err) + } + lines := strings.Split(strings.TrimSuffix(string(b), "\n"), "\n") + wantTypes := []RecordType{RecordHeader, RecordPoll, RecordStopped} + if len(lines) != len(wantTypes) { + t.Fatalf("got %d lines, want %d:\n%s", len(lines), len(wantTypes), b) + } + for i, line := range lines { + var m map[string]json.RawMessage + if err := json.Unmarshal([]byte(line), &m); err != nil { + t.Fatalf("line %d is not one JSON object: %v", i+1, err) + } + var gotType RecordType + if err := json.Unmarshal(m["type"], &gotType); err != nil { + t.Fatalf("line %d has no type tag: %v", i+1, err) + } + if gotType != wantTypes[i] { + t.Errorf("line %d type = %q, want %q", i+1, gotType, wantTypes[i]) + } + if _, nested := m["header"]; nested { + t.Errorf("line %d wraps its payload instead of being flat: %s", i+1, line) + } + } +} diff --git a/grafana-alertcheck/internal/gate/parse_state.go b/grafana-alertcheck/internal/gate/parse_state.go index d49dca0fa..7ffff1cfc 100644 --- a/grafana-alertcheck/internal/gate/parse_state.go +++ b/grafana-alertcheck/internal/gate/parse_state.go @@ -24,12 +24,16 @@ const ( // is the opaque suffix of a "State (Reason)" composite ("" when the API gave a // bare state). Reason is reporting-only except for the H2 MissingSeries routing // done downstream in the log markers. +// +// The json tags are for the JSONL log's abnormal-instance list (P5) only — +// parsing an API response never goes through them, because parseInstance +// decodes field by field through req/opt to keep H1's presence checks explicit. type Instance struct { - Labels map[string]string - State State - Reason string - ActiveAt time.Time - Value string + Labels map[string]string `json:"labels"` + State State `json:"state"` + Reason string `json:"reason,omitempty"` + ActiveAt time.Time `json:"active_at"` + Value string `json:"value,omitempty"` } // StateRule is one rule from the state endpoint diff --git a/grafana-alertcheck/internal/gate/schedule.go b/grafana-alertcheck/internal/gate/schedule.go index 023bf10ba..b473fd389 100644 --- a/grafana-alertcheck/internal/gate/schedule.go +++ b/grafana-alertcheck/internal/gate/schedule.go @@ -97,6 +97,67 @@ func DeriveTimings(defs []Definition, override time.Duration) (rules map[string] return rules, deriveGlobalTimings(defs), notes } +// DeriveTimingsFromLog is DeriveTimings' log-mode counterpart, and the two +// authorities of P5 are the whole reason it exists as a separate function. +// pollEvery comes from the header — the cadence the recording ACTUALLY used, +// after any --poll-interval override — and maxGap and healthGrace follow from +// it. Re-deriving pollEvery from defs here would compare gaps recorded at the +// override cadence against thresholds computed from the default: exit 2 on a +// clean window when the override was slower, and, worse, a real recorder gap +// passing silently when it was faster. +// +// evalStaleAfter still comes from defs (2 x intervalSeconds): it is a property +// of the rule's own evaluation cadence and is unaffected by how often the gate +// polled. +// +// Three shapes of header are errors rather than a best-effort derivation, +// because each one would otherwise widen a threshold silently: +// +// - a rule with no matching definition — a log that names a rule nobody can +// resolve cannot have that rule's coverage proved; +// - a non-positive recorded cadence — a log that cannot say how often it was +// written cannot have maxGap derived, and defaulting the cadence would +// prove a window that was never observed; +// - the same UID twice — last-one-wins would take whichever cadence happened +// to be written last, and a slower duplicate widens maxGap. That is a +// fail-open reachable through nothing but log corruption. +// +// It checks only the header-to-defs direction. The opposite direction — a +// resolved definition absent from the header — is NOT this function's to +// judge: it is §19.1 step 3's log-identity validation, and it belongs to P9's +// Check, which is the only caller that knows both sets and can name the +// mismatch. Without that check a definition simply gets no timings entry, and +// a downstream lookup would read a zero maxGap: fail-closed (every gap +// exceeds it) but silent, so P9 must reject the set mismatch by name rather +// than let a rule fail for an unexplained reason. +func DeriveTimingsFromLog(h Header, defs []Definition) (rules map[string]ruleTimings, global globalTimings, err error) { + byUID := make(map[string]Definition, len(defs)) + for _, d := range defs { + byUID[d.UID] = d + } + + rules = make(map[string]ruleTimings, len(h.Rules)) + for _, lr := range h.Rules { + def, ok := byUID[lr.UID] + if !ok { + return nil, globalTimings{}, fmt.Errorf( + "log header names rule %s (%q), which no current definition matches", lr.UID, lr.Title) + } + if _, duplicate := rules[lr.UID]; duplicate { + return nil, globalTimings{}, fmt.Errorf( + "log header names rule %s (%q) twice; its recorded cadence is ambiguous", lr.UID, lr.Title) + } + if lr.PollEverySeconds <= 0 { + return nil, globalTimings{}, fmt.Errorf( + "log header records poll_every_seconds=%v for rule %s (%q); the recorded cadence is required to derive maxGap", + lr.PollEverySeconds, lr.UID, lr.Title) + } + pollEvery := time.Duration(lr.PollEverySeconds * float64(time.Second)) + rules[lr.UID] = newRuleTimings(pollEvery, def.IntervalSeconds) + } + return rules, deriveGlobalTimings(defs), nil +} + // deriveGlobalTimings computes transitionGrace and drainTimeout over defs // (§5, §13.1, §19). A rule paused before the window opened — skipped, §12 — // is excluded from the transitionGrace max: its `for` value can never fire From bfce7794db64255486030ad99ab00fdb5b663857 Mon Sep 17 00:00:00 2001 From: Bartek Tofel Date: Fri, 4 Sep 2026 17:14:55 +0200 Subject: [PATCH 14/43] chore: remove unix build tags --- .../internal/gate/{flock_unix.go => flock.go} | 6 ------ 1 file changed, 6 deletions(-) rename grafana-alertcheck/internal/gate/{flock_unix.go => flock.go} (67%) diff --git a/grafana-alertcheck/internal/gate/flock_unix.go b/grafana-alertcheck/internal/gate/flock.go similarity index 67% rename from grafana-alertcheck/internal/gate/flock_unix.go rename to grafana-alertcheck/internal/gate/flock.go index 7b927f315..9d7f7e750 100644 --- a/grafana-alertcheck/internal/gate/flock_unix.go +++ b/grafana-alertcheck/internal/gate/flock.go @@ -1,5 +1,3 @@ -//go:build unix - package gate import ( @@ -12,10 +10,6 @@ import ( // point (§8): a second writer must fail immediately with an error the operator // sees, not queue behind the first and start appending to a log somebody else // already finished. -// -// There is deliberately no Windows implementation — runners are Linux and -// goreleaser builds linux+darwin only (P6, P12) — so the package does not -// build there at all rather than silently skipping the lock. func lockExclusive(f *os.File) error { if err := syscall.Flock(int(f.Fd()), syscall.LOCK_EX|syscall.LOCK_NB); err != nil { return fmt.Errorf("flock: %w", err) From 58ed7f962494b573a74753a03cdcab2e4537082a Mon Sep 17 00:00:00 2001 From: Bartek Tofel Date: Mon, 7 Sep 2026 16:16:54 +0200 Subject: [PATCH 15/43] chore: address code review comments --- grafana-alertcheck/internal/gate/flock.go | 7 ++++ .../internal/gate/flock_test.go | 38 +++++++++++++++++++ grafana-alertcheck/internal/gate/log.go | 5 ++- 3 files changed, 49 insertions(+), 1 deletion(-) create mode 100644 grafana-alertcheck/internal/gate/flock_test.go diff --git a/grafana-alertcheck/internal/gate/flock.go b/grafana-alertcheck/internal/gate/flock.go index 9d7f7e750..af4cf99fe 100644 --- a/grafana-alertcheck/internal/gate/flock.go +++ b/grafana-alertcheck/internal/gate/flock.go @@ -1,6 +1,7 @@ package gate import ( + "errors" "fmt" "os" "syscall" @@ -16,3 +17,9 @@ func lockExclusive(f *os.File) error { } return nil } + +// isLockContention reports whether a flock failure means another writer holds +// the lock, as opposed to an unrelated failure. +func isLockContention(err error) bool { + return errors.Is(err, syscall.EWOULDBLOCK) || errors.Is(err, syscall.EAGAIN) +} diff --git a/grafana-alertcheck/internal/gate/flock_test.go b/grafana-alertcheck/internal/gate/flock_test.go new file mode 100644 index 000000000..3e9ef268c --- /dev/null +++ b/grafana-alertcheck/internal/gate/flock_test.go @@ -0,0 +1,38 @@ +package gate + +import ( + "errors" + "fmt" + "syscall" + "testing" +) + +func TestIsLockContention(t *testing.T) { + contended := []error{syscall.EWOULDBLOCK, syscall.EAGAIN} + for _, e := range contended { + if !isLockContention(e) { + t.Errorf("isLockContention(%v) = false, want true", e) + } + // lockExclusive wraps the raw error via fmt.Errorf("flock: %w", ...). + if !isLockContention(fmt.Errorf("flock: %w", e)) { + t.Errorf("isLockContention(wrapped %v) = false, want true", e) + } + } + + notContended := []error{ + syscall.EROFS, + syscall.ENOTSUP, + syscall.ENOLCK, + syscall.EBADF, + syscall.EIO, + errors.New("something else"), + } + for _, e := range notContended { + if isLockContention(e) { + t.Errorf("isLockContention(%v) = true, want false (not a contender)", e) + } + if isLockContention(fmt.Errorf("flock: %w", e)) { + t.Errorf("isLockContention(wrapped %v) = true, want false", e) + } + } +} diff --git a/grafana-alertcheck/internal/gate/log.go b/grafana-alertcheck/internal/gate/log.go index 3f940f7d0..321330d74 100644 --- a/grafana-alertcheck/internal/gate/log.go +++ b/grafana-alertcheck/internal/gate/log.go @@ -335,7 +335,10 @@ func NewWriter(path string, clock Clock) (*Writer, error) { } if err := lockExclusive(f); err != nil { f.Close() - return nil, fmt.Errorf("lock log %s: %w (another writer holds it)", path, err) + if isLockContention(err) { + return nil, fmt.Errorf("lock log %s: %w (another writer holds it)", path, err) + } + return nil, fmt.Errorf("lock log %s: %w", path, err) } return &Writer{f: f, enc: json.NewEncoder(f), clock: clock}, nil } From 09902568016108de32247d0e9bb8879b7cdc379f Mon Sep 17 00:00:00 2001 From: Bartek Tofel Date: Mon, 31 Aug 2026 15:49:08 +0200 Subject: [PATCH 16/43] chore: implement phase 6 Invariant defended: H2. The one question: can watch return success over a window that nothing is recording? Watch() records the first observation of each non-skipped rule, then detaches a child that polls at the cadence in the header. The parent returns only after the child reports ready on an inherited pipe, and writes the pidfile after that. A clean stop writes the sentinel; a hard error does not. --- grafana-alertcheck/internal/gate/log.go | 28 + grafana-alertcheck/internal/gate/schedule.go | 44 +- .../internal/gate/schedule_test.go | 8 +- .../internal/gate/source_fake_test.go | 40 +- grafana-alertcheck/internal/gate/watch.go | 764 ++++++++++++++++++ .../internal/gate/watch_daemon_test.go | 308 +++++++ .../internal/gate/watch_test.go | 702 ++++++++++++++++ .../internal/gate/watch_unix.go | 112 +++ 8 files changed, 1988 insertions(+), 18 deletions(-) create mode 100644 grafana-alertcheck/internal/gate/watch.go create mode 100644 grafana-alertcheck/internal/gate/watch_daemon_test.go create mode 100644 grafana-alertcheck/internal/gate/watch_test.go create mode 100644 grafana-alertcheck/internal/gate/watch_unix.go diff --git a/grafana-alertcheck/internal/gate/log.go b/grafana-alertcheck/internal/gate/log.go index 321330d74..a66b508e1 100644 --- a/grafana-alertcheck/internal/gate/log.go +++ b/grafana-alertcheck/internal/gate/log.go @@ -237,6 +237,34 @@ func (r *Reducer) Reduce(uid string, obs Observation) Poll { return p } +// seedFrom restores the marker state above from polls that are already in the +// log, so the first poll a NEW Reducer produces compares against the last poll +// the previous one wrote instead of against an empty set. +// +// It exists for the one place a recording changes hands: watch's parent takes +// the first observation of every rule and its detached child continues from +// there (P6). Without the seed, an instance that is abnormal in the parent's +// observation and gone by the child's first poll produces no marker at all — +// it leaves the record as though it had never been bad, which is H2's +// fail-open reached through the handoff rather than through a reason string. +// +// Not-found polls are skipped, mirroring Reduce: an absent rule leaves the +// previous abnormal set untouched rather than emptying it. +func (r *Reducer) seedFrom(polls []Poll) { + r.mu.Lock() + defer r.mu.Unlock() + for _, p := range polls { + if !p.Found { + continue + } + keys := make(map[string]struct{}, len(p.Abnormal)) + for _, inst := range p.Abnormal { + keys[instanceKey(inst.Labels)] = struct{}{} + } + r.prevAbnormal[p.RuleUID] = keys + } +} + // reasonNames reports whether reason names want. Newer Grafana versions // comma-join several reasons into one string, so this tests membership rather // than equality (P7 check 9 needs the same test for KeepLast). diff --git a/grafana-alertcheck/internal/gate/schedule.go b/grafana-alertcheck/internal/gate/schedule.go index b473fd389..5712cda51 100644 --- a/grafana-alertcheck/internal/gate/schedule.go +++ b/grafana-alertcheck/internal/gate/schedule.go @@ -193,21 +193,27 @@ type Scheduler struct { every map[string]time.Duration } -// NewScheduler builds a Scheduler over rules (keyed by UID), staggering each -// rule's initial next-due time across [0, pollEvery) so the fleet does not -// start phase-aligned (§5's burst-bound proof depends on this: an -// already-staggered fleet only re-aligns by chance, briefly, not by +// NewScheduler builds a Scheduler over per-rule cadences (keyed by UID), +// staggering each rule's initial next-due time across [0, pollEvery) so the +// fleet does not start phase-aligned (§5's burst-bound proof depends on this: +// an already-staggered fleet only re-aligns by chance, briefly, not by // construction). -func NewScheduler(rules map[string]ruleTimings, now time.Time) *Scheduler { +// +// It takes cadences rather than whole ruleTimings on purpose: a scheduler +// decides when to poll and nothing else, so it must not be handed maxGap, +// healthGrace or evalStaleAfter. Those are coverage thresholds, they are +// applied by the pure layer at classification time, and the recorder that +// drives this scheduler never applies them at all. +func NewScheduler(every map[string]time.Duration, now time.Time) *Scheduler { s := &Scheduler{ - next: make(map[string]time.Time, len(rules)), - every: make(map[string]time.Duration, len(rules)), + next: make(map[string]time.Time, len(every)), + every: make(map[string]time.Duration, len(every)), } - for uid, rt := range rules { - s.every[uid] = rt.pollEvery + for uid, pollEvery := range every { + s.every[uid] = pollEvery var offset time.Duration - if rt.pollEvery > 0 { - offset = rand.N(rt.pollEvery) + if pollEvery > 0 { + offset = rand.N(pollEvery) } s.next[uid] = now.Add(offset) } @@ -255,6 +261,22 @@ func (s *Scheduler) Mark(uid string, now time.Time) error { return nil } +// earliestDue returns the earliest scheduled next-due time, and false when the +// scheduler holds no rules at all. The recorder's loop (P6) waits exactly that +// long instead of waking on a fixed tick: a fixed tick either polls a slack +// rule early — spending request budget the §5 formulas already accounted for — +// or wakes too late for the tightest rule and opens a gap inside its own +// maxGap. +func (s *Scheduler) earliestDue() (time.Time, bool) { + var earliest time.Time + for _, t := range s.next { + if earliest.IsZero() || t.Before(earliest) { + earliest = t + } + } + return earliest, !earliest.IsZero() +} + // CheckBudget applies §5's error-at-start check to a fully resolved schedule. // t and measured are both keyed by rule UID; measured must carry every UID in // t; a rule this run never measured can't have its budget proved, and a diff --git a/grafana-alertcheck/internal/gate/schedule_test.go b/grafana-alertcheck/internal/gate/schedule_test.go index f0eb0351d..07f0184ff 100644 --- a/grafana-alertcheck/internal/gate/schedule_test.go +++ b/grafana-alertcheck/internal/gate/schedule_test.go @@ -180,9 +180,9 @@ func TestScheduler_MarkUnknownUIDFails(t *testing.T) { // forced onto the tight rule's cadence. func TestScheduler_PerRuleCadenceOverTime(t *testing.T) { start := time.Date(2026, 1, 1, 0, 0, 0, 0, time.UTC) - rules := map[string]ruleTimings{ - "tight": {pollEvery: 10 * time.Second}, - "slack": {pollEvery: 300 * time.Second}, + rules := map[string]time.Duration{ + "tight": 10 * time.Second, + "slack": 300 * time.Second, } s := NewScheduler(rules, start) @@ -215,7 +215,7 @@ func TestScheduler_PerRuleCadenceOverTime(t *testing.T) { func TestNewScheduler_StaggersWithinPollEvery(t *testing.T) { now := time.Date(2026, 1, 1, 0, 0, 0, 0, time.UTC) - rules := map[string]ruleTimings{"r1": {pollEvery: 100 * time.Second}} + rules := map[string]time.Duration{"r1": 100 * time.Second} s := NewScheduler(rules, now) offset := s.next["r1"].Sub(now) if offset < 0 || offset >= 100*time.Second { diff --git a/grafana-alertcheck/internal/gate/source_fake_test.go b/grafana-alertcheck/internal/gate/source_fake_test.go index 197788a24..40a2a8a05 100644 --- a/grafana-alertcheck/internal/gate/source_fake_test.go +++ b/grafana-alertcheck/internal/gate/source_fake_test.go @@ -15,9 +15,9 @@ import ( // That is sufficient here: every retry/backoff test in this phase only needs // to avoid a real sleep. It is NOT sufficient for a test that must prove a // wait did not fire early — e.g. a P4 scheduler test asserting Due() doesn't -// return a rule before its next-due time. That needs a clock with a real -// waiter list keyed off Advance, which does not exist yet; build it when a -// phase actually needs it rather than guessing its shape now. +// return a rule before its next-due time. Use virtualClock below for that: it +// is the clock P6's recorder-loop tests needed, and it makes a wait and the +// passage of time the same event. type fakeClock struct { mu sync.Mutex now time.Time @@ -46,6 +46,40 @@ func (c *fakeClock) After(d time.Duration) <-chan time.Time { return ch } +// virtualClock is a Clock in which time moves only when something waits for +// it: After(d) jumps Now() forward by d and fires at once. That makes a +// recorder-loop test both instant and exact — a loop that waits for its next +// scheduled poll gets that poll's time, never an early or a late wake — and it +// terminates, which a clock whose After fires without advancing Now does not +// (the loop would spin forever on a rule that never comes due). +// +// It is goroutine-safe, but a test that advances time from two goroutines gets +// what it deserves: use it from the loop under test only. +type virtualClock struct { + mu sync.Mutex + now time.Time +} + +func newVirtualClock(now time.Time) *virtualClock { return &virtualClock{now: now} } + +func (c *virtualClock) Now() time.Time { + c.mu.Lock() + defer c.mu.Unlock() + return c.now +} + +func (c *virtualClock) After(d time.Duration) <-chan time.Time { + c.mu.Lock() + if d > 0 { + c.now = c.now.Add(d) + } + fireAt := c.now + c.mu.Unlock() + ch := make(chan time.Time, 1) + ch <- fireAt + return ch +} + // steppingClock advances by a fixed step on every Now() call, so a test can // assert exact latency/skew-bound arithmetic (doRequest's three clock reads // per attempt) without depending on real wall-clock timing. diff --git a/grafana-alertcheck/internal/gate/watch.go b/grafana-alertcheck/internal/gate/watch.go new file mode 100644 index 000000000..61e1cce63 --- /dev/null +++ b/grafana-alertcheck/internal/gate/watch.go @@ -0,0 +1,764 @@ +package gate + +import ( + "context" + "fmt" + "io" + "os" + "os/signal" + "strconv" + "strings" + "sync" + "syscall" + "time" +) + +// DaemonChildFlag is the hidden flag the parent passes when it re-execs itself +// as the detached recorder (§4.4). It is deliberately absent from the CLI's +// usage text: an operator never types it, and a child started by hand against +// a log no parent prepared fails immediately on the header read. +const DaemonChildFlag = "--daemon-child" + +// ReadyFDFlag names the inherited descriptor the child reports readiness on. +// The parent passes the write end of a pipe as descriptor 3 and waits for one +// byte, so "the recorder is running" is a POSITIVE signal from the child +// itself — it has read the header, taken the log's flock and entered its poll +// loop — and not an assumption drawn from surviving a timer. A timer cannot +// tell a healthy child from one that is about to die on a slow runner, and +// getting that wrong means watch returns success over a recording that never +// happened (§4.3). +const ReadyFDFlag = "--ready-fd" + +// childReadyTimeout bounds that wait. Everything before the signal is local — +// fork, exec, one read of a log holding a header and a handful of polls — so +// the real figure is milliseconds; this is loose enough for a badly overloaded +// runner and still fails closed rather than hanging the pipeline. +const childReadyTimeout = 30 * time.Second + +// daemonLogTailBytes bounds how much of a dead child's output the parent +// quotes back. A child dies in its first few lines or not at all. +const daemonLogTailBytes = 4096 + +// WatchConfig is the record step's whole input. +// +// It has no To field and must never gain one: watch writes the stopped +// sentinel with its OWN stop time and makes no comparison against `to`, which +// only check knows (§4.5). Passing `to` here would give two components an +// opinion about the same comparison, and the recorder's opinion is the one +// that cannot be trusted — it exits before the grace it would have to wait for. +// +// It has no States field either, and watch has no --states flag: recording is +// deliberately unfiltered. The reduction keeps every non-normal instance and +// the transition markers key off the same predicate, so neither consults +// States. The payoff is real — because the log is raw evidence, one recording +// can be re-classified under different --states without re-recording — and the +// Header carries no States field for the same reason. +type WatchConfig struct { + // URL and Token are the connection details. The CLI reads both from the + // environment and never from a flag (§20.2); Token is never logged and + // never enters an error string. + URL, Token string + + // Alerts are the operator-supplied names, one per line, in any of §17's + // forms. Empty lines are discarded by Resolve. + Alerts []string + Folder string + + // Out is the JSONL log path. PidFile and DaemonLog default to + // .pid and .daemon.log — the same convention check uses to find + // the recorder it must stop (P9), so nothing has to be wired by hand. + Out string + PidFile string + DaemonLog string + + // Until is an optional hard stop for the child. Zero means "record until + // signalled", which is the normal case: check sends SIGTERM when its + // collection loop ends. + Until time.Time + + // PollEvery is the --poll-interval override, used verbatim for every rule + // and never clamped (§5.1). Zero means each rule polls at half its own + // evaluation interval. Whatever this resolves to is written into the header + // as the cadence actually used, and that header value — never a + // re-derivation from the definitions — is what check derives maxGap from + // (P5, "two authorities"). + PollEvery time.Duration + + Concurrency int + Clock Clock + + // Notes is where the parent prints what an operator has to see before the + // deploy step runs: resolve notes, the cadence per rule, the rules it will + // not wait for. nil discards them. The library prints nothing else — the + // CLI owns presentation (§20.2). + Notes io.Writer +} + +func (cfg WatchConfig) withDefaults() WatchConfig { + if cfg.Clock == nil { + cfg.Clock = SystemClock{} + } + if cfg.Notes == nil { + cfg.Notes = io.Discard + } + if cfg.Concurrency < 1 { + cfg.Concurrency = 1 + } + if cfg.PidFile == "" && cfg.Out != "" { + cfg.PidFile = cfg.Out + ".pid" + } + if cfg.DaemonLog == "" && cfg.Out != "" { + cfg.DaemonLog = cfg.Out + ".daemon.log" + } + return cfg +} + +func (cfg WatchConfig) validate() error { + if cfg.URL == "" { + return fmt.Errorf("watch: no grafana url (it is the log's identity, which check validates)") + } + if cfg.Out == "" { + return fmt.Errorf("watch: no log path") + } + named := 0 + for _, a := range cfg.Alerts { + if strings.TrimSpace(a) != "" { + named++ + } + } + if named == 0 { + return fmt.Errorf("watch: no alert names given; there is nothing to record") + } + // An --until already in the past would make the child stop before it ever + // polled, and the parent would then report a child that never reported + // ready — a true statement about a config mistake, but a confusing one. + if !cfg.Until.IsZero() && !cfg.Until.After(cfg.Clock.Now()) { + return fmt.Errorf("watch: --until %s is not in the future", cfg.Until.Format(time.RFC3339)) + } + return nil +} + +// Watch is the record step's parent process (§4.3). It returns only once the +// window is genuinely being recorded: +// +// version gate -> resolve definitions and names -> derive timings -> +// open the log and write the header -> ONE observation of every non-skipped +// rule -> verify §3.2 -> check the schedule budget -> detach the child -> +// wait for the child to report that it is recording -> write the pidfile -> +// return. +// +// The first-observation wait is not a convenience. Returning before it would +// leave the deploy inside [from, first_poll] with no evidence — the exact +// blind interval the two-phase model exists to remove — and it is also what +// surfaces auth, name-resolution and parse failures BEFORE deploy.sh runs +// rather than ten minutes later. +func Watch(ctx context.Context, cfg WatchConfig) error { + cfg = cfg.withDefaults() + if err := cfg.validate(); err != nil { + return err + } + + src := NewHTTPSource(cfg.URL, cfg.Token, cfg.Clock) + prep, err := prepareWatch(ctx, cfg, src) + if err != nil { + return err + } + + // Hand the log over with Close, never Stop: a sentinel here would tell + // check the recording ended before the child had even started (§4.5). + // Closing also releases the flock the child is about to take. + if err := prep.writer.Close(); err != nil { + return err + } + + child, err := spawnChild(cfg) + if err != nil { + return fmt.Errorf("detach recorder: %w", err) + } + if err := waitForChildReady(cfg, child); err != nil { + return err + } + + // The PARENT writes the pidfile, not the child: check must find the pid the + // instant Watch returns, and a child writing its own would race the very + // next step of the pipeline. A deviation from P6's argv list, and the + // reason the child is never given --pidfile at all. + // + // It is written only once the child has reported ready, so no path through + // this function leaves a pidfile naming a process that is not recording. + // Pids are reused: a stale pidfile is a live process somewhere else, and a + // pipeline that ignored this function's error would SIGTERM it. + if err := writePidFile(cfg.PidFile, child.cmd.Process.Pid); err != nil { + _ = child.cmd.Process.Kill() + _ = os.Remove(cfg.PidFile) + return err + } + fmt.Fprintf(cfg.Notes, "recording to %s (pid %d, pidfile %s, output %s)\n", + cfg.Out, child.cmd.Process.Pid, cfg.PidFile, cfg.DaemonLog) + return nil +} + +// waitForChildReady waits for the child's own readiness byte, and treats every +// other outcome as a failure to record: the pipe closing without a byte (the +// child died on its way to the loop), the process exiting, or the timeout. +// Each of the three quotes the daemon log, which is the only place a detached +// process can explain itself. +func waitForChildReady(cfg WatchConfig, child detachedChild) error { + defer child.ready.Close() + + // This goroutine outlives the function when the child keeps running. It + // stays behind only to reap a child that dies while this short-lived parent + // is still alive, and costs nothing. + exited := make(chan error, 1) + go func() { exited <- child.cmd.Wait() }() + + signalled := make(chan error, 1) + go func() { + _, err := child.ready.Read(make([]byte, 1)) + signalled <- err + }() + + fail := func(format string, args ...any) error { + _ = child.cmd.Process.Kill() + return fmt.Errorf("%s; its output was:\n%s", + fmt.Sprintf(format, args...), daemonLogTail(cfg.DaemonLog, child.logOffset)) + } + + select { + case err := <-signalled: + if err == nil { + return nil + } + // The child closed the pipe — by exiting — without ever reporting that + // it had the log and was polling. + return fail("the detached recorder never reported ready (%v)", err) + case waitErr := <-exited: + status := "exit status 0" + if waitErr != nil { + status = waitErr.Error() + } + return fail("the detached recorder exited before it started recording (%s)", status) + case <-cfg.Clock.After(childReadyTimeout): + return fail("the detached recorder did not report ready within %s", childReadyTimeout) + } +} + +// daemonLogTail quotes the end of the daemon log, starting at from — the size +// the file had when THIS run opened it. The offset is what keeps the quote +// honest when several runs share one --daemon-log path: without it the tail can +// name a previous run's failure as the current one's cause. +func daemonLogTail(path string, from int64) string { + b, err := os.ReadFile(path) + if err != nil { + return fmt.Sprintf("(daemon log %s is unreadable: %v)", path, err) + } + if from > 0 && from <= int64(len(b)) { + b = b[from:] + } + if len(b) > daemonLogTailBytes { + b = b[len(b)-daemonLogTailBytes:] + } + if len(b) == 0 { + return fmt.Sprintf("(this run wrote nothing to the daemon log %s)", path) + } + return strings.TrimRight(string(b), "\n") +} + +// preparedWatch is what the parent has established by the time it is willing +// to detach: an open log with a header and one poll per non-skipped rule, the +// timings that produced them, and the latencies it measured doing so. +type preparedWatch struct { + writer *Writer + header Header + timings map[string]ruleTimings + measured map[string]time.Duration +} + +// prepareWatch is everything the parent does before it detaches. It takes a +// Source rather than building one so the paused-rule, first-observation, +// §3.2 and budget behaviours are all testable with a scripted fake — only the +// process spawning needs a real binary. +func prepareWatch(ctx context.Context, cfg WatchConfig, src Source) (*preparedWatch, error) { + version, err := src.Version(ctx) + if err != nil { + return nil, fmt.Errorf("read grafana version: %w", err) + } + if err := CheckGrafanaVersion(version); err != nil { + return nil, err + } + + defs, err := src.Definitions(ctx) + if err != nil { + return nil, fmt.Errorf("read rule definitions: %w", err) + } + resolved, notes, err := Resolve(defs, cfg.Alerts, cfg.Folder) + if err != nil { + return nil, err + } + for _, n := range notes { + fmt.Fprintf(cfg.Notes, "note: %s\n", n) + } + + rt, _, timingNotes := DeriveTimings(resolved, cfg.PollEvery) + for _, n := range timingNotes { + fmt.Fprintf(cfg.Notes, "note: %s\n", n) + } + for _, d := range resolved { + // A cadence of zero would make the child spin: every rule is due the + // instant it was marked. It also cannot be written into the header, + // where check requires a positive value to derive maxGap from (P5). + if rt[d.UID].pollEvery <= 0 { + return nil, fmt.Errorf("rule %q (%s) reports intervalSeconds=%d: there is no poll cadence to record at", + d.Title, d.UID, d.IntervalSeconds) + } + } + + writer, err := NewWriter(cfg.Out, cfg.Clock) + if err != nil { + return nil, err + } + prep, err := openRecording(ctx, cfg, src, writer, version, resolved, rt) + if err != nil { + // Close, never Stop. The log keeps whatever was written and gets no + // sentinel, so nothing can later mistake it for a finished recording. + _ = writer.Close() + return nil, err + } + return prep, nil +} + +// openRecording writes the header, takes the first observation of every rule +// the recorder will actually watch, appends those observations as the log's +// first heartbeats, and only then decides whether the schedule is feasible. +func openRecording(ctx context.Context, cfg WatchConfig, src Source, writer *Writer, + version string, resolved []Definition, rt map[string]ruleTimings) (*preparedWatch, error) { + + header := Header{ + SchemaVersion: LogSchemaVersion, + URL: cfg.URL, + GrafanaVersion: version, + StartedAt: cfg.Clock.Now(), + Rules: loggedRules(resolved, rt), + } + if err := writer.WriteHeader(header); err != nil { + return nil, err + } + + // A rule whose DEFINITION says is_paused is skipped (§12): it is not + // waited for, not scheduled and never polled. Waiting for one either hangs + // forever or errors before the deploy (§4.3), and recording polls for it + // would report an in-window pause (coverage check 7) for a rule that was + // already paused when the window opened — turning §12's exit 1 into an + // exit 2. The header still names it, with is_paused true, so check reports + // it as skipped from the definitions. + var active []Definition + activeTimings := make(map[string]ruleTimings, len(resolved)) + titles := make(map[string]string, len(resolved)) + for _, d := range resolved { + if d.IsPaused { + fmt.Fprintf(cfg.Notes, "note: rule %q (%s) is paused: recorded as skipped, not waited for (§4.3)\n", d.Title, d.UID) + continue + } + active = append(active, d) + activeTimings[d.UID] = rt[d.UID] + titles[d.UID] = d.Title + fmt.Fprintf(cfg.Notes, "recording %q (%s) every %s (maxGap %s)\n", d.Title, d.UID, rt[d.UID].pollEvery, rt[d.UID].maxGap) + } + + uids := make([]string, 0, len(active)) + for _, d := range active { + uids = append(uids, d.UID) + } + observed, err := observeAll(ctx, src, titles, uids, cfg.Concurrency) + if err != nil { + return nil, err + } + + // Verify §3.2 before anything downstream relies on it: if the state + // endpoint ever stops returning normal instances, the reduction's "keep + // the non-normal ones" silently becomes "keep everything it happened to + // send" and the transition markers lose their ground truth. + for _, d := range active { + if err := VerifyNormalInstancesVisible(observed[d.UID].Rules); err != nil { + return nil, err + } + } + + // One poll record per rule, in resolve order so the log is byte-stable for + // a given set of observations. These ARE the log's first heartbeats: they + // predate the deploy step, which is the whole point of §4.3. + reducer := NewReducer() + measured := make(map[string]time.Duration, len(active)) + for _, d := range active { + obs := observed[d.UID] + measured[d.UID] = obs.Latency + poll := reducer.Reduce(d.UID, obs) + if !poll.Found { + // Authoritative, not transient (P2 already retried transport + // failures): the rule resolved in the ruler API but the state + // endpoint does not serve it. Recorded as Found=false, which P7 + // turns into unobservable — a note rather than an error here, + // because the state endpoint can lag a freshly created rule and + // check re-resolves and fails closed either way. + fmt.Fprintf(cfg.Notes, "warning: rule %q (%s) is absent from the state endpoint; recorded as not found\n", d.Title, d.UID) + } + if err := writer.WritePoll(poll); err != nil { + return nil, err + } + } + + // Budget last, on the latencies just measured — never on a fixed estimate + // (§5.2). Only the active rules count: a skipped rule is never polled and + // consumes none of the capacity. + if err := CheckBudget(activeTimings, measured, cfg.Concurrency); err != nil { + return nil, err + } + + return &preparedWatch{writer: writer, header: header, timings: rt, measured: measured}, nil +} + +// loggedRules snapshots the resolved definitions into the header's rule list. +// Every field but PollEverySeconds is forensic — a resolve-time snapshot that +// makes an uploaded log self-describing (§21.3) — while PollEverySeconds is +// load-bearing: it is the cadence this recording actually used, and check +// derives maxGap from it rather than from the definitions (P5). +func loggedRules(defs []Definition, rt map[string]ruleTimings) []LoggedRule { + out := make([]LoggedRule, 0, len(defs)) + for _, d := range defs { + out = append(out, LoggedRule{ + UID: d.UID, + Title: d.Title, + Folder: d.Folder, + Group: d.Group, + ForSeconds: d.For.Seconds(), + IntervalSeconds: d.IntervalSeconds, + IsPaused: d.IsPaused, + NoDataState: d.NoDataState, + ExecErrState: d.ExecErrState, + PollEverySeconds: rt[d.UID].pollEvery.Seconds(), + }) + } + return out +} + +// observeAll polls every rule in uids concurrently, bounded by concurrency, +// and returns one Observation per rule that answered. Every rule is polled by +// TITLE (the ?rule_name= filter, §2.8) and selected out of the response by +// UID (§14.5) — a filtered response can carry several rules sharing one title. +// +// It returns the successful observations alongside the first error in UID +// order, so a caller that wants to keep the good heartbeats can, and the error +// message is the same on every run. +func observeAll(ctx context.Context, src Source, titles map[string]string, uids []string, concurrency int) (map[string]Observation, error) { + if concurrency < 1 { + concurrency = 1 + } + var ( + mu sync.Mutex + out = make(map[string]Observation, len(uids)) + firstErr error + firstErrUID string + ) + sem := make(chan struct{}, concurrency) + var wg sync.WaitGroup + for _, uid := range uids { + wg.Go(func() { + sem <- struct{}{} + defer func() { <-sem }() + + obs, err := src.RuleState(ctx, titles[uid]) + + mu.Lock() + defer mu.Unlock() + if err != nil { + if firstErr == nil || uid < firstErrUID { + firstErr, firstErrUID = err, uid + } + return + } + out[uid] = obs + }) + } + wg.Wait() + + if firstErr != nil { + return out, fmt.Errorf("poll rule %q (%s): %w", titles[firstErrUID], firstErrUID, firstErr) + } + return out, nil +} + +// DaemonChildConfig is the detached recorder's whole input, and it is +// deliberately tiny. The rule set and every cadence come from the header the +// parent already wrote — one source of truth, no parent/child drift, and it +// exercises ReadLog's header path — and the connection details come from the +// inherited environment. Only the run facts the header does not carry travel +// in argv (§4.4). +type DaemonChildConfig struct { + URL, Token string // from the inherited environment, never from argv (§20.2) + Out string + Until time.Time + Concurrency int + Clock Clock + // ReadyFD is the inherited descriptor to report readiness on (ReadyFDFlag). + // Zero means nobody is waiting — a hand-started child — and the report is + // then skipped rather than written to stdin. + ReadyFD int +} + +// RunDaemonChild is the detached recorder. The CLI dispatches to it when it +// sees DaemonChildFlag; nothing else ever calls it. +// +// It re-reads the log the parent wrote, restores the transition-marker state +// from the polls already in it, reopens the log for appending, takes the flock +// the parent released, and then polls until it is signalled or reaches Until. +func RunDaemonChild(ctx context.Context, cfg DaemonChildConfig) error { + if cfg.Clock == nil { + cfg.Clock = SystemClock{} + } + if cfg.Concurrency < 1 { + cfg.Concurrency = 1 + } + if cfg.Out == "" { + return fmt.Errorf("recorder: no log path") + } + + // Safe to read: the parent closed its writer before spawning this process, + // and no other writer can hold the log's flock (§4.4 step 4). + header, polls, sentinel, err := ReadLog(cfg.Out) + if err != nil { + return err + } + if sentinel != nil { + return fmt.Errorf("log %s already carries a stopped sentinel: another recorder finished it", cfg.Out) + } + // The header's URL is the log's identity (§19.1 step 3). Checking it here + // catches a child that inherited an environment pointing somewhere else, + // before it appends a single poll from the wrong Grafana. + if header.URL != cfg.URL { + return fmt.Errorf("log %s records url %q but this recorder is configured for %q", cfg.Out, header.URL, cfg.URL) + } + + titles, cadence, err := childSchedule(header) + if err != nil { + return err + } + + writer, err := NewWriter(cfg.Out, cfg.Clock) + if err != nil { + return err + } + + reducer := NewReducer() + reducer.seedFrom(polls) + + // SIGTERM is how check stops the recorder (§4.4 step 1); SIGINT is the + // same request from a human at a terminal. Both are clean stops, so both + // end with a sentinel. Registered before the readiness report, so a signal + // arriving the moment the parent unblocks is already handled. + sigCtx, stop := signal.NotifyContext(ctx, syscall.SIGTERM, syscall.SIGINT) + defer stop() + + // Everything that can fail before a single poll has now succeeded: the + // header parsed, the identity matched, the flock is held. That — and not + // the mere fact of having been started — is what the parent waits for. + if err := reportReady(cfg.ReadyFD); err != nil { + return err + } + + return watchLoop(sigCtx, watchLoopConfig{ + Src: NewHTTPSource(cfg.URL, cfg.Token, cfg.Clock), + Writer: writer, + Reducer: reducer, + Titles: titles, + Cadence: cadence, + Until: cfg.Until, + Concurrency: cfg.Concurrency, + Clock: cfg.Clock, + }) +} + +// reportReady writes one byte to the inherited readiness descriptor and closes +// it. fd 0 means no parent is waiting: descriptor 0 is stdin, so it can never +// be a readiness pipe, which makes the zero value safe as "absent". +func reportReady(fd int) error { + if fd == 0 { + return nil + } + pipe := os.NewFile(uintptr(fd), "ready") + if pipe == nil { + return fmt.Errorf("readiness descriptor %d is not open", fd) + } + defer pipe.Close() + if _, err := pipe.Write([]byte{'1'}); err != nil { + return fmt.Errorf("report ready on descriptor %d: %w", fd, err) + } + return nil +} + +// childSchedule derives what the child polls, and how often, from the header +// alone. The cadence comes from PollEverySeconds — the cadence the recording +// actually uses — and is never re-derived from the rule's evaluation interval: +// that is P5's "two authorities", and getting it wrong is fail-open in the +// faster-override direction. Paused rules are excluded here for the same +// reason the parent never polls them (§4.3, §12). +// +// It returns cadences and nothing else. maxGap, healthGrace and evalStaleAfter +// are coverage thresholds applied by the pure layer at classification time, so +// the recorder must not carry them: it would only be able to misuse them. +func childSchedule(h Header) (titles map[string]string, cadence map[string]time.Duration, err error) { + titles = make(map[string]string, len(h.Rules)) + cadence = make(map[string]time.Duration, len(h.Rules)) + for _, lr := range h.Rules { + if lr.IsPaused { + continue + } + if lr.PollEverySeconds <= 0 { + return nil, nil, fmt.Errorf("log header records poll_every_seconds=%v for rule %s (%q): there is no cadence to record at", + lr.PollEverySeconds, lr.UID, lr.Title) + } + if _, duplicate := titles[lr.UID]; duplicate { + return nil, nil, fmt.Errorf("log header names rule %s (%q) twice; its recorded cadence is ambiguous", lr.UID, lr.Title) + } + titles[lr.UID] = lr.Title + cadence[lr.UID] = time.Duration(lr.PollEverySeconds * float64(time.Second)) + } + return titles, cadence, nil +} + +// watchLoopConfig is the child's working state: what to poll, how often, and +// where to append it. There is no threshold in here and no policy — the child +// records and classifies nothing (H5). +type watchLoopConfig struct { + Src Source + Writer *Writer + Reducer *Reducer + Titles map[string]string // uid -> title: poll by title, select by UID + Cadence map[string]time.Duration // uid -> pollEvery, as recorded in the header + Until time.Time + Concurrency int + Clock Clock +} + +// watchLoop is the child's whole working life: poll the rules that are due, +// reduce each observation to one poll record, append it, and — on a clean stop +// only — finish the log with the stopped sentinel. +// +// The sentinel policy is the load-bearing part. A clean stop (a signal, or +// Until) writes it; a hard error does NOT. A recorder that died must look +// exactly like a coverage gap to check, because it is one (§4.5) — writing a +// sentinel on the way out of a failure would hand check a "recording finished" +// claim about a window that stopped being observed. +func watchLoop(ctx context.Context, cfg watchLoopConfig) error { + sched := NewScheduler(cfg.Cadence, cfg.Clock.Now()) + + for { + if ctx.Err() != nil { + return cfg.Writer.Stop() + } + now := cfg.Clock.Now() + if !cfg.Until.IsZero() && !now.Before(cfg.Until) { + return cfg.Writer.Stop() + } + + due := sched.Due(now) + if len(due) == 0 { + wait, ok := untilNextPoll(sched, cfg.Until, now) + if !ok { + // Nothing will ever come due: every watched rule is paused and + // there is no hard stop. Wait for the signal — and still write + // a sentinel, because "the recorder ran and finished" is + // exactly what check needs to prove about the window. + <-ctx.Done() + return cfg.Writer.Stop() + } + select { + case <-ctx.Done(): + return cfg.Writer.Stop() + case <-cfg.Clock.After(wait): + } + continue + } + + // Mark before polling, against the batch's own now: the next poll is + // one cadence after this one was DUE, not one cadence after it + // returned, so request latency cannot make the heartbeat spacing drift + // towards maxGap. + for _, uid := range due { + sched.Mark(uid, now) + } + + pollErr := cfg.pollBatch(ctx, due) + if ctx.Err() != nil { + // Signalled while a poll was in flight. The aborted poll's error is + // not a recorder failure, and a clean stop wins over it (§4.4 step + // 1: finish the in-flight write, then the sentinel). + return cfg.Writer.Stop() + } + if pollErr != nil { + return pollErr + } + } +} + +// pollBatch polls one round of due rules and appends every poll that +// succeeded, in due order, before returning the first failure. Writing the +// successes first is deliberate: a heartbeat that was genuinely observed is +// evidence, and dropping it because a different rule failed would turn one +// rule's transport failure into a coverage gap for the others. +func (cfg watchLoopConfig) pollBatch(ctx context.Context, uids []string) error { + observed, obsErr := observeAll(ctx, cfg.Src, cfg.Titles, uids, cfg.Concurrency) + for _, uid := range uids { + obs, ok := observed[uid] + if !ok { + continue + } + if err := cfg.Writer.WritePoll(cfg.Reducer.Reduce(uid, obs)); err != nil { + return err + } + } + return obsErr +} + +// untilNextPoll returns how long to wait for the next scheduled poll, cut +// short by Until when that comes first. ok is false when nothing will ever +// come due: no rules to poll and no hard stop. +func untilNextPoll(sched *Scheduler, until, now time.Time) (time.Duration, bool) { + next, hasNext := sched.earliestDue() + switch { + case hasNext && (until.IsZero() || next.Before(until)): + // keep next + case !until.IsZero(): + next = until + default: + return 0, false + } + return max(next.Sub(now), 0), true +} + +// writePidFile records the child's pid where check looks for it (P9's +// --pidfile, default .pid). The format is the decimal pid and a newline, +// so `kill $(cat log.jsonl.pid)` works and ReadPidFile stays trivial. +func writePidFile(path string, pid int) error { + if err := os.WriteFile(path, []byte(strconv.Itoa(pid)+"\n"), 0o644); err != nil { + return fmt.Errorf("write pidfile %s: %w", path, err) + } + return nil +} + +// ReadPidFile is the other side of that contract: the pid of the recorder +// check must stop before it may read the log (§4.4 steps 1-4). +func ReadPidFile(path string) (int, error) { + b, err := os.ReadFile(path) + if err != nil { + return 0, fmt.Errorf("read pidfile %s: %w", path, err) + } + text := strings.TrimSpace(string(b)) + pid, err := strconv.Atoi(text) + if err != nil { + return 0, fmt.Errorf("pidfile %s: unparseable pid %q", path, text) + } + if pid <= 0 { + return 0, fmt.Errorf("pidfile %s: %d is not a pid", path, pid) + } + return pid, nil +} diff --git a/grafana-alertcheck/internal/gate/watch_daemon_test.go b/grafana-alertcheck/internal/gate/watch_daemon_test.go new file mode 100644 index 000000000..8a044c7e9 --- /dev/null +++ b/grafana-alertcheck/internal/gate/watch_daemon_test.go @@ -0,0 +1,308 @@ +//go:build unix + +package gate + +import ( + "context" + "encoding/json" + "fmt" + "net/http" + "net/http/httptest" + "os" + "path/filepath" + "slices" + "strconv" + "strings" + "syscall" + "testing" + "time" +) + +// TestMain doubles this test binary as the detached recorder. Watch spawns +// os.Executable(), which under `go test` is this binary, so the one integration +// test below exercises the real thing — a real fork/exec, a real setsid, a real +// inherited environment, a real SIGTERM — with this function standing in for +// the CLI's `watch --daemon-child` dispatch, which lands in P10. +func TestMain(m *testing.M) { + if slices.Contains(os.Args, DaemonChildFlag) { + os.Exit(runTestDaemonChild(os.Args[1:])) + } + os.Exit(m.Run()) +} + +// runTestDaemonChild parses the child argv childArgs() writes, and reads the +// connection details from the environment — never from argv (§20.2). P10's +// `watch` FlagSet does the same four flags. +func runTestDaemonChild(args []string) int { + cfg := DaemonChildConfig{ + URL: os.Getenv("GRAFANA_URL"), + Token: os.Getenv("GRAFANA_TOKEN"), + } + for i := 0; i < len(args); i++ { + value := func() string { + if i+1 >= len(args) { + fmt.Fprintf(os.Stderr, "flag %s wants a value\n", args[i]) + os.Exit(2) + } + i++ + return args[i] + } + switch args[i] { + case "--out": + cfg.Out = value() + case "--until": + until, err := time.Parse(time.RFC3339, value()) + if err != nil { + fmt.Fprintf(os.Stderr, "--until: %v\n", err) + return 2 + } + cfg.Until = until + case "--concurrency": + n, err := strconv.Atoi(value()) + if err != nil { + fmt.Fprintf(os.Stderr, "--concurrency: %v\n", err) + return 2 + } + cfg.Concurrency = n + case ReadyFDFlag: + fd, err := strconv.Atoi(value()) + if err != nil { + fmt.Fprintf(os.Stderr, "%s: %v\n", ReadyFDFlag, err) + return 2 + } + cfg.ReadyFD = fd + } + } + if err := RunDaemonChild(context.Background(), cfg); err != nil { + fmt.Fprintln(os.Stderr, err) + return 2 + } + return 0 +} + +// testBearerToken is what every request to grafanaTestServer must carry. The +// child never receives it in argv (§20.2), so a request that arrives +// authenticated is proof that the token reached the detached process through +// the inherited environment — and a 401 is what a test sees if that ever +// breaks. +const testBearerToken = "test-token" + +// grafanaTestServer serves the three endpoints the record step reads, from the +// real captured fixtures: /api/health, the ruler definitions, and the +// rule_name-filtered state response. The state body is the one-instance +// fixture with its uid and name patched to the ruler fixture's live rule, so +// the reducer's select-by-UID finds it. +func grafanaTestServer(t *testing.T) *httptest.Server { + t.Helper() + ruler := readFixture(t, "ruler_rules.json") + state := patchedStateBody(t) + + srv := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) { + if got := r.Header.Get("Authorization"); got != "Bearer "+testBearerToken { + // Not t.Errorf: this must reach the client as a real 401, so the + // parent fails its version gate and the child fails its polls. + http.Error(w, "unauthorized: Authorization = "+got, http.StatusUnauthorized) + return + } + w.Header().Set("Content-Type", "application/json") + switch { + case r.URL.Path == "/api/health": + fmt.Fprint(w, healthBody("13.1.0")) + case strings.HasPrefix(r.URL.Path, "/api/ruler/"): + _, _ = w.Write(ruler) + case strings.HasPrefix(r.URL.Path, "/api/prometheus/"): + if r.URL.Query().Get("rule_name") == "" { + // §2.8: the gate must never read the state endpoint unfiltered. + http.Error(w, "unfiltered state read", http.StatusBadRequest) + return + } + _, _ = w.Write(state) + default: + http.Error(w, "unexpected path "+r.URL.Path, http.StatusNotFound) + } + })) + t.Cleanup(srv.Close) + return srv +} + +func patchedStateBody(t *testing.T) []byte { + t.Helper() + var body map[string]any + if err := json.Unmarshal(readFixture(t, "state_one_instance.json"), &body); err != nil { + t.Fatalf("unmarshal state fixture: %v", err) + } + data, ok := body["data"].(map[string]any) + if !ok { + t.Fatal("state fixture: no data object") + } + groups, ok := data["groups"].([]any) + if !ok || len(groups) == 0 { + t.Fatal("state fixture: no groups") + } + group, ok := groups[0].(map[string]any) + if !ok { + t.Fatal("state fixture: group 0 is not an object") + } + rules, ok := group["rules"].([]any) + if !ok || len(rules) == 0 { + t.Fatal("state fixture: group 0 has no rules") + } + rule, ok := rules[0].(map[string]any) + if !ok { + t.Fatal("state fixture: rule 0 is not an object") + } + rule["uid"] = watchActiveUID + rule["name"] = watchActiveTitle + rule["lastEvaluation"] = time.Now().UTC().Format(time.RFC3339Nano) + + b, err := json.Marshal(body) + if err != nil { + t.Fatalf("marshal patched state fixture: %v", err) + } + return b +} + +// waitFor polls cond until it holds. This is the one tier of the project where +// a test waits on real time: it drives real processes over real HTTP, so there +// is no clock to fake. +func waitFor(t *testing.T, what string, timeout time.Duration, cond func() bool) { + t.Helper() + deadline := time.Now().Add(timeout) + for time.Now().Before(deadline) { + if cond() { + return + } + time.Sleep(20 * time.Millisecond) + } + t.Fatalf("timed out after %s waiting for %s", timeout, what) +} + +// TestWatchSpawnsADetachedRecorder is P6's one integration test: everything +// from the version gate to the sentinel, through a real detached process. +// +// It asserts the four things only a real spawn can show — the pidfile points +// at a live process, that process is in its own session (setsid, not a bare +// `&`), it keeps appending after Watch returned, and SIGTERM makes it finish +// the log in the §4.4 order — and it uses a 200ms --poll-interval to do it in +// about a second, which also exercises the unclamped-override path (§5.1). +func TestWatchSpawnsADetachedRecorder(t *testing.T) { + srv := grafanaTestServer(t) + t.Setenv("GRAFANA_URL", srv.URL) + t.Setenv("GRAFANA_TOKEN", testBearerToken) + + out := filepath.Join(t.TempDir(), "log.jsonl") + var notes strings.Builder + cfg := WatchConfig{ + URL: srv.URL, + Token: testBearerToken, + Alerts: []string{"uid:" + watchActiveUID}, + Out: out, + PollEvery: 200 * time.Millisecond, + Concurrency: 2, + Notes: ¬es, + } + + if err := Watch(context.Background(), cfg); err != nil { + t.Fatalf("Watch: %v\nnotes:\n%s", err, notes.String()) + } + t.Cleanup(func() { + if t.Failed() { + t.Logf("notes:\n%s", notes.String()) + t.Logf("daemon log:\n%s", daemonLogTail(out+".daemon.log", 0)) + } + }) + + pid, err := ReadPidFile(out + ".pid") + if err != nil { + t.Fatalf("ReadPidFile: %v", err) + } + if err := syscall.Kill(pid, 0); err != nil { + t.Fatalf("recorder pid %d is not running right after Watch returned: %v", pid, err) + } + // Setsid, not a bare `&`: a session leader's process group id is its own + // pid. Without this the child would still share the parent's process group + // and die with the step that started it. + if pgid, err := syscall.Getpgid(pid); err != nil { + t.Errorf("Getpgid(%d): %v", pid, err) + } else if pgid != pid { + t.Errorf("recorder pgid = %d, want %d: it did not get its own session", pgid, pid) + } + + // The parent already wrote the first heartbeat before it returned (§4.3); + // these later ones prove the detached child is the one appending now. + waitFor(t, "the detached recorder to append its own polls", 10*time.Second, func() bool { + _, polls, _, err := ReadLog(out) + return err == nil && len(polls) >= 3 + }) + + // Stop it exactly the way check does. + if err := syscall.Kill(pid, syscall.SIGTERM); err != nil { + t.Fatalf("SIGTERM %d: %v", pid, err) + } + waitFor(t, "the stopped sentinel", 10*time.Second, func() bool { + _, _, sentinel, err := ReadLog(out) + return err == nil && sentinel != nil + }) + + header, polls, sentinel, err := ReadLog(out) + if err != nil { + t.Fatalf("ReadLog: %v", err) + } + if header.URL != srv.URL || header.GrafanaVersion != "13.1.0" { + t.Errorf("header identity = %q/%q, want %q/13.1.0", header.URL, header.GrafanaVersion, srv.URL) + } + if len(header.Rules) != 1 || header.Rules[0].PollEverySeconds != 0.2 { + t.Errorf("header rules = %+v, want one rule recorded at 0.2s", header.Rules) + } + for i, p := range polls { + if p.RuleUID != watchActiveUID || !p.Found { + t.Fatalf("poll %d = %+v, want a found observation of %s", i, p, watchActiveUID) + } + if p.GrafanaNow.IsZero() { + t.Fatalf("poll %d has no grafana_now; H4 needs the Date header of its own response", i) + } + } + if sentinel.Before(header.StartedAt) { + t.Errorf("sentinel at %s precedes the record start %s", sentinel, header.StartedAt) + } + + waitFor(t, "the recorder to exit", 10*time.Second, func() bool { + return syscall.Kill(pid, 0) != nil + }) +} + +// TestWatchFailsWhenTheChildCannotStartRecording is the other half of the +// readiness contract. The child dies on its identity check, so it never reports +// ready — and Watch must say so instead of returning success over a window +// nothing is recording, and must leave no pidfile naming a dead process for the +// next step to signal. +// +// The child is made to fail through the environment, which is the only channel +// it takes its connection details from: the header says one URL and the +// inherited GRAFANA_URL says another. +func TestWatchFailsWhenTheChildCannotStartRecording(t *testing.T) { + srv := grafanaTestServer(t) + t.Setenv("GRAFANA_URL", srv.URL+"/somewhere-else") + t.Setenv("GRAFANA_TOKEN", testBearerToken) + + out := filepath.Join(t.TempDir(), "log.jsonl") + var notes strings.Builder + err := Watch(context.Background(), WatchConfig{ + URL: srv.URL, // what the parent uses, and what the header records + Token: testBearerToken, + Alerts: []string{"uid:" + watchActiveUID}, + Out: out, + PollEvery: 200 * time.Millisecond, + Concurrency: 2, + Notes: ¬es, + }) + if err == nil { + t.Fatal("Watch: no error, but the child could never have started recording") + } + if !strings.Contains(err.Error(), "records url") { + t.Errorf("error does not quote the child's own reason:\n%v", err) + } + if _, statErr := os.Stat(out + ".pid"); !os.IsNotExist(statErr) { + t.Errorf("a pidfile survived a failed detach (%v); pids are reused, so the next step would signal a stranger", statErr) + } +} diff --git a/grafana-alertcheck/internal/gate/watch_test.go b/grafana-alertcheck/internal/gate/watch_test.go new file mode 100644 index 000000000..cecd3a45e --- /dev/null +++ b/grafana-alertcheck/internal/gate/watch_test.go @@ -0,0 +1,702 @@ +package gate + +import ( + "context" + "errors" + "fmt" + "os" + "path/filepath" + "slices" + "strings" + "sync" + "testing" + "time" +) + +// The two fixture rules every prepareWatch test below uses: one live, one +// paused in its definition. Both are addressed by uid:, because the ruler +// fixture deliberately contains a 2-way title collision and a title would make +// the tests depend on which side of it they hit. +const ( + watchActiveUID = "rule0000009" + watchActiveTitle = "Example Failure Ratio Above 10 Percent" + watchPausedUID = "rule0000007" + watchPausedTitle = "example_workflow_paused_rule" +) + +// loopSource answers every RuleState call from a responder that also sees the +// call count, so a recorder-loop test can make the answer depend on virtual +// time or fail on the Nth poll. The loop never reads Version or Definitions — +// the parent did that before detaching — so both fail loudly here. +type loopSource struct { + mu sync.Mutex + calls map[string]int + respond func(title string, call int) (Observation, error) +} + +func newLoopSource(respond func(title string, call int) (Observation, error)) *loopSource { + return &loopSource{calls: map[string]int{}, respond: respond} +} + +func (s *loopSource) Version(context.Context) (string, error) { + return "", errors.New("loopSource: the recorder loop must not read the version") +} + +func (s *loopSource) Definitions(context.Context) ([]Definition, error) { + return nil, errors.New("loopSource: the recorder loop must not read the definitions") +} + +func (s *loopSource) RuleState(_ context.Context, title string) (Observation, error) { + s.mu.Lock() + s.calls[title]++ + call := s.calls[title] + s.mu.Unlock() + return s.respond(title, call) +} + +var _ Source = (*loopSource)(nil) + +// testStateRule is one rule as the state endpoint would return it, healthy and +// evaluated at grafanaNow. +func testStateRule(uid, title string, interval time.Duration, grafanaNow time.Time, instances ...Instance) StateRule { + totals := map[string]int{"normal": len(instances)} + return StateRule{ + UID: uid, Title: title, Folder: "F", Group: "G", + Interval: interval, State: "inactive", Health: "ok", + LastEvaluation: grafanaNow, Totals: totals, Instances: instances, + } +} + +// newLoopWriter opens a log with a header already written, exactly as the +// parent hands it to the child. +func newLoopWriter(t *testing.T, path string, clock Clock) *Writer { + t.Helper() + w, err := NewWriter(path, clock) + if err != nil { + t.Fatalf("NewWriter: %v", err) + } + if err := w.WriteHeader(testHeader()); err != nil { + t.Fatalf("WriteHeader: %v", err) + } + return w +} + +func countPolls(polls []Poll, uid string) int { + n := 0 + for _, p := range polls { + if p.RuleUID == uid { + n++ + } + } + return n +} + +// TestWatchLoopPollsEachRuleAtItsOwnCadence is §5's per-rule schedule seen +// from the recorder: a 10s rule beside a 300s one keeps its own 5s cadence +// instead of dragging the slack rule along with it or being slowed to its pace. +func TestWatchLoopPollsEachRuleAtItsOwnCadence(t *testing.T) { + const tightUID, slackUID = "tight", "slack" + path := filepath.Join(t.TempDir(), "log.jsonl") + clock := newVirtualClock(testNow) + w := newLoopWriter(t, path, clock) + + src := newLoopSource(func(title string, _ int) (Observation, error) { + uid := tightUID + if title == "Slack Rule" { + uid = slackUID + } + now := clock.Now() + return observation(now, testStateRule(uid, title, time.Minute, now)), nil + }) + + err := watchLoop(context.Background(), watchLoopConfig{ + Src: src, + Writer: w, + Reducer: NewReducer(), + Titles: map[string]string{tightUID: "Tight Rule", slackUID: "Slack Rule"}, + Cadence: map[string]time.Duration{ + tightUID: 5 * time.Second, + slackUID: 150 * time.Second, + }, + Until: testNow.Add(300 * time.Second), + Concurrency: 2, + Clock: clock, + }) + if err != nil { + t.Fatalf("watchLoop: %v", err) + } + + _, polls, sentinel, readErr := ReadLog(path) + if readErr != nil { + t.Fatalf("ReadLog: %v", readErr) + } + if sentinel == nil { + t.Fatal("no stopped sentinel after a clean stop") + } + if sentinel.Before(testNow.Add(300 * time.Second)) { + t.Errorf("sentinel at %s, want >= the stop time %s", sentinel, testNow.Add(300*time.Second)) + } + // 300s of window at 5s and 150s, minus the initial stagger offset of up to + // one cadence: 59-60 and 1-2. The assertion is the ratio, not the exact + // count — a single global cycle would give both rules the same number. + if got := countPolls(polls, tightUID); got < 59 || got > 61 { + t.Errorf("tight rule polled %d times, want ~60 (300s at 5s)", got) + } + if got := countPolls(polls, slackUID); got < 1 || got > 3 { + t.Errorf("slack rule polled %d times, want ~2 (300s at 150s)", got) + } +} + +// TestWatchLoopHardErrorLeavesNoSentinel is §4.5's fail-closed rule from the +// recorder's side: a recorder that dies must look exactly like a coverage gap, +// so it must not sign off the log on its way out. +func TestWatchLoopHardErrorLeavesNoSentinel(t *testing.T) { + path := filepath.Join(t.TempDir(), "log.jsonl") + clock := newVirtualClock(testNow) + w := newLoopWriter(t, path, clock) + + boom := errors.New("grafana went away for good") + src := newLoopSource(func(title string, call int) (Observation, error) { + if call >= 2 { + return Observation{}, boom + } + now := clock.Now() + return observation(now, testStateRule("r1", title, time.Minute, now)), nil + }) + + err := watchLoop(context.Background(), watchLoopConfig{ + Src: src, + Writer: w, + Reducer: NewReducer(), + Titles: map[string]string{"r1": "Example"}, + Cadence: map[string]time.Duration{"r1": 30 * time.Second}, + Until: testNow.Add(time.Hour), + Concurrency: 1, + Clock: clock, + }) + if !errors.Is(err, boom) { + t.Fatalf("watchLoop error = %v, want %v", err, boom) + } + + _, polls, sentinel, readErr := ReadLog(path) + if readErr != nil { + t.Fatalf("ReadLog: %v", readErr) + } + if sentinel != nil { + t.Errorf("sentinel at %s after a failed recording; check would read that as a finished window", sentinel) + } + if len(polls) != 1 { + t.Errorf("kept %d polls, want the 1 that succeeded before the failure", len(polls)) + } +} + +// TestWatchLoopSignalDuringPollIsACleanStop pins §4.4 step 1: SIGTERM arriving +// while a poll is in flight is a clean stop, so the aborted poll's error must +// not suppress the sentinel — otherwise every normal check run, which stops the +// recorder exactly this way, would end unobservable. +func TestWatchLoopSignalDuringPollIsACleanStop(t *testing.T) { + path := filepath.Join(t.TempDir(), "log.jsonl") + clock := newVirtualClock(testNow) + w := newLoopWriter(t, path, clock) + + ctx, cancel := context.WithCancel(context.Background()) + defer cancel() + + src := newLoopSource(func(title string, call int) (Observation, error) { + if call >= 2 { + // The signal lands while this request is out. + cancel() + return Observation{}, ctx.Err() + } + now := clock.Now() + return observation(now, testStateRule("r1", title, time.Minute, now)), nil + }) + + if err := watchLoop(ctx, watchLoopConfig{ + Src: src, + Writer: w, + Reducer: NewReducer(), + Titles: map[string]string{"r1": "Example"}, + Cadence: map[string]time.Duration{"r1": 30 * time.Second}, + Concurrency: 1, + Clock: clock, + }); err != nil { + t.Fatalf("watchLoop: %v", err) + } + + if _, _, sentinel, err := ReadLog(path); err != nil { + t.Fatalf("ReadLog: %v", err) + } else if sentinel == nil { + t.Error("no sentinel after a signalled stop; check would call a fully observed window unobservable") + } +} + +// TestWatchLoopWithNothingToPollStillFinishesTheLog covers the every-rule-is- +// paused case: there is nothing to record, but "the recorder ran and finished" +// is still what check has to prove about the window. +func TestWatchLoopWithNothingToPollStillFinishesTheLog(t *testing.T) { + path := filepath.Join(t.TempDir(), "log.jsonl") + clock := newVirtualClock(testNow) + w := newLoopWriter(t, path, clock) + + src := newLoopSource(func(title string, _ int) (Observation, error) { + return Observation{}, fmt.Errorf("nothing should be polled, got %q", title) + }) + + if err := watchLoop(context.Background(), watchLoopConfig{ + Src: src, + Writer: w, + Reducer: NewReducer(), + Titles: map[string]string{}, + Cadence: map[string]time.Duration{}, + Until: testNow.Add(time.Minute), + Concurrency: 1, + Clock: clock, + }); err != nil { + t.Fatalf("watchLoop: %v", err) + } + + _, polls, sentinel, err := ReadLog(path) + if err != nil { + t.Fatalf("ReadLog: %v", err) + } + if len(polls) != 0 { + t.Errorf("wrote %d polls with nothing to poll", len(polls)) + } + if sentinel == nil { + t.Error("no sentinel: check cannot tell this recording from one that died") + } +} + +// TestWatchLoopPollBatchKeepsTheHeartbeatsItGot: one rule's failure must not +// discard another rule's observed heartbeat, or a single transport failure +// turns into a coverage gap for every rule that answered. +func TestWatchLoopPollBatchKeepsTheHeartbeatsItGot(t *testing.T) { + path := filepath.Join(t.TempDir(), "log.jsonl") + clock := newVirtualClock(testNow) + w := newLoopWriter(t, path, clock) + + boom := errors.New("one rule is unreachable") + src := newLoopSource(func(title string, _ int) (Observation, error) { + if title == "Broken" { + return Observation{}, boom + } + now := clock.Now() + return observation(now, testStateRule("ok", title, time.Minute, now)), nil + }) + + cfg := watchLoopConfig{ + Src: src, + Writer: w, + Reducer: NewReducer(), + Titles: map[string]string{"ok": "Healthy", "bad": "Broken"}, + Cadence: map[string]time.Duration{"ok": 30 * time.Second, "bad": 30 * time.Second}, + Concurrency: 2, + Clock: clock, + } + if err := cfg.pollBatch(context.Background(), []string{"ok", "bad"}); !errors.Is(err, boom) { + t.Fatalf("pollBatch error = %v, want %v", err, boom) + } + if err := w.Close(); err != nil { + t.Fatalf("Close: %v", err) + } + + _, polls, _, err := ReadLog(path) + if err != nil { + t.Fatalf("ReadLog: %v", err) + } + if len(polls) != 1 || polls[0].RuleUID != "ok" { + t.Errorf("polls = %+v, want the one heartbeat that was actually observed", polls) + } +} + +// TestReducerSeedFromKeepsMarkersAcrossTheHandoff is H2 at the one seam P6 +// introduces. The parent observes a firing instance; the child starts with a +// fresh Reducer and sees the instance gone. Seeded, that is a vanish — a +// discontinuity. Unseeded, it is nothing at all, and the instance silently +// leaves the record as if it had never been bad. +func TestReducerSeedFromKeepsMarkersAcrossTheHandoff(t *testing.T) { + firing := testInstance(StateFiring, "", "b") + key := instanceKey(firing.Labels) + parentPoll := Poll{RuleUID: "r1", Found: true, Abnormal: []Instance{firing}} + // The child's first response: the instance is gone from the response + // entirely, which is a vanish and never a clear (§4.7). + childObs := observation(testNow, testStateRule("r1", "Example", time.Minute, testNow)) + + t.Run("seeded", func(t *testing.T) { + r := NewReducer() + r.seedFrom([]Poll{parentPoll}) + p := r.Reduce("r1", childObs) + if !slices.Contains(p.Vanished, key) { + t.Errorf("vanished = %v, want it to contain %q", p.Vanished, key) + } + if len(p.Cleared) != 0 { + t.Errorf("cleared = %v, want none: a vanish is not a recovery", p.Cleared) + } + }) + + t.Run("unseeded loses the transition", func(t *testing.T) { + p := NewReducer().Reduce("r1", childObs) + if len(p.Vanished) != 0 { + t.Fatalf("vanished = %v; this subtest exists to show the seed is what produces the marker", p.Vanished) + } + }) + + t.Run("a not-found poll does not clear the seed", func(t *testing.T) { + r := NewReducer() + r.seedFrom([]Poll{parentPoll, {RuleUID: "r1", Found: false}}) + if p := r.Reduce("r1", childObs); !slices.Contains(p.Vanished, key) { + t.Errorf("vanished = %v, want it to contain %q: an absent rule leaves the abnormal set untouched", p.Vanished, key) + } + }) +} + +// watchTestConfig is a prepareWatch config over a temp log, with the notes +// captured so the tests can assert on what an operator is told. +func watchTestConfig(t *testing.T, notes *strings.Builder, alerts ...string) WatchConfig { + t.Helper() + return WatchConfig{ + URL: "https://grafana.example.com", + Token: "secret-token", + Alerts: alerts, + Out: filepath.Join(t.TempDir(), "log.jsonl"), + Concurrency: 2, + Clock: newFakeClock(testNow), + Notes: notes, + }.withDefaults() +} + +// watchTestSource is a fakeSource with the real ruler fixture and a scripted +// state response for the live rule only. The paused rule is deliberately +// unscripted: fakeSource errors on an unscripted title, so any attempt to poll +// it fails the test rather than passing silently. +func watchTestSource(t *testing.T, obs Observation) *fakeSource { + t.Helper() + src := newFakeSource() + src.version = "13.1.0" + src.defs = rulerDefs(t) + src.script(watchActiveTitle, obs, nil) + return src +} + +func liveObservation(grafanaNow time.Time) Observation { + return observation(grafanaNow, testStateRule(watchActiveUID, watchActiveTitle, time.Minute, grafanaNow, + testInstance(StateNormal, "", "a"))) +} + +// TestPrepareWatchDoesNotWaitForPausedRules is §22.4's regression test: a rule +// paused in its definition is skipped, never waited for. Waiting for one either +// hangs forever or errors before the deploy — and the header must still name +// it, so check can report it as skipped rather than lose it. +func TestPrepareWatchDoesNotWaitForPausedRules(t *testing.T) { + var notes strings.Builder + cfg := watchTestConfig(t, ¬es, "uid:"+watchActiveUID, "uid:"+watchPausedUID) + src := watchTestSource(t, liveObservation(testNow)) + + prep, err := prepareWatch(context.Background(), cfg, src) + if err != nil { + t.Fatalf("prepareWatch: %v", err) + } + if err := prep.writer.Close(); err != nil { + t.Fatalf("Close: %v", err) + } + + header, polls, sentinel, err := ReadLog(cfg.Out) + if err != nil { + t.Fatalf("ReadLog: %v", err) + } + if sentinel != nil { + t.Error("the parent wrote a sentinel; that would tell check the recording ended before the child started") + } + + if len(header.Rules) != 2 { + t.Fatalf("header names %d rules, want both the live and the paused one", len(header.Rules)) + } + for _, lr := range header.Rules { + if lr.PollEverySeconds <= 0 { + t.Errorf("header rule %s records poll_every_seconds=%v; check needs a positive cadence to derive maxGap from", lr.UID, lr.PollEverySeconds) + } + if lr.UID == watchPausedUID && !lr.IsPaused { + t.Errorf("header rule %s: is_paused = false, want the resolve-time snapshot to say true", lr.UID) + } + } + + // One poll, for the live rule only — and it is already in the log before + // prepareWatch returned, which is the whole point of §4.3. + if len(polls) != 1 || polls[0].RuleUID != watchActiveUID { + t.Fatalf("polls = %+v, want exactly one first observation of %s", polls, watchActiveUID) + } + if !polls[0].Found || !polls[0].GrafanaNow.Equal(testNow) { + t.Errorf("first poll = %+v, want a found observation at %s", polls[0], testNow) + } + if !strings.Contains(notes.String(), watchPausedTitle) || !strings.Contains(notes.String(), "paused") { + t.Errorf("notes do not mention the paused rule:\n%s", notes.String()) + } +} + +// TestPrepareWatchHeaderRecordsTheOverriddenCadence is P5's "two authorities" +// from the writing side: whatever --poll-interval resolves to is what the +// header records, because that is the only value check may derive maxGap from. +func TestPrepareWatchHeaderRecordsTheOverriddenCadence(t *testing.T) { + var notes strings.Builder + cfg := watchTestConfig(t, ¬es, "uid:"+watchActiveUID) + cfg.PollEvery = 120 * time.Second // the rule evaluates every 60s + src := watchTestSource(t, liveObservation(testNow)) + + prep, err := prepareWatch(context.Background(), cfg, src) + if err != nil { + t.Fatalf("prepareWatch: %v", err) + } + defer prep.writer.Close() + + if got := prep.header.Rules[0].PollEverySeconds; got != 120 { + t.Errorf("header poll_every_seconds = %v, want 120 (the override, used verbatim and never clamped)", got) + } + if got := prep.timings[watchActiveUID].maxGap; got != 240*time.Second { + t.Errorf("maxGap = %s, want 240s (2 x the recorded cadence)", got) + } + if !strings.Contains(notes.String(), "--poll-interval") { + t.Errorf("notes do not report that the override exceeds half the evaluation interval:\n%s", notes.String()) + } +} + +// TestPrepareWatchFailsWhenTheScheduleDoesNotFit: the budget check runs on the +// latencies the parent just measured, before the deploy runs (§5.2). +func TestPrepareWatchFailsWhenTheScheduleDoesNotFit(t *testing.T) { + var notes strings.Builder + cfg := watchTestConfig(t, ¬es, "uid:"+watchActiveUID) + + obs := liveObservation(testNow) + obs.Latency = 60 * time.Second // against a 30s cadence + src := watchTestSource(t, obs) + + _, err := prepareWatch(context.Background(), cfg, src) + if err == nil { + t.Fatal("prepareWatch: no error on a schedule that cannot hold its own cadence") + } + assertBudgetMessage(t, err.Error()) +} + +// TestPrepareWatchVerifiesNormalInstancesAreVisible is the §3.2 check at the +// one place it can still be cheap: the first observation. If the state endpoint +// stops returning normal instances, the reduction's predicate quietly inverts. +func TestPrepareWatchVerifiesNormalInstancesAreVisible(t *testing.T) { + var notes strings.Builder + cfg := watchTestConfig(t, ¬es, "uid:"+watchActiveUID) + + rule := testStateRule(watchActiveUID, watchActiveTitle, time.Minute, testNow, testInstance(StateFiring, "", "b")) + rule.Totals = map[string]int{"alerting": 1, "normal": 4} // claims normals it did not return + src := watchTestSource(t, observation(testNow, rule)) + + _, err := prepareWatch(context.Background(), cfg, src) + if err == nil { + t.Fatal("prepareWatch: no error when totals claim normal instances the response omitted") + } + if !strings.Contains(err.Error(), "3.2") { + t.Errorf("error does not name §3.2: %v", err) + } + + // The failure happens before any poll is appended, so the log holds a + // header and nothing else. + if _, polls, _, readErr := ReadLog(cfg.Out); readErr != nil { + t.Fatalf("ReadLog: %v", readErr) + } else if len(polls) != 0 { + t.Errorf("wrote %d polls from an observation it refused to trust", len(polls)) + } +} + +func TestPrepareWatchRejectsAnUnsupportedGrafana(t *testing.T) { + var notes strings.Builder + cfg := watchTestConfig(t, ¬es, "uid:"+watchActiveUID) + src := watchTestSource(t, liveObservation(testNow)) + src.version = "12.4.0" + + if _, err := prepareWatch(context.Background(), cfg, src); err == nil { + t.Fatal("prepareWatch: no error on an unsupported grafana version") + } else if !strings.Contains(err.Error(), "12.4.0") || !strings.Contains(err.Error(), "13.0.0") { + t.Errorf("error names neither what was found nor what is supported: %v", err) + } +} + +// TestPrepareWatchNotesAnAbsentRule: a rule that resolved in the ruler API but +// is absent from the state endpoint is recorded as Found=false — authoritative +// evidence P7 turns into unobservable — not silently dropped. +func TestPrepareWatchNotesAnAbsentRule(t *testing.T) { + var notes strings.Builder + cfg := watchTestConfig(t, ¬es, "uid:"+watchActiveUID) + src := watchTestSource(t, observation(testNow)) // an authoritative, empty 2xx + + prep, err := prepareWatch(context.Background(), cfg, src) + if err != nil { + t.Fatalf("prepareWatch: %v", err) + } + if err := prep.writer.Close(); err != nil { + t.Fatalf("Close: %v", err) + } + + _, polls, _, err := ReadLog(cfg.Out) + if err != nil { + t.Fatalf("ReadLog: %v", err) + } + if len(polls) != 1 || polls[0].Found { + t.Fatalf("polls = %+v, want one poll recorded as not found", polls) + } + if !strings.Contains(notes.String(), "absent from the state endpoint") { + t.Errorf("notes do not warn about the absent rule:\n%s", notes.String()) + } +} + +func TestWatchConfigValidation(t *testing.T) { + base := func() WatchConfig { + return WatchConfig{ + URL: "https://grafana.example.com", + Alerts: []string{"Example"}, + Out: filepath.Join(t.TempDir(), "log.jsonl"), + Clock: newFakeClock(testNow), + } + } + + for _, tc := range []struct { + name string + mutate func(*WatchConfig) + want string + }{ + {"no url", func(c *WatchConfig) { c.URL = "" }, "url"}, + {"no log path", func(c *WatchConfig) { c.Out = "" }, "no log path"}, + {"no alerts", func(c *WatchConfig) { c.Alerts = nil }, "no alert names"}, + {"blank alerts only", func(c *WatchConfig) { c.Alerts = []string{"", " "} }, "no alert names"}, + {"until in the past", func(c *WatchConfig) { c.Until = testNow.Add(-time.Second) }, "not in the future"}, + } { + t.Run(tc.name, func(t *testing.T) { + cfg := base() + tc.mutate(&cfg) + err := cfg.withDefaults().validate() + if err == nil { + t.Fatalf("validate: no error, want one naming %q", tc.want) + } + if !strings.Contains(err.Error(), tc.want) { + t.Errorf("validate error = %v, want it to name %q", err, tc.want) + } + }) + } + + t.Run("defaults derive the pidfile and daemon log from the log path", func(t *testing.T) { + cfg := base().withDefaults() + if cfg.PidFile != cfg.Out+".pid" { + t.Errorf("PidFile = %q, want %q — check finds the recorder by this convention", cfg.PidFile, cfg.Out+".pid") + } + if cfg.DaemonLog == "" { + t.Error("DaemonLog is empty: a detached child would have nowhere to explain a failure") + } + if err := cfg.validate(); err != nil { + t.Errorf("validate: %v", err) + } + }) +} + +// TestChildScheduleUsesTheRecordedCadence is P5's fail-open direction, checked +// on the child's side: a log recorded at 5s on a 300s rule must schedule at 5s. +// Re-deriving from the interval would give 150s — and every real 250s hole in +// that recording would pass. +func TestChildScheduleUsesTheRecordedCadence(t *testing.T) { + h := Header{Rules: []LoggedRule{ + {UID: "fast", Title: "Fast", IntervalSeconds: 300, PollEverySeconds: 5}, + {UID: "paused", Title: "Paused", IntervalSeconds: 60, PollEverySeconds: 30, IsPaused: true}, + }} + + titles, cadence, err := childSchedule(h) + if err != nil { + t.Fatalf("childSchedule: %v", err) + } + if _, ok := titles["paused"]; ok { + t.Error("the child scheduled a rule that was paused when the window opened (§4.3)") + } + if got := cadence["fast"]; got != 5*time.Second { + t.Errorf("pollEvery = %s, want 5s from the header, not %s from the interval", got, defaultPollEvery(300)) + } +} + +func TestChildScheduleRejectsAnUnusableHeader(t *testing.T) { + for _, tc := range []struct { + name string + h Header + want string + }{ + { + "no recorded cadence", + Header{Rules: []LoggedRule{{UID: "r1", Title: "Example", IntervalSeconds: 60}}}, + "no cadence", + }, + { + "the same rule twice", + Header{Rules: []LoggedRule{ + {UID: "r1", Title: "Example", IntervalSeconds: 60, PollEverySeconds: 30}, + {UID: "r1", Title: "Example", IntervalSeconds: 60, PollEverySeconds: 300}, + }}, + "twice", + }, + } { + t.Run(tc.name, func(t *testing.T) { + if _, _, err := childSchedule(tc.h); err == nil { + t.Fatalf("childSchedule: no error, want one naming %q", tc.want) + } else if !strings.Contains(err.Error(), tc.want) { + t.Errorf("error = %v, want it to name %q", err, tc.want) + } + }) + } +} + +// TestChildArgsCarryNoSecretsAndNoRuleSet: the child's command line lands in +// the process table and in CI logs. Everything it needs about the rules comes +// from the header, and everything about the connection comes from the +// environment — so argv holds the log path and the two run facts only. +func TestChildArgsCarryNoSecretsAndNoRuleSet(t *testing.T) { + cfg := WatchConfig{ + URL: "https://grafana.example.com", + Token: "secret-token", + Alerts: []string{"Example"}, + Folder: "F", + Out: "/tmp/log.jsonl", + PidFile: "/tmp/log.jsonl.pid", + Until: testNow.Add(time.Hour), + PollEvery: 17 * time.Second, + Concurrency: 3, + } + args := childArgs(cfg) + joined := strings.Join(args, " ") + + for _, want := range []string{DaemonChildFlag, "--out /tmp/log.jsonl", "--concurrency 3", "--until ", ReadyFDFlag + " 3"} { + if !strings.Contains(joined, want) { + t.Errorf("child args %q do not contain %q", joined, want) + } + } + for _, forbidden := range []string{"secret-token", "Example", "--folder", "--poll-interval", "--pidfile"} { + if strings.Contains(joined, forbidden) { + t.Errorf("child args %q contain %q, which must not reach argv", joined, forbidden) + } + } +} + +func TestPidFileRoundTrip(t *testing.T) { + path := filepath.Join(t.TempDir(), "log.jsonl.pid") + if err := writePidFile(path, 4242); err != nil { + t.Fatalf("writePidFile: %v", err) + } + pid, err := ReadPidFile(path) + if err != nil { + t.Fatalf("ReadPidFile: %v", err) + } + if pid != 4242 { + t.Errorf("pid = %d, want 4242", pid) + } + + t.Run("garbage is an error, never a pid", func(t *testing.T) { + bad := filepath.Join(t.TempDir(), "bad.pid") + if err := os.WriteFile(bad, []byte("not-a-pid\n"), 0o644); err != nil { + t.Fatalf("write: %v", err) + } + if _, err := ReadPidFile(bad); err == nil { + t.Error("ReadPidFile: no error on an unparseable pidfile") + } + }) +} diff --git a/grafana-alertcheck/internal/gate/watch_unix.go b/grafana-alertcheck/internal/gate/watch_unix.go new file mode 100644 index 000000000..6d6096786 --- /dev/null +++ b/grafana-alertcheck/internal/gate/watch_unix.go @@ -0,0 +1,112 @@ +//go:build unix + +package gate + +import ( + "fmt" + "os" + "os/exec" + "strconv" + "syscall" + "time" +) + +// readyFD is where ExtraFiles[0] lands in the child: exec.Cmd starts extra +// descriptors at 3, after stdin, stdout and stderr. +const readyFD = 3 + +// detachedChild is a started recorder: the process, the read end of its +// readiness pipe, and the size the daemon log had before it wrote anything. +type detachedChild struct { + cmd *exec.Cmd + ready *os.File + // logOffset is where this run's output starts in the daemon log, so a + // failure quotes this child and never a previous run's. + logOffset int64 +} + +// spawnChild re-execs this binary as the detached recorder (§4.4). A trailing +// `&` is NOT sufficient: the child would keep the parent's session and process +// group, so it would still take the terminal's signals and, on a runner, die +// with the step that started it. Setsid gives it a new session AND a new +// process group, which is what makes it survive to the end of the window. +// +// There is deliberately no Windows counterpart — runners are Linux and +// goreleaser builds linux+darwin only (P12) — so the package does not build +// there at all rather than silently recording in the foreground. +func spawnChild(cfg WatchConfig) (detachedChild, error) { + exe, err := os.Executable() + if err != nil { + return detachedChild{}, fmt.Errorf("find own executable: %w", err) + } + + // The child is detached, so its output has nowhere to go but a file, and + // that file is the only place it can ever explain a failure. Opened + // O_APPEND, never O_TRUNC: --daemon-log is an operator-supplied path and + // this process does not get to destroy what is already in it. The offset + // below is what keeps a shared path from misattributing a failure. + logFile, err := os.OpenFile(cfg.DaemonLog, os.O_APPEND|os.O_CREATE|os.O_WRONLY, 0o644) + if err != nil { + return detachedChild{}, fmt.Errorf("open daemon log %s: %w", cfg.DaemonLog, err) + } + // The child inherits the descriptor at Start; this process does not need + // its own copy afterwards. + defer logFile.Close() + + var logOffset int64 + if info, err := logFile.Stat(); err == nil { + logOffset = info.Size() + } + + // The readiness pipe (§ReadyFDFlag): the child gets the write end as + // descriptor 3 and reports on it once it holds the log and is polling. + readyRead, readyWrite, err := os.Pipe() + if err != nil { + return detachedChild{}, fmt.Errorf("open readiness pipe: %w", err) + } + + cmd := exec.Command(exe, childArgs(cfg)...) + cmd.Stdin = nil // /dev/null + cmd.Stdout = logFile + cmd.Stderr = logFile + cmd.ExtraFiles = []*os.File{readyWrite} // descriptor 3 in the child + // The environment is how the connection details reach the child (§20.2): + // the token must never appear in argv, where it would land in the process + // table and in CI logs. + cmd.Env = os.Environ() + cmd.SysProcAttr = &syscall.SysProcAttr{Setsid: true} + + if err := cmd.Start(); err != nil { + readyRead.Close() + readyWrite.Close() + return detachedChild{}, fmt.Errorf("start recorder %s: %w", exe, err) + } + // Drop the parent's copy of the write end at once: with only the child + // holding it, a child that dies before signalling closes the pipe and the + // parent reads EOF instead of waiting out the whole timeout. + readyWrite.Close() + + return detachedChild{cmd: cmd, ready: readyRead, logOffset: logOffset}, nil +} + +// childArgs builds the child's command line. The rule set, every cadence and +// the recording's identity all come from the header the parent already wrote, +// and the connection details come from the environment — so what is left here +// is the log path plus the two run facts the header does not carry: the +// optional hard stop and the concurrency limit. +// +// Notably absent: --pidfile (the parent writes it, so check can find the pid +// the instant Watch returns), --alerts, --folder, --poll-interval, and +// anything derived from them. The CLI's `watch` FlagSet needs one flag of its +// own for this path — ReadyFDFlag — and dispatches to RunDaemonChild when it +// sees DaemonChildFlag. +func childArgs(cfg WatchConfig) []string { + args := []string{"watch", DaemonChildFlag, "--out", cfg.Out, ReadyFDFlag, strconv.Itoa(readyFD)} + if !cfg.Until.IsZero() { + args = append(args, "--until", cfg.Until.Format(time.RFC3339)) + } + if cfg.Concurrency > 0 { + args = append(args, "--concurrency", strconv.Itoa(cfg.Concurrency)) + } + return args +} From 6015157b6615e903d668231bd11011ab3ccb09f0 Mon Sep 17 00:00:00 2001 From: Bartek Tofel Date: Wed, 2 Sep 2026 11:44:28 +0200 Subject: [PATCH 17/43] chore: add a unit test, remove build tags --- .../internal/gate/watch_daemon_test.go | 36 +++++++++++++++++-- .../gate/{watch_unix.go => watch_process.go} | 6 ---- 2 files changed, 34 insertions(+), 8 deletions(-) rename grafana-alertcheck/internal/gate/{watch_unix.go => watch_process.go} (94%) diff --git a/grafana-alertcheck/internal/gate/watch_daemon_test.go b/grafana-alertcheck/internal/gate/watch_daemon_test.go index 8a044c7e9..35e9c424a 100644 --- a/grafana-alertcheck/internal/gate/watch_daemon_test.go +++ b/grafana-alertcheck/internal/gate/watch_daemon_test.go @@ -1,5 +1,3 @@ -//go:build unix - package gate import ( @@ -271,6 +269,40 @@ func TestWatchSpawnsADetachedRecorder(t *testing.T) { }) } +// TestDaemonChildRejectsAnAlreadyFinishedLog covers the RunDaemonChild guard +// against a reused --out path: a log that already carries a stopped sentinel is +// a finished recording, and a child starting against it would either append to +// a window already declared over, or take a flock over evidence that is about +// to be classified — so it must refuse before polling once. This is the +// fail-closed counterpart of §4.5 on the recorder's own startup path. +func TestDaemonChildRejectsAnAlreadyFinishedLog(t *testing.T) { + path := filepath.Join(t.TempDir(), "log.jsonl") + clock := newFakeClock(testNow) + + w, err := NewWriter(path, clock) + if err != nil { + t.Fatalf("NewWriter: %v", err) + } + if err := w.WriteHeader(testHeader()); err != nil { + t.Fatalf("WriteHeader: %v", err) + } + if err := w.Stop(); err != nil { + t.Fatalf("Stop: %v", err) + } + + err = RunDaemonChild(context.Background(), DaemonChildConfig{ + URL: testHeader().URL, + Out: path, + Clock: clock, + }) + if err == nil { + t.Fatal("RunDaemonChild: no error against a log that already carries a stopped sentinel") + } + if !strings.Contains(err.Error(), "sentinel") { + t.Errorf("error = %v, want it to name the stopped sentinel", err) + } +} + // TestWatchFailsWhenTheChildCannotStartRecording is the other half of the // readiness contract. The child dies on its identity check, so it never reports // ready — and Watch must say so instead of returning success over a window diff --git a/grafana-alertcheck/internal/gate/watch_unix.go b/grafana-alertcheck/internal/gate/watch_process.go similarity index 94% rename from grafana-alertcheck/internal/gate/watch_unix.go rename to grafana-alertcheck/internal/gate/watch_process.go index 6d6096786..629606188 100644 --- a/grafana-alertcheck/internal/gate/watch_unix.go +++ b/grafana-alertcheck/internal/gate/watch_process.go @@ -1,5 +1,3 @@ -//go:build unix - package gate import ( @@ -30,10 +28,6 @@ type detachedChild struct { // group, so it would still take the terminal's signals and, on a runner, die // with the step that started it. Setsid gives it a new session AND a new // process group, which is what makes it survive to the end of the window. -// -// There is deliberately no Windows counterpart — runners are Linux and -// goreleaser builds linux+darwin only (P12) — so the package does not build -// there at all rather than silently recording in the foreground. func spawnChild(cfg WatchConfig) (detachedChild, error) { exe, err := os.Executable() if err != nil { From 155f65e007127ab37d06ca6407e20784b499b5bb Mon Sep 17 00:00:00 2001 From: Bartek Tofel Date: Mon, 7 Sep 2026 16:28:58 +0200 Subject: [PATCH 18/43] chore: address code review comments --- grafana-alertcheck/.tool-versions | 3 ++ grafana-alertcheck/internal/gate/schedule.go | 6 ++- .../internal/gate/schedule_test.go | 42 ++++++++++++++++ grafana-alertcheck/internal/gate/watch.go | 29 +++++++++-- .../internal/gate/watch_test.go | 48 +++++++++++++++++++ 5 files changed, 121 insertions(+), 7 deletions(-) create mode 100644 grafana-alertcheck/.tool-versions diff --git a/grafana-alertcheck/.tool-versions b/grafana-alertcheck/.tool-versions new file mode 100644 index 000000000..ef699cf97 --- /dev/null +++ b/grafana-alertcheck/.tool-versions @@ -0,0 +1,3 @@ +# golangci-lint: keep in sync with devbox.json (used by CI in .github/workflows/linters.yml via `devbox run -- just lint`). +golang 1.26.6 +golangci-lint 2.12.2 diff --git a/grafana-alertcheck/internal/gate/schedule.go b/grafana-alertcheck/internal/gate/schedule.go index 5712cda51..ec89dcb13 100644 --- a/grafana-alertcheck/internal/gate/schedule.go +++ b/grafana-alertcheck/internal/gate/schedule.go @@ -269,12 +269,14 @@ func (s *Scheduler) Mark(uid string, now time.Time) error { // maxGap. func (s *Scheduler) earliestDue() (time.Time, bool) { var earliest time.Time + ok := false for _, t := range s.next { - if earliest.IsZero() || t.Before(earliest) { + if !ok || t.Before(earliest) { earliest = t + ok = true } } - return earliest, !earliest.IsZero() + return earliest, ok } // CheckBudget applies §5's error-at-start check to a fully resolved schedule. diff --git a/grafana-alertcheck/internal/gate/schedule_test.go b/grafana-alertcheck/internal/gate/schedule_test.go index 07f0184ff..9985a0bd3 100644 --- a/grafana-alertcheck/internal/gate/schedule_test.go +++ b/grafana-alertcheck/internal/gate/schedule_test.go @@ -223,6 +223,48 @@ func TestNewScheduler_StaggersWithinPollEvery(t *testing.T) { } } +func TestScheduler_EarliestDueEmpty(t *testing.T) { + s := &Scheduler{next: map[string]time.Time{}, every: map[string]time.Duration{}} + if _, ok := s.earliestDue(); ok { + t.Fatalf("earliestDue on an empty scheduler = ok=true, want false") + } +} + +// TestScheduler_EarliestDueZeroTime pins the empty-detection fix: a +// non-empty scheduler whose earliest next-due time is the zero time must still +// report ok=true. The old IsZero() sentinel misread exactly this as "no rules". +func TestScheduler_EarliestDueZeroTime(t *testing.T) { + s := &Scheduler{ + next: map[string]time.Time{"r1": {}}, + every: map[string]time.Duration{"r1": time.Second}, + } + earliest, ok := s.earliestDue() + if !ok { + t.Fatalf("earliestDue = ok=false, want true (the zero time is a real next-due, not an empty scheduler)") + } + if !earliest.IsZero() { + t.Errorf("earliestDue = %v, want the zero time", earliest) + } +} + +func TestScheduler_EarliestDuePicksMinimum(t *testing.T) { + now := time.Date(2026, 1, 1, 0, 0, 0, 0, time.UTC) + s := &Scheduler{ + next: map[string]time.Time{ + "later": now.Add(2 * time.Minute), + "soon": now.Add(time.Minute), + }, + every: map[string]time.Duration{"later": time.Minute, "soon": time.Minute}, + } + earliest, ok := s.earliestDue() + if !ok { + t.Fatalf("earliestDue = ok=false, want true") + } + if !earliest.Equal(now.Add(time.Minute)) { + t.Errorf("earliestDue = %v, want the earliest next-due time", earliest) + } +} + // TestCheckBudget_MixedIntervalRegression is §22.3's sanity check from the // plan: one rule at 10s beside twenty at 300s, all measured ~1.8s, must not // error at any reasonable concurrency — the exact case a naive worst-case-slot diff --git a/grafana-alertcheck/internal/gate/watch.go b/grafana-alertcheck/internal/gate/watch.go index 61e1cce63..f0f24f247 100644 --- a/grafana-alertcheck/internal/gate/watch.go +++ b/grafana-alertcheck/internal/gate/watch.go @@ -248,15 +248,34 @@ func waitForChildReady(cfg WatchConfig, child detachedChild) error { // honest when several runs share one --daemon-log path: without it the tail can // name a previous run's failure as the current one's cause. func daemonLogTail(path string, from int64) string { - b, err := os.ReadFile(path) + f, err := os.Open(path) + if err != nil { + return fmt.Sprintf("(daemon log %s is unreadable: %v)", path, err) + } + defer f.Close() + + info, err := f.Stat() if err != nil { return fmt.Sprintf("(daemon log %s is unreadable: %v)", path, err) } - if from > 0 && from <= int64(len(b)) { - b = b[from:] + size := info.Size() + + // Only the tail is ever quoted, so read at most daemonLogTailBytes from + // disk rather than the whole (possibly unbounded, shared-across-runs) file. + start := int64(0) + if from > 0 && from <= size { + start = from + } + if tailStart := size - daemonLogTailBytes; tailStart > start { + start = tailStart + } + if _, err := f.Seek(start, io.SeekStart); err != nil { + return fmt.Sprintf("(daemon log %s is unreadable: %v)", path, err) } - if len(b) > daemonLogTailBytes { - b = b[len(b)-daemonLogTailBytes:] + + b, err := io.ReadAll(io.LimitReader(f, daemonLogTailBytes)) + if err != nil { + return fmt.Sprintf("(daemon log %s is unreadable: %v)", path, err) } if len(b) == 0 { return fmt.Sprintf("(this run wrote nothing to the daemon log %s)", path) diff --git a/grafana-alertcheck/internal/gate/watch_test.go b/grafana-alertcheck/internal/gate/watch_test.go index cecd3a45e..fa2af889a 100644 --- a/grafana-alertcheck/internal/gate/watch_test.go +++ b/grafana-alertcheck/internal/gate/watch_test.go @@ -700,3 +700,51 @@ func TestPidFileRoundTrip(t *testing.T) { } }) } + +func TestDaemonLogTail(t *testing.T) { + t.Run("missing file is unreadable", func(t *testing.T) { + out := daemonLogTail(filepath.Join(t.TempDir(), "nope.daemon.log"), 0) + if !strings.Contains(out, "unreadable") { + t.Errorf("daemonLogTail = %q, want it to name the file as unreadable", out) + } + }) + + t.Run("small file returns its content", func(t *testing.T) { + path := filepath.Join(t.TempDir(), "small.daemon.log") + if err := os.WriteFile(path, []byte("line one\nline two\n"), 0o644); err != nil { + t.Fatalf("write: %v", err) + } + if out := daemonLogTail(path, 0); out != "line one\nline two" { + t.Errorf("daemonLogTail = %q, want the full trimmed content", out) + } + }) + + t.Run("large file keeps only the tail", func(t *testing.T) { + path := filepath.Join(t.TempDir(), "large.daemon.log") + prefix := strings.Repeat("P", 1000) + suffix := strings.Repeat("S", daemonLogTailBytes) + if err := os.WriteFile(path, []byte(prefix+suffix), 0o644); err != nil { + t.Fatalf("write: %v", err) + } + out := daemonLogTail(path, 0) + if out != suffix { + t.Errorf("daemonLogTail = %q, want exactly the trailing %d bytes (the %d leading bytes dropped)", out, daemonLogTailBytes, len(prefix)) + } + }) + + t.Run("offset skips a previous run's content", func(t *testing.T) { + path := filepath.Join(t.TempDir(), "shared.daemon.log") + prior := strings.Repeat("p", 2000) + if err := os.WriteFile(path, []byte(prior), 0o644); err != nil { + t.Fatalf("write: %v", err) + } + from := int64(len(prior)) + thisRun := "this run's output\n" + if err := os.WriteFile(path, []byte(prior+thisRun), 0o644); err != nil { + t.Fatalf("write: %v", err) + } + if out := daemonLogTail(path, from); out != "this run's output" { + t.Errorf("daemonLogTail = %q, want only this run's bytes after offset %d", out, from) + } + }) +} From 1bb0b9d8e0eeab03756f10210bea7bf4237514b9 Mon Sep 17 00:00:00 2001 From: Bartek Tofel Date: Mon, 31 Aug 2026 16:16:32 +0200 Subject: [PATCH 19/43] chore: implement phase 7 MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Invariant defended: H3. The one question: can a rule be called alive because it looked alive one poll ago? proveCoverage (grafana-alertcheck/internal/gate/coverage.go) is the pure coverage function: nine checks over one rule's polls — sentinel, from-bounds, heartbeat continuity, health error/nodata, liveness, in-window pause, rule absence, KeepLast. Liveness is absolute, never a delta. Cross-domain comparisons translate by each poll's own skew and widen boundary segments by its skew bound, fail-closed. --- grafana-alertcheck/internal/gate/coverage.go | 352 ++++++++++ .../internal/gate/coverage_test.go | 628 ++++++++++++++++++ 2 files changed, 980 insertions(+) create mode 100644 grafana-alertcheck/internal/gate/coverage.go create mode 100644 grafana-alertcheck/internal/gate/coverage_test.go diff --git a/grafana-alertcheck/internal/gate/coverage.go b/grafana-alertcheck/internal/gate/coverage.go new file mode 100644 index 000000000..36d2c3dac --- /dev/null +++ b/grafana-alertcheck/internal/gate/coverage.go @@ -0,0 +1,352 @@ +package gate + +import ( + "fmt" + "slices" + "time" +) + +// keepLastReason is the instance Reason that check 9 watches for (§10.2). +const keepLastReason = "KeepLast" + +// Obligations this phase leaves for later ones — carried forward the same +// way P6's own deviations list did, so a later review has something concrete +// to check against: +// +// - fromFutureTolerance (§5: 60s) has no constant and no hard-error check +// anywhere yet. Check 2 below implements only "from < StartedAt"; the +// second clause — from more than fromFutureTolerance ahead is a hard +// error — is once-per-run input validation, not a per-rule coverage +// check, and belongs to Check's construction in a later phase (P9). +// - decide (P8) must read a rule's skipped status from the definitions +// (LoggedRule.IsPaused / Definition.IsPaused), never from the polls, and +// must do so BEFORE calling proveCoverage for that rule: a rule paused +// before the window opened is never scheduled or polled (§4.3), so it +// reaches this function with zero polls and today reads as one large +// heartbeat_gap, not skipped (pinned by +// TestProveCoverage_SkippedRuleWithZeroPollsPinnedAsHeartbeatGap). + +// UnobservableReason names why proveCoverage could not prove a rule's window. +// It is machine-readable — this reaches the action's JSON outputs, so it is a +// published vocabulary like Outcome (§19.0); prose belongs in Notes. +type UnobservableReason string + +const ( + ReasonNoSentinel UnobservableReason = "no_sentinel" + ReasonSentinelEarly UnobservableReason = "sentinel_early" + ReasonFromBeforeRecord UnobservableReason = "from_before_record" + ReasonHeartbeatGap UnobservableReason = "heartbeat_gap" + ReasonHealthError UnobservableReason = "health_error" + ReasonStaleEvaluation UnobservableReason = "stale_evaluation" + ReasonPausedInWindow UnobservableReason = "paused_in_window" + ReasonRuleAbsent UnobservableReason = "rule_absent" + // ReasonDrainTimeout is set by check.go's drain wait (a later phase), + // never by proveCoverage: the wait is I/O and must not be added to this + // pure function — that would put HTTP inside the pure layer and destroy + // the seam §2's architecture depends on. + ReasonDrainTimeout UnobservableReason = "drain_timeout" +) + +// CoverageResult is proveCoverage's whole answer for one rule. No interval +// list: proved-or-not plus the largest gap and where is everything a human +// reads on exit 2, and everything §20.2's table needs. +type CoverageResult struct { + Proved bool + LargestGap time.Duration + LargestGapAt time.Time + Unobservable bool + Reason UnobservableReason + Notes []string + // BlindFor is the worst staleness (GrafanaNow - LastEvaluation) that + // tripped check 6; zero when check 6 never fired. + BlindFor time.Duration +} + +// proveCoverage applies the nine coverage checks (§6, §10, §14) to one rule's +// polls and is PURE: no HTTP, no files, no clock reads — everything it needs +// arrives as an argument, which is what lets §22's tests build []Poll literals +// instead of a fixture server (§2). +// +// polls need not be pre-filtered to this rule: proveCoverage selects by +// def.UID itself, exactly as Reduce selects by UID rather than by title +// (§14.5) — a caller handing it a whole log's polls must not have to +// pre-filter to get a correct answer. +// +// Every check always runs, even once an earlier one has already set +// Unobservable: LargestGap and the notes are diagnostics an operator reads on +// exit 2 regardless of which check actually failed (§20.2). Reason names the +// FIRST check, in the order below, that failed; a later failure still adds +// its own Note. +func proveCoverage(h Header, polls []Poll, sentinel *time.Time, t ruleTimings, def Definition, + from, to time.Time, grace time.Duration) CoverageResult { + + windowEnd := to.Add(grace) + + var rulePolls []Poll + for _, p := range polls { + if p.RuleUID == def.UID { + rulePolls = append(rulePolls, p) + } + } + // Stable, not sort.Slice: two polls sharing a GrafanaNow (a coarse Date + // header, or a corrupted/replayed log) must not reorder nondeterministically + // in a function that promises to be pure. + slices.SortStableFunc(rulePolls, func(a, b Poll) int { return a.GrafanaNow.Compare(b.GrafanaNow) }) + + var res CoverageResult + fail := func(reason UnobservableReason, note string) { + res.Unobservable = true + if res.Reason == "" { + res.Reason = reason + } + res.Notes = append(res.Notes, fmt.Sprintf("rule %q: %s", def.Title, note)) + } + + // Check 1 — sentinel (§4.5). Present and At >= to+grace -> coverage + // provable; absent, or short of it, is never a pass. A recorder that died + // early must look exactly like a coverage gap, because it is one. + switch { + case sentinel == nil: + fail(ReasonNoSentinel, "no stopped sentinel: the recorder never reported finishing") + case sentinel.Before(windowEnd): + fail(ReasonSentinelEarly, fmt.Sprintf("stopped sentinel at %s is before the required %s (to+grace)", + sentinel.Format(time.RFC3339), windowEnd.Format(time.RFC3339))) + } + + // Check 2 — from bounds (§7), first sentence only: from < StartedAt makes + // coverage unprovable, no matter how healthy the polls that DO exist look. + // Both are runner-domain clock reads (the recorder's own Clock.Now()), so + // no cross-domain translation applies here. The second sentence — from + // more than fromFutureTolerance ahead is a hard error — is Check's input + // validation, once per run rather than per rule, and belongs to a later + // phase: this function has no error return, only a per-rule verdict. + if from.Before(h.StartedAt) { + fail(ReasonFromBeforeRecord, fmt.Sprintf( + "requested from %s is before recording started at %s", from.Format(time.RFC3339), h.StartedAt.Format(time.RFC3339))) + } + + // Filtered once, here, and threaded through every remaining check — + // ruleHeartbeatGap included — rather than re-filtered per check: two + // independent filters over the same polls would only invite one of them + // drifting from the other's membership test. + inWindow := inWindowPolls(rulePolls, from, windowEnd) + + // Check 3 — heartbeat continuity (§6). Data at both ends with a hole in + // between is not enough (§22.4): this scans every gap inside the window, + // not just its edges. + res.LargestGap, res.LargestGapAt = ruleHeartbeatGap(inWindow, from, windowEnd) + if res.LargestGap > t.maxGap { + fail(ReasonHeartbeatGap, fmt.Sprintf( + "gap of %s starting at %s exceeds maxGap %s", res.LargestGap, res.LargestGapAt.Format(time.RFC3339), t.maxGap)) + } + + // Check 4 — health=="error" (§10.1). A short blip is a note only (§22.1: + // one failed evaluation must not exit 2 over an otherwise clean window); + // only a run longer than healthGrace consumes coverage. + if runLen, sawAny := longestHealthRun(inWindow, "error"); sawAny { + res.Notes = append(res.Notes, fmt.Sprintf("rule %q: health=error observed (longest run %s)", def.Title, runLen)) + if runLen > t.healthGrace { + fail(ReasonHealthError, fmt.Sprintf("health=error for %s exceeds healthGrace %s", runLen, t.healthGrace)) + } + } + + // Check 5 — health=="nodata" (§10.1/§10.2). Never fatal here: 96% of the + // fleet runs no_data_state:OK, so treating this as fatal by default would + // block nearly every healthy deploy in an idle environment. Escalating it + // under Policy.NodataIsUnobservable is decide's job (a later phase), + // applied directly against the raw polls — this pure function has no + // Policy to consult and must not invent one. + if _, sawAny := longestHealthRun(inWindow, "nodata"); sawAny { + res.Notes = append(res.Notes, fmt.Sprintf("rule %q: health=nodata observed (not fatal; see --nodata-is-unobservable)", def.Title)) + } + + // Check 6 — liveness (H3). Absolute only, per poll: GrafanaNow and + // LastEvaluation are both Grafana-domain reads off the SAME response, so + // this is a same-domain comparison and uses raw values — never a delta + // against a previous poll, which reports stale on ~half the polls of a + // perfectly healthy rule (polling runs at intervalSeconds/2). + // + // Skipped only for a poll whose own flags SAY there is nothing to check: + // IsPaused (a zero LastEvaluation is legal only while paused, §2.3; check + // 7 is its detector) or !Found (no rule, no evaluation; check 8 is its + // detector). Deliberately NOT skipped merely because LastEvaluation is + // zero: ReadLog does no field validation, so a corrupted or hand-edited + // log line can claim found:true, is_paused:false and still carry a zero + // LastEvaluation, and that combination must read as maximally stale + // rather than being silently waved through. + var staleCount int + var worstStale time.Duration + var worstStaleAt time.Time + for _, p := range inWindow { + if p.IsPaused || !p.Found { + continue + } + if stale := p.GrafanaNow.Sub(p.LastEvaluation); stale > t.evalStaleAfter { + staleCount++ + if stale > worstStale { + worstStale, worstStaleAt = stale, p.GrafanaNow + } + } + } + if staleCount > 0 { + res.BlindFor = worstStale + fail(ReasonStaleEvaluation, fmt.Sprintf( + "lastEvaluation stale on %d poll(s); worst %s (> evalStaleAfter %s) as of %s", + staleCount, worstStale, t.evalStaleAfter, worstStaleAt.Format(time.RFC3339))) + } + + // Check 7 — isPaused in-window (§12.2, §14.8). The PRIMARY pause + // detector: liveness (check 6) is only the backup for what IsPaused + // cannot show (a deleted rule, a stopped scheduler, a blocked + // evaluation). This is what catches pause-then-unpause, which the drain + // wait alone passes (§14.7). + var pausedCount int + var pausedAt time.Time + for _, p := range inWindow { + if p.IsPaused { + pausedCount++ + if pausedAt.IsZero() { + pausedAt = p.GrafanaNow + } + } + } + if pausedCount > 0 { + fail(ReasonPausedInWindow, fmt.Sprintf("observed paused on %d poll(s), first at %s", pausedCount, pausedAt.Format(time.RFC3339))) + } + + // Check 8 — rule absent (§14.5). Found==false is authoritative (P2 + // already retried every transport failure before a Poll record ever + // exists): the rule resolved at resolve time but the state endpoint + // stopped serving it. Never drop a watched rule from the verdict set + // silently. + var absentCount int + var absentAt time.Time + for _, p := range inWindow { + if !p.Found { + absentCount++ + if absentAt.IsZero() { + absentAt = p.GrafanaNow + } + } + } + if absentCount > 0 { + fail(ReasonRuleAbsent, fmt.Sprintf("state endpoint returned no rule on %d poll(s), first at %s", absentCount, absentAt.Format(time.RFC3339))) + } + + // Check 9 — KeepLast (§10.2). A note, never fatal. It surfaces only as an + // instance Reason after P1.2a's parsing, and Reasons keys can be + // comma-joined composites, so membership (reasonsContain) is required — + // indexing "KeepLast" directly would miss "KeepLast, MissingSeries". + for _, p := range inWindow { + if reasonsContain(p.Reasons, keepLastReason) { + res.Notes = append(res.Notes, fmt.Sprintf( + "rule %q: KeepLast observed at %s: a held-over state may hide a real blind spot", def.Title, p.GrafanaNow.Format(time.RFC3339))) + break + } + } + + res.Proved = !res.Unobservable + return res +} + +// inWindowPolls filters polls to those inside [from, windowEnd] using the +// CROSS-DOMAIN membership test (§16): each poll's Grafana-domain GrafanaNow +// is translated to the runner domain by its OWN skew, and its own skew bound +// is the membership tolerance, so a poll that is genuinely inside the window +// is never excluded by ordinary clock imprecision. +// +// Everything downstream of this filter (health runs, liveness, pause, +// absence) reads the poll's raw fields: GrafanaNow paired with +// LastEvaluation on the SAME response, or one poll's GrafanaNow against the +// next's, are same-domain comparisons and need no translation (§16, "Clock +// domains" — only window membership and check 3's two boundary segments do). +func inWindowPolls(polls []Poll, from, windowEnd time.Time) []Poll { + var out []Poll + for _, p := range polls { + bound := p.SkewBound() + runner := p.GrafanaNow.Add(-p.Skew()) + if runner.Before(from.Add(-bound)) || runner.After(windowEnd.Add(bound)) { + continue + } + out = append(out, p) + } + return out +} + +// ruleHeartbeatGap finds the largest unobserved span inside [from, windowEnd] +// (§6), including the two boundary segments — which is why "data at both +// ends with a hole in the middle" still fails (§22.4): the segment between +// the polls just inside each edge is exactly what this measures. in must +// already be filtered to this window (inWindowPolls) and sorted by +// GrafanaNow — proveCoverage computes that filter once and threads it through +// every check, this one included, rather than each check re-filtering. +// +// The two boundary segments compare a Grafana-domain poll time against the +// runner-domain from/windowEnd, so each is translated by its own poll's skew +// AND widened by that same poll's skew bound (§16: "with that poll's bound as +// the tolerance") — on the side that makes the segment larger, never smaller, +// so an uncertain boundary reads as at least as big a gap as it might really +// be. Understating it by up to the bound would be fail-open. The spacing +// BETWEEN consecutive polls compares two Grafana-domain reads to each other — +// same domain — and uses the raw GrafanaNow difference, no bound needed. +func ruleHeartbeatGap(in []Poll, from, windowEnd time.Time) (largestGap time.Duration, largestGapAt time.Time) { + if len(in) == 0 { + return windowEnd.Sub(from), from + } + + runnerOf := func(p Poll) time.Time { return p.GrafanaNow.Add(-p.Skew()) } + + first := in[0] + if gap := runnerOf(first).Sub(from) + first.SkewBound(); gap > largestGap { + largestGap, largestGapAt = gap, from + } + for i := 1; i < len(in); i++ { + if gap := in[i].GrafanaNow.Sub(in[i-1].GrafanaNow); gap > largestGap { + largestGap, largestGapAt = gap, runnerOf(in[i-1]) + } + } + last := in[len(in)-1] + if gap := windowEnd.Sub(runnerOf(last)) + last.SkewBound(); gap > largestGap { + largestGap, largestGapAt = gap, runnerOf(last) + } + return largestGap, largestGapAt +} + +// longestHealthRun returns the longest contiguous wall-clock span (§10.1) +// during which polls — already sorted by GrafanaNow, same-domain spacing +// (§16) — read the given rule-level Health, and whether any poll matched it +// at all. +// +// It detects the span as it accumulates rather than waiting for the run to +// end, so an open-ended run that is still failing at the last poll in the +// window is measured correctly without needing data past the window: waiting +// for the run to "end" would have to assume the best case about what happens +// next, which is exactly what this gate must not do (§1). +func longestHealthRun(polls []Poll, health string) (longest time.Duration, sawAny bool) { + var runStart time.Time + for _, p := range polls { + if p.Health != health { + runStart = time.Time{} + continue + } + sawAny = true + if runStart.IsZero() { + runStart = p.GrafanaNow + } + if span := p.GrafanaNow.Sub(runStart); span > longest { + longest = span + } + } + return longest, sawAny +} + +// reasonsContain reports whether any key of reasons names want, honoring +// Grafana's comma-joined composite reason strings via reasonNames (log.go). +func reasonsContain(reasons map[string]int, want string) bool { + for reason := range reasons { + if reasonNames(reason, want) { + return true + } + } + return false +} diff --git a/grafana-alertcheck/internal/gate/coverage_test.go b/grafana-alertcheck/internal/gate/coverage_test.go new file mode 100644 index 000000000..429139602 --- /dev/null +++ b/grafana-alertcheck/internal/gate/coverage_test.go @@ -0,0 +1,628 @@ +package gate + +import ( + "strings" + "testing" + "time" +) + +func TestProveCoverage_CleanWindowIsProved(t *testing.T) { + from := time.Date(2026, 1, 1, 0, 0, 0, 0, time.UTC) + to := from.Add(10 * time.Minute) + rt := newRuleTimings(30*time.Second, 60) + def := Definition{UID: "r1", Title: "R1"} + + var polls []Poll + for ts := from; !ts.After(to); ts = ts.Add(30 * time.Second) { + polls = append(polls, Poll{RuleUID: "r1", GrafanaNow: ts, Found: true, Health: "ok", State: "inactive", LastEvaluation: ts}) + } + sentinel := to + + res := proveCoverage(Header{StartedAt: from.Add(-time.Hour)}, polls, &sentinel, rt, def, from, to, 0) + if !res.Proved || res.Unobservable || res.Reason != "" { + t.Fatalf("res = %+v, want a clean proved window", res) + } +} + +func TestProveCoverage_FiltersPollsByUID(t *testing.T) { + from := time.Date(2026, 1, 1, 0, 0, 0, 0, time.UTC) + to := from.Add(10 * time.Minute) + rt := newRuleTimings(30*time.Second, 60) + def := Definition{UID: "r1", Title: "R1"} + + var polls []Poll + for ts := from; !ts.After(to); ts = ts.Add(30 * time.Second) { + polls = append(polls, Poll{RuleUID: "r1", GrafanaNow: ts, Found: true, Health: "ok", LastEvaluation: ts}) + // A different rule's polls, deliberately broken, must never + // contaminate r1's verdict: proveCoverage selects by UID itself. + polls = append(polls, Poll{RuleUID: "other", GrafanaNow: ts, Found: false}) + } + sentinel := to + + res := proveCoverage(Header{StartedAt: from.Add(-time.Hour)}, polls, &sentinel, rt, def, from, to, 0) + if !res.Proved { + t.Fatalf("res = %+v, want proved: a different rule's broken polls must not affect this rule's verdict", res) + } +} + +// --- Check 1: sentinel (§4.5) --- + +func TestProveCoverage_NoSentinelIsUnobservable(t *testing.T) { + from := time.Date(2026, 1, 1, 0, 0, 0, 0, time.UTC) + to := from.Add(time.Minute) + rt := newRuleTimings(30*time.Second, 60) + def := Definition{UID: "r1", Title: "R1"} + + res := proveCoverage(Header{StartedAt: from.Add(-time.Hour)}, nil, nil, rt, def, from, to, 0) + if res.Proved || res.Reason != ReasonNoSentinel { + t.Fatalf("res = %+v, want unobservable/no_sentinel: an absent sentinel must never be a pass", res) + } +} + +func TestProveCoverage_SentinelBeforeGraceIsUnobservable(t *testing.T) { + from := time.Date(2026, 1, 1, 0, 0, 0, 0, time.UTC) + to := from.Add(10 * time.Minute) + grace := 2 * time.Minute + rt := newRuleTimings(30*time.Second, 60) + def := Definition{UID: "r1", Title: "R1"} + + sentinel := to.Add(grace).Add(-time.Second) // one second short of to+grace + res := proveCoverage(Header{StartedAt: from.Add(-time.Hour)}, nil, &sentinel, rt, def, from, to, grace) + if res.Reason != ReasonSentinelEarly { + t.Fatalf("Reason = %q, want sentinel_early", res.Reason) + } +} + +func TestProveCoverage_SentinelExactlyAtGraceIsFine(t *testing.T) { + from := time.Date(2026, 1, 1, 0, 0, 0, 0, time.UTC) + to := from.Add(10 * time.Minute) + grace := 2 * time.Minute + windowEnd := to.Add(grace) + rt := newRuleTimings(30*time.Second, 60) + def := Definition{UID: "r1", Title: "R1"} + + var polls []Poll + for ts := from; !ts.After(windowEnd); ts = ts.Add(30 * time.Second) { + polls = append(polls, Poll{RuleUID: "r1", GrafanaNow: ts, Found: true, Health: "ok", LastEvaluation: ts}) + } + sentinel := windowEnd + + res := proveCoverage(Header{StartedAt: from.Add(-time.Hour)}, polls, &sentinel, rt, def, from, to, grace) + if !res.Proved { + t.Fatalf("Proved = false, want true: sentinel exactly at to+grace must satisfy check 1: %+v", res) + } +} + +// --- Check 2: from bounds (§7) --- + +func TestProveCoverage_FromBeforeRecordIsUnobservable(t *testing.T) { + started := time.Date(2026, 1, 1, 1, 0, 0, 0, time.UTC) + from := started.Add(-time.Minute) // the requested window opens before recording started + to := from.Add(10 * time.Minute) + rt := newRuleTimings(30*time.Second, 60) + def := Definition{UID: "r1", Title: "R1"} + + sentinel := to + res := proveCoverage(Header{StartedAt: started}, nil, &sentinel, rt, def, from, to, 0) + if res.Reason != ReasonFromBeforeRecord { + t.Fatalf("Reason = %q, want from_before_record", res.Reason) + } +} + +// --- Check 3: heartbeat continuity (§6) --- + +// TestProveCoverage_HeartbeatGapBetweenBoundariesIsUnobservable is §22.4's +// core regression: data at both ends with a hole between is not enough. +func TestProveCoverage_HeartbeatGapBetweenBoundariesIsUnobservable(t *testing.T) { + from := time.Date(2026, 1, 1, 0, 0, 0, 0, time.UTC) + to := from.Add(10 * time.Minute) + rt := newRuleTimings(30*time.Second, 60) // maxGap = 60s + def := Definition{UID: "r1", Title: "R1"} + + polls := []Poll{ + {RuleUID: "r1", GrafanaNow: from.Add(time.Second), Found: true, Health: "ok", LastEvaluation: from}, + {RuleUID: "r1", GrafanaNow: to.Add(-time.Second), Found: true, Health: "ok", LastEvaluation: to}, + } + sentinel := to + + res := proveCoverage(Header{StartedAt: from.Add(-time.Hour)}, polls, &sentinel, rt, def, from, to, 0) + if res.Reason != ReasonHeartbeatGap { + t.Fatalf("Reason = %q, want heartbeat_gap: healthy edges with a hole in the middle must still fail (§22.4)", res.Reason) + } + // The gap is the SPACING between the two polls (598s), not either + // boundary segment (1s each) — pin the actual values, not just the verdict. + if res.LargestGap != 598*time.Second { + t.Fatalf("LargestGap = %s, want 598s (the spacing between the two polls, not a boundary segment)", res.LargestGap) + } + wantAt := from.Add(time.Second) + if !res.LargestGapAt.Equal(wantAt) { + t.Fatalf("LargestGapAt = %s, want %s (where the gap starts, at the first poll)", res.LargestGapAt, wantAt) + } +} + +// --- Check 4/5: health (§10.1/§10.2) --- + +func TestProveCoverage_HealthErrorShortBlipPassesWithNote(t *testing.T) { + from := time.Date(2026, 1, 1, 0, 0, 0, 0, time.UTC) + to := from.Add(10 * time.Minute) + rt := newRuleTimings(30*time.Second, 60) // healthGrace = 60s + def := Definition{UID: "r1", Title: "R1"} + + var polls []Poll + blip := from.Add(2 * time.Minute) + for ts := from; !ts.After(to); ts = ts.Add(30 * time.Second) { + health := "ok" + if ts.Equal(blip) { + health = "error" // one isolated failed evaluation + } + polls = append(polls, Poll{RuleUID: "r1", GrafanaNow: ts, Found: true, Health: health, LastEvaluation: ts}) + } + sentinel := to + + res := proveCoverage(Header{StartedAt: from.Add(-time.Hour)}, polls, &sentinel, rt, def, from, to, 0) + if !res.Proved { + t.Fatalf("Proved = false, want true: one failed evaluation must not fail an otherwise clean window (§22.1): %+v", res) + } + if !anyContains(res.Notes, "health=error") { + t.Fatalf("Notes = %v, want a health=error note even though it did not fail the window", res.Notes) + } +} + +func TestProveCoverage_HealthErrorSustainedIsUnobservable(t *testing.T) { + from := time.Date(2026, 1, 1, 0, 0, 0, 0, time.UTC) + to := from.Add(10 * time.Minute) + rt := newRuleTimings(30*time.Second, 60) // healthGrace = 60s + def := Definition{UID: "r1", Title: "R1"} + + runStart, runEnd := from.Add(2*time.Minute), from.Add(5*time.Minute) // a 3-minute run, well past healthGrace + var polls []Poll + for ts := from; !ts.After(to); ts = ts.Add(30 * time.Second) { + health := "ok" + if !ts.Before(runStart) && !ts.After(runEnd) { + health = "error" + } + polls = append(polls, Poll{RuleUID: "r1", GrafanaNow: ts, Found: true, Health: health, LastEvaluation: ts}) + } + sentinel := to + + res := proveCoverage(Header{StartedAt: from.Add(-time.Hour)}, polls, &sentinel, rt, def, from, to, 0) + if res.Reason != ReasonHealthError { + t.Fatalf("Reason = %q, want health_error for a run that outlasts healthGrace", res.Reason) + } +} + +func TestProveCoverage_HealthNodataNeverFatalHere(t *testing.T) { + from := time.Date(2026, 1, 1, 0, 0, 0, 0, time.UTC) + to := from.Add(10 * time.Minute) + rt := newRuleTimings(30*time.Second, 60) + def := Definition{UID: "r1", Title: "R1"} + + var polls []Poll + for ts := from; !ts.After(to); ts = ts.Add(30 * time.Second) { + polls = append(polls, Poll{RuleUID: "r1", GrafanaNow: ts, Found: true, Health: "nodata", LastEvaluation: ts}) + } + sentinel := to + + res := proveCoverage(Header{StartedAt: from.Add(-time.Hour)}, polls, &sentinel, rt, def, from, to, 0) + if !res.Proved { + t.Fatalf("Proved = false, want true: health=nodata for the WHOLE window must still not be fatal by itself "+ + "(escalating it is Policy.NodataIsUnobservable's job, applied by decide in a later phase): %+v", res) + } + if !anyContains(res.Notes, "health=nodata") { + t.Fatalf("Notes = %v, want a health=nodata note", res.Notes) + } +} + +// --- Check 6: liveness / H3 --- + +// TestProveCoverage_LivenessAbsoluteNeverFalseStale is §22.7's disproportionate +// test: a healthy rule polled at intervalSeconds/2, across the full window, +// must show zero staleness violations. lastEvaluation only advances once per +// full evaluation interval here — the realistic shape a delta check +// misreads as stale on roughly half of all polls (H3). +func TestProveCoverage_LivenessAbsoluteNeverFalseStale(t *testing.T) { + from := time.Date(2026, 1, 1, 0, 0, 0, 0, time.UTC) + pollEvery := 30 * time.Second + intervalSeconds := 60 + windowEnd := from.Add(10 * time.Minute) + rt := newRuleTimings(pollEvery, intervalSeconds) + def := Definition{UID: "r1", Title: "R1"} + + var polls []Poll + lastEval := from + for ts := from; !ts.After(windowEnd); ts = ts.Add(pollEvery) { + if ts.Sub(lastEval) >= time.Duration(intervalSeconds)*time.Second { + lastEval = ts + } + polls = append(polls, Poll{RuleUID: "r1", GrafanaNow: ts, Found: true, Health: "ok", LastEvaluation: lastEval}) + } + sentinel := windowEnd + + res := proveCoverage(Header{StartedAt: from.Add(-time.Hour)}, polls, &sentinel, rt, def, from, windowEnd, 0) + if res.Reason == ReasonStaleEvaluation || res.BlindFor != 0 { + t.Fatalf("proveCoverage flagged staleness on a healthy rule polled at intervalSeconds/2 — H3 must be absolute, "+ + "never a delta against a previous poll: %+v", res) + } + if !res.Proved { + t.Fatalf("Proved = false, want true: %+v (notes: %v)", res, res.Notes) + } +} + +func TestProveCoverage_StaleEvaluationIsUnobservable(t *testing.T) { + from := time.Date(2026, 1, 1, 0, 0, 0, 0, time.UTC) + to := from.Add(10 * time.Minute) + rt := newRuleTimings(30*time.Second, 60) // evalStaleAfter = 120s + def := Definition{UID: "r1", Title: "R1"} + + // Dense, otherwise-healthy polling so heartbeat continuity (check 3) + // stays intact — only check 6 should be able to fire. + polls := denseHealthyPolls("r1", from, to, 30*time.Second) + staleAt := from.Add(5 * time.Minute) + for i := range polls { + if polls[i].GrafanaNow.Equal(staleAt) { + polls[i].LastEvaluation = staleAt.Add(-3 * time.Minute) + } + } + sentinel := to + + res := proveCoverage(Header{StartedAt: from.Add(-time.Hour)}, polls, &sentinel, rt, def, from, to, 0) + if res.Reason != ReasonStaleEvaluation { + t.Fatalf("Reason = %q, want stale_evaluation", res.Reason) + } + if res.BlindFor != 3*time.Minute { + t.Fatalf("BlindFor = %s, want 3m", res.BlindFor) + } +} + +func TestProveCoverage_ZeroLastEvaluationNeverFalseStale(t *testing.T) { + from := time.Date(2026, 1, 1, 0, 0, 0, 0, time.UTC) + to := from.Add(10 * time.Minute) + rt := newRuleTimings(30*time.Second, 60) + def := Definition{UID: "r1", Title: "R1"} + + // A paused rule legitimately reports the zero time (§2.3); check 6 must + // not read that as an enormous staleness violation. Check 7 is its + // detector. + polls := []Poll{ + {RuleUID: "r1", GrafanaNow: from.Add(time.Minute), Found: true, IsPaused: true}, + } + sentinel := to + + res := proveCoverage(Header{StartedAt: from.Add(-time.Hour)}, polls, &sentinel, rt, def, from, to, 0) + if res.Reason == ReasonStaleEvaluation { + t.Fatalf("a zero lastEvaluation on a paused poll must not trigger check 6: %+v", res) + } +} + +// --- Check 7: isPaused in-window (§12.2, §14.8) --- + +func TestProveCoverage_PausedInWindowIsUnobservable(t *testing.T) { + from := time.Date(2026, 1, 1, 0, 0, 0, 0, time.UTC) + to := from.Add(10 * time.Minute) + rt := newRuleTimings(30*time.Second, 60) + def := Definition{UID: "r1", Title: "R1"} + + // Dense, otherwise-healthy polling so heartbeat continuity (check 3) + // stays intact — only check 7 should be able to fire. + polls := denseHealthyPolls("r1", from, to, 30*time.Second) + pausedAt := from.Add(5 * time.Minute) + for i := range polls { + if polls[i].GrafanaNow.Equal(pausedAt) { + polls[i].IsPaused = true + polls[i].LastEvaluation = time.Time{} // legal only while paused, §2.3 + } + } + sentinel := to + + res := proveCoverage(Header{StartedAt: from.Add(-time.Hour)}, polls, &sentinel, rt, def, from, to, 0) + if res.Reason != ReasonPausedInWindow { + t.Fatalf("Reason = %q, want paused_in_window", res.Reason) + } +} + +// --- Check 8: rule absent (§14.5) --- + +func TestProveCoverage_RuleAbsentIsUnobservable(t *testing.T) { + from := time.Date(2026, 1, 1, 0, 0, 0, 0, time.UTC) + to := from.Add(10 * time.Minute) + rt := newRuleTimings(30*time.Second, 60) + def := Definition{UID: "r1", Title: "R1"} + + // Dense, otherwise-healthy polling so heartbeat continuity (check 3) + // stays intact — only check 8 should be able to fire. + polls := denseHealthyPolls("r1", from, to, 30*time.Second) + absentAt := from.Add(5 * time.Minute) + for i := range polls { + if polls[i].GrafanaNow.Equal(absentAt) { + polls[i].Found = false + polls[i].Health = "" + polls[i].LastEvaluation = time.Time{} + } + } + sentinel := to + + res := proveCoverage(Header{StartedAt: from.Add(-time.Hour)}, polls, &sentinel, rt, def, from, to, 0) + if res.Reason != ReasonRuleAbsent { + t.Fatalf("Reason = %q, want rule_absent", res.Reason) + } +} + +// denseHealthyPolls builds a clean poll sequence at a fixed cadence, with +// zero staleness and nothing abnormal — the baseline the single-check tests +// mutate exactly one poll of, so heartbeat continuity (check 3) never +// confounds the check under test. +func denseHealthyPolls(uid string, from, to time.Time, every time.Duration) []Poll { + var out []Poll + for ts := from; !ts.After(to); ts = ts.Add(every) { + out = append(out, Poll{RuleUID: uid, GrafanaNow: ts, Found: true, Health: "ok", LastEvaluation: ts}) + } + return out +} + +// --- Check 9: KeepLast (§10.2) --- + +func TestProveCoverage_KeepLastIsNoteOnly(t *testing.T) { + from := time.Date(2026, 1, 1, 0, 0, 0, 0, time.UTC) + to := from.Add(10 * time.Minute) + rt := newRuleTimings(30*time.Second, 60) + def := Definition{UID: "r1", Title: "R1"} + + var polls []Poll + for ts := from; !ts.After(to); ts = ts.Add(30 * time.Second) { + polls = append(polls, Poll{ + RuleUID: "r1", GrafanaNow: ts, Found: true, Health: "ok", LastEvaluation: ts, + // A comma-joined composite — reasonsContain must match by + // membership, never by an exact key, per P5's markers. + Reasons: map[string]int{"KeepLast, MissingSeries": 1}, + }) + } + sentinel := to + + res := proveCoverage(Header{StartedAt: from.Add(-time.Hour)}, polls, &sentinel, rt, def, from, to, 0) + if !res.Proved { + t.Fatalf("Proved = false, want true: KeepLast is a note, never fatal: %+v", res) + } + if !anyContains(res.Notes, "KeepLast") { + t.Fatalf("Notes = %v, want a KeepLast note (comma-joined membership, not a literal-key match)", res.Notes) + } +} + +// --- Clock domains (§16) --- + +// TestProveCoverage_SkewTranslationAtWindowBoundary pins §16's "Clock +// domains" rule: a constant clock skew on every poll must not itself read as +// a coverage gap or a from-before-record violation, because every +// cross-domain comparison translates by that poll's own skew first. +func TestProveCoverage_SkewTranslationAtWindowBoundary(t *testing.T) { + from := time.Date(2026, 1, 1, 0, 0, 0, 0, time.UTC) + to := from.Add(10 * time.Minute) + rt := newRuleTimings(30*time.Second, 60) + def := Definition{UID: "r1", Title: "R1"} + + const skew = 45 * time.Second // Grafana's clock reads 45s ahead of the runner's + const bound = 5 * time.Second + + var polls []Poll + for ts := from; !ts.After(to); ts = ts.Add(30 * time.Second) { + grafanaTime := ts.Add(skew) + polls = append(polls, Poll{ + RuleUID: "r1", GrafanaNow: grafanaTime, SkewMS: skew.Milliseconds(), SkewBoundMS: bound.Milliseconds(), + Found: true, Health: "ok", LastEvaluation: grafanaTime, + }) + } + sentinel := to + + res := proveCoverage(Header{StartedAt: from.Add(-time.Hour)}, polls, &sentinel, rt, def, from, to, 0) + if !res.Proved { + t.Fatalf("res = %+v, want proved: a constant clock skew must not itself read as a coverage gap (§16)", res) + } +} + +// --- Override round-trip (P5's "two authorities") --- + +// TestProveCoverage_OverrideRoundTrip is P7's other disproportionate done-gate +// test: it exercises DeriveTimingsFromLog and proveCoverage together, exactly +// as check will, to prove maxGap tracks the RECORDED cadence, never a +// re-derivation from the rule's own evaluation interval. +func TestProveCoverage_OverrideRoundTrip(t *testing.T) { + from := time.Date(2026, 1, 1, 0, 0, 0, 0, time.UTC) + + t.Run("slower override on a tighter rule classifies clean", func(t *testing.T) { + windowEnd := from.Add(10 * time.Minute) + h := Header{ + StartedAt: from.Add(-time.Hour), + Rules: []LoggedRule{{UID: "r1", Title: "R1", IntervalSeconds: 60, PollEverySeconds: 120}}, + } + defs := []Definition{{UID: "r1", Title: "R1", IntervalSeconds: 60}} + rt, _, err := DeriveTimingsFromLog(h, defs) + if err != nil { + t.Fatalf("DeriveTimingsFromLog: %v", err) + } + + var polls []Poll + for ts := from; !ts.After(windowEnd); ts = ts.Add(120 * time.Second) { + polls = append(polls, Poll{RuleUID: "r1", GrafanaNow: ts, Found: true, Health: "ok", LastEvaluation: ts}) + } + sentinel := windowEnd + + res := proveCoverage(h, polls, &sentinel, rt["r1"], defs[0], from, windowEnd, 0) + if !res.Proved { + t.Fatalf("Proved = false, want true (maxGap must come from the recorded 120s cadence, not the 30s default): %+v", res) + } + }) + + t.Run("faster override still catches a real recorder gap", func(t *testing.T) { + windowEnd := from.Add(20 * time.Minute) + h := Header{ + StartedAt: from.Add(-time.Hour), + Rules: []LoggedRule{{UID: "r1", Title: "R1", IntervalSeconds: 300, PollEverySeconds: 5}}, + } + defs := []Definition{{UID: "r1", Title: "R1", IntervalSeconds: 300}} + rt, _, err := DeriveTimingsFromLog(h, defs) + if err != nil { + t.Fatalf("DeriveTimingsFromLog: %v", err) + } + + var polls []Poll + ts := from + for range 20 { + polls = append(polls, Poll{RuleUID: "r1", GrafanaNow: ts, Found: true, Health: "ok", LastEvaluation: ts}) + ts = ts.Add(5 * time.Second) + } + // The one real gap: 250s, nowhere near this recording's actual 5s + // cadence. Resume 5s polling afterward all the way to windowEnd, so + // this hole is the ONLY gap in the window — otherwise an uncovered + // tail would exceed even the WRONG (definition-derived) 300s maxGap + // on its own, and the test could not tell the two derivations apart. + ts = ts.Add(250 * time.Second) + for !ts.After(windowEnd) { + polls = append(polls, Poll{RuleUID: "r1", GrafanaNow: ts, Found: true, Health: "ok", LastEvaluation: ts}) + ts = ts.Add(5 * time.Second) + } + sentinel := windowEnd + + res := proveCoverage(h, polls, &sentinel, rt["r1"], defs[0], from, windowEnd, 0) + if res.Reason != ReasonHeartbeatGap { + t.Fatalf("Reason = %q, want heartbeat_gap: if maxGap had been re-derived from the 300s definition instead of "+ + "the recorded 5s cadence, this 250s gap would pass silently — the fail-open direction P5 warns about", res.Reason) + } + }) +} + +func anyContains(notes []string, substr string) bool { + for _, n := range notes { + if strings.Contains(n, substr) { + return true + } + } + return false +} + +// --- Check 6, tightened: a corrupted log must not silently disable liveness --- + +// TestProveCoverage_ZeroLastEvaluationWithoutPauseIsStale guards check 6's +// skip condition. ReadLog does no field validation, so a log line can claim +// found:true, is_paused:false and still carry a zero LastEvaluation (a +// corrupted write, a hand-edited fixture, a future log format bug). That +// combination must read as maximally stale, not be waved through the way a +// legitimately paused poll's zero time is (§2.3) — the skip must key off +// IsPaused/Found, never off LastEvaluation being zero. +func TestProveCoverage_ZeroLastEvaluationWithoutPauseIsStale(t *testing.T) { + from := time.Date(2026, 1, 1, 0, 0, 0, 0, time.UTC) + to := from.Add(10 * time.Minute) + rt := newRuleTimings(30*time.Second, 60) + def := Definition{UID: "r1", Title: "R1"} + + polls := denseHealthyPolls("r1", from, to, 30*time.Second) + corruptAt := from.Add(5 * time.Minute) + for i := range polls { + if polls[i].GrafanaNow.Equal(corruptAt) { + polls[i].LastEvaluation = time.Time{} // found:true, is_paused:false, yet zero + } + } + sentinel := to + + res := proveCoverage(Header{StartedAt: from.Add(-time.Hour)}, polls, &sentinel, rt, def, from, to, 0) + if res.Reason != ReasonStaleEvaluation { + t.Fatalf("Reason = %q, want stale_evaluation: a zero lastEvaluation on a found, non-paused poll must fail "+ + "closed, not be silently skipped as if it were a legitimately paused observation", res.Reason) + } +} + +// --- Check 3, tightened: the boundary segments must widen by the skew bound --- + +// TestProveCoverage_BoundaryGapWidensBySkewBound pins §16's "with that +// poll's bound as the tolerance" for the two boundary segments specifically: +// a boundary gap that lands EXACTLY at maxGap must still fail once the +// poll's own skew bound is added, because the translation is only a best +// estimate and understating the gap by up to the bound would be fail-open. +func TestProveCoverage_BoundaryGapWidensBySkewBound(t *testing.T) { + from := time.Date(2026, 1, 1, 0, 0, 0, 0, time.UTC) + to := from.Add(10 * time.Minute) + rt := newRuleTimings(30*time.Second, 60) // maxGap = 60s + def := Definition{UID: "r1", Title: "R1"} + + const bound = 5 * time.Second + first := Poll{ + RuleUID: "r1", GrafanaNow: from.Add(rt.maxGap), Found: true, Health: "ok", + LastEvaluation: from.Add(rt.maxGap), SkewBoundMS: bound.Milliseconds(), + } + rest := denseHealthyPolls("r1", from.Add(rt.maxGap+30*time.Second), to, 30*time.Second) + polls := append([]Poll{first}, rest...) + sentinel := to + + res := proveCoverage(Header{StartedAt: from.Add(-time.Hour)}, polls, &sentinel, rt, def, from, to, 0) + if res.Reason != ReasonHeartbeatGap { + t.Fatalf("Reason = %q, want heartbeat_gap: the leading boundary segment sits at EXACTLY maxGap (60s) before "+ + "widening; the poll's own %s skew bound must push it past the threshold (§16), not just the skew translation", res.Reason, bound) + } +} + +// --- Multi-failure contract --- + +// TestProveCoverage_MultipleFailuresReasonIsFirstButAllNoted exercises two +// checks failing in the same rule: check 7 (paused in-window) precedes check +// 8 (rule absent) in the §5 order, so Reason must name the pause even though +// the rule also goes absent later — and the later failure must still add its +// own Note rather than being swallowed once Reason is set. +func TestProveCoverage_MultipleFailuresReasonIsFirstButAllNoted(t *testing.T) { + from := time.Date(2026, 1, 1, 0, 0, 0, 0, time.UTC) + to := from.Add(10 * time.Minute) + rt := newRuleTimings(30*time.Second, 60) + def := Definition{UID: "r1", Title: "R1"} + + polls := denseHealthyPolls("r1", from, to, 30*time.Second) + pausedAt := from.Add(3 * time.Minute) + absentAt := from.Add(6 * time.Minute) + for i := range polls { + switch { + case polls[i].GrafanaNow.Equal(pausedAt): + polls[i].IsPaused = true + polls[i].LastEvaluation = time.Time{} + case polls[i].GrafanaNow.Equal(absentAt): + polls[i].Found = false + polls[i].Health = "" + polls[i].LastEvaluation = time.Time{} + } + } + sentinel := to + + res := proveCoverage(Header{StartedAt: from.Add(-time.Hour)}, polls, &sentinel, rt, def, from, to, 0) + if res.Reason != ReasonPausedInWindow { + t.Fatalf("Reason = %q, want paused_in_window (the FIRST check to fail, in §5's order)", res.Reason) + } + if !anyContains(res.Notes, "paused") { + t.Fatalf("Notes = %v, want a note about the pause", res.Notes) + } + if !anyContains(res.Notes, "no rule") { + t.Fatalf("Notes = %v, want a note about the absence too — a later failure must still be recorded, "+ + "not swallowed once Reason is already set", res.Notes) + } +} + +// --- Skipped rules (P6/P8 obligation) --- + +// TestProveCoverage_SkippedRuleWithZeroPollsPinnedAsHeartbeatGap pins a known +// gap in this function's contract, not a bug in it: a rule paused BEFORE the +// window opened is never scheduled or polled (watch.go, §4.3), so it reaches +// proveCoverage with zero polls at all. proveCoverage has no notion of +// "skipped" — that classification belongs to the definitions +// (LoggedRule.IsPaused / Definition.IsPaused), never to the polls — so today +// it reports the whole window as one big heartbeat_gap instead. decide (P8) +// MUST read skipped status from the definitions and either skip calling this +// function for that rule entirely, or override this result — this test pins +// today's behavior so that review has something concrete to check against. +func TestProveCoverage_SkippedRuleWithZeroPollsPinnedAsHeartbeatGap(t *testing.T) { + from := time.Date(2026, 1, 1, 0, 0, 0, 0, time.UTC) + to := from.Add(10 * time.Minute) + rt := newRuleTimings(30*time.Second, 60) + def := Definition{UID: "r1", Title: "R1", IsPaused: true} + + sentinel := to + res := proveCoverage(Header{StartedAt: from.Add(-time.Hour)}, nil, &sentinel, rt, def, from, to, 0) + if res.Reason != ReasonHeartbeatGap { + t.Fatalf("Reason = %q, want heartbeat_gap (pinned, not the desired end state): proveCoverage has no "+ + "'skipped' concept, so decide (P8) must handle a skipped rule's classification itself, before or "+ + "instead of calling this function", res.Reason) + } +} From 2db19c7e219b64da13edf2db2590d58a2be2b886 Mon Sep 17 00:00:00 2001 From: Bartek Tofel Date: Wed, 2 Sep 2026 11:56:39 +0200 Subject: [PATCH 20/43] chore: enhance unit tests --- .../internal/gate/coverage_test.go | 27 +++++++++++++++++++ 1 file changed, 27 insertions(+) diff --git a/grafana-alertcheck/internal/gate/coverage_test.go b/grafana-alertcheck/internal/gate/coverage_test.go index 429139602..7ef4b5427 100644 --- a/grafana-alertcheck/internal/gate/coverage_test.go +++ b/grafana-alertcheck/internal/gate/coverage_test.go @@ -320,6 +320,33 @@ func TestProveCoverage_PausedInWindowIsUnobservable(t *testing.T) { } } +// TestProveCoverage_PausedAfterWindowIsFine pins check 7's respect for the +// window boundary: a poll that reports paused but lands beyond windowEnd (a +// rule paused only after THIS release window closed) is filtered out by +// inWindowPolls and must not fail the window. Without that filter, a pause in +// the next release's window would wrongly fail this one. +func TestProveCoverage_PausedAfterWindowIsFine(t *testing.T) { + from := time.Date(2026, 1, 1, 0, 0, 0, 0, time.UTC) + to := from.Add(10 * time.Minute) + rt := newRuleTimings(30*time.Second, 60) + def := Definition{UID: "r1", Title: "R1"} + + polls := denseHealthyPolls("r1", from, to, 30*time.Second) + polls = append(polls, Poll{ + RuleUID: "r1", GrafanaNow: to.Add(2 * time.Minute), + Found: true, Health: "ok", IsPaused: true, + }) + sentinel := to + + res := proveCoverage(Header{StartedAt: from.Add(-time.Hour)}, polls, &sentinel, rt, def, from, to, 0) + if res.Reason == ReasonPausedInWindow { + t.Fatalf("a paused poll after windowEnd tripped check 7: %+v", res.Notes) + } + if !res.Proved { + t.Fatalf("Proved = false, want a clean window: %+v", res.Notes) + } +} + // --- Check 8: rule absent (§14.5) --- func TestProveCoverage_RuleAbsentIsUnobservable(t *testing.T) { From 82dc549295c95eb2532fcb45d98e211a163e9cfe Mon Sep 17 00:00:00 2001 From: Bartek Tofel Date: Mon, 7 Sep 2026 16:36:05 +0200 Subject: [PATCH 21/43] chore: address code review comments --- grafana-alertcheck/internal/gate/coverage.go | 13 ++++++++++- .../internal/gate/coverage_test.go | 23 +++++++++++++++++++ .../internal/gate/parse_ruler_test.go | 5 +--- .../internal/gate/parse_state_test.go | 5 +--- .../internal/gate/schedule_test.go | 4 +--- 5 files changed, 38 insertions(+), 12 deletions(-) diff --git a/grafana-alertcheck/internal/gate/coverage.go b/grafana-alertcheck/internal/gate/coverage.go index 36d2c3dac..3d3704b90 100644 --- a/grafana-alertcheck/internal/gate/coverage.go +++ b/grafana-alertcheck/internal/gate/coverage.go @@ -38,6 +38,7 @@ const ( ReasonHeartbeatGap UnobservableReason = "heartbeat_gap" ReasonHealthError UnobservableReason = "health_error" ReasonStaleEvaluation UnobservableReason = "stale_evaluation" + ReasonFutureEvaluation UnobservableReason = "future_evaluation" ReasonPausedInWindow UnobservableReason = "paused_in_window" ReasonRuleAbsent UnobservableReason = "rule_absent" // ReasonDrainTimeout is set by check.go's drain wait (a later phase), @@ -181,6 +182,16 @@ func proveCoverage(h Header, polls []Poll, sentinel *time.Time, t ruleTimings, d if p.IsPaused || !p.Found { continue } + // lastEvaluation in the future of its own poll's grafana_now is + // corrupted or hand-edited data (ReadLog does no field validation); + // GrafanaNow-LastEvaluation would go negative and silently read as + // fresh — fail-open. Treat it as unobservable instead. + if p.LastEvaluation.After(p.GrafanaNow) { + fail(ReasonFutureEvaluation, fmt.Sprintf( + "lastEvaluation %s is after grafana_now %s (corrupted poll)", + p.LastEvaluation.Format(time.RFC3339), p.GrafanaNow.Format(time.RFC3339))) + continue + } if stale := p.GrafanaNow.Sub(p.LastEvaluation); stale > t.evalStaleAfter { staleCount++ if stale > worstStale { @@ -191,7 +202,7 @@ func proveCoverage(h Header, polls []Poll, sentinel *time.Time, t ruleTimings, d if staleCount > 0 { res.BlindFor = worstStale fail(ReasonStaleEvaluation, fmt.Sprintf( - "lastEvaluation stale on %d poll(s); worst %s (> evalStaleAfter %s) as of %s", + "lastEvaluation stale on %d poll(s); worst %s (> evalStaleAfter %s) as of grafana_now %s", staleCount, worstStale, t.evalStaleAfter, worstStaleAt.Format(time.RFC3339))) } diff --git a/grafana-alertcheck/internal/gate/coverage_test.go b/grafana-alertcheck/internal/gate/coverage_test.go index 7ef4b5427..6419a8b2e 100644 --- a/grafana-alertcheck/internal/gate/coverage_test.go +++ b/grafana-alertcheck/internal/gate/coverage_test.go @@ -556,6 +556,29 @@ func TestProveCoverage_ZeroLastEvaluationWithoutPauseIsStale(t *testing.T) { } } +// A lastEvaluation in the future of grafana_now (corrupted log) must fail closed. +func TestProveCoverage_FutureLastEvaluationIsUnobservable(t *testing.T) { + from := time.Date(2026, 1, 1, 0, 0, 0, 0, time.UTC) + to := from.Add(10 * time.Minute) + rt := newRuleTimings(30*time.Second, 60) + def := Definition{UID: "r1", Title: "R1"} + + polls := denseHealthyPolls("r1", from, to, 30*time.Second) + corruptAt := from.Add(5 * time.Minute) + for i := range polls { + if polls[i].GrafanaNow.Equal(corruptAt) { + polls[i].LastEvaluation = corruptAt.Add(2 * time.Minute) // in the future of its own grafana_now + } + } + sentinel := to + + res := proveCoverage(Header{StartedAt: from.Add(-time.Hour)}, polls, &sentinel, rt, def, from, to, 0) + if res.Reason != ReasonFutureEvaluation { + t.Fatalf("Reason = %q, want future_evaluation: a lastEvaluation in the future of grafana_now must fail "+ + "closed rather than read its negative staleness as fresh", res.Reason) + } +} + // --- Check 3, tightened: the boundary segments must widen by the skew bound --- // TestProveCoverage_BoundaryGapWidensBySkewBound pins §16's "with that diff --git a/grafana-alertcheck/internal/gate/parse_ruler_test.go b/grafana-alertcheck/internal/gate/parse_ruler_test.go index dda66b77e..3cf2e3261 100644 --- a/grafana-alertcheck/internal/gate/parse_ruler_test.go +++ b/grafana-alertcheck/internal/gate/parse_ruler_test.go @@ -115,10 +115,7 @@ func TestParseDefinitions_Recording(t *testing.T) { } } -// A datasource-managed rule with neither an "alert" nor a "record" name has no -// identity (its only name is the Prometheus rule name), and an empty Title -// would make P3's refusal-by-name unreachable. It must fail parsing, not hand -// back a silently unusable Definition. +// A datasource-managed rule with no alert/record name must fail parsing. func TestParseDefinitions_DatasourceManagedNoName(t *testing.T) { body := []byte(`{"ExampleMetrics":[{"name":"g","rules":[{"expr":"up == 0","for":"5m"}]}]}`) if _, err := ParseDefinitions(body); err == nil { diff --git a/grafana-alertcheck/internal/gate/parse_state_test.go b/grafana-alertcheck/internal/gate/parse_state_test.go index e370c0f5e..8fca6e7ab 100644 --- a/grafana-alertcheck/internal/gate/parse_state_test.go +++ b/grafana-alertcheck/internal/gate/parse_state_test.go @@ -283,10 +283,7 @@ func TestInstanceKey(t *testing.T) { } } -// TestInstanceKey_NoCollision guards against ambiguous identities: label -// values may legally contain "\n" or "=", and a naive "k=v\n" join would -// collide e.g. {a:"1\nb=2"} with {a:"1",b:"2"}. The JSON encoding must keep -// such sets distinct. +// Label values may contain "\n" or "="; the JSON encoding must keep them distinct. func TestInstanceKey_NoCollision(t *testing.T) { if instanceKey(map[string]string{"a": "1\nb=2"}) == instanceKey(map[string]string{"a": "1", "b": "2"}) { t.Errorf("instanceKey collided for sets {a:1\\nb=2} and {a:1,b:2}") diff --git a/grafana-alertcheck/internal/gate/schedule_test.go b/grafana-alertcheck/internal/gate/schedule_test.go index 9985a0bd3..e897414bc 100644 --- a/grafana-alertcheck/internal/gate/schedule_test.go +++ b/grafana-alertcheck/internal/gate/schedule_test.go @@ -230,9 +230,7 @@ func TestScheduler_EarliestDueEmpty(t *testing.T) { } } -// TestScheduler_EarliestDueZeroTime pins the empty-detection fix: a -// non-empty scheduler whose earliest next-due time is the zero time must still -// report ok=true. The old IsZero() sentinel misread exactly this as "no rules". +// A zero next-due time is real, not an empty scheduler. func TestScheduler_EarliestDueZeroTime(t *testing.T) { s := &Scheduler{ next: map[string]time.Time{"r1": {}}, From 6ac0240c6987840304f2320d7863df9bec6eb625 Mon Sep 17 00:00:00 2001 From: Bartek Tofel Date: Mon, 31 Aug 2026 17:16:00 +0200 Subject: [PATCH 22/43] chore: implement phase 8 Invariant defended: H6/H7. The one question: can a violation ever outrank an unobservable rule, or can a pass happen without Violations empty and err nil? Adds classify.go: the pure per-instance classifier (outcome table, preexisting policy, BadFor) and decide(), the seam combining proveCoverage with those timelines under one Policy. Consolidates rule-poll filtering and skew translation onto pollsForRule/runnerTime, shared with coverage.go. --- grafana-alertcheck/internal/gate/classify.go | 601 +++++++++++++ .../internal/gate/classify_test.go | 823 ++++++++++++++++++ grafana-alertcheck/internal/gate/coverage.go | 21 +- 3 files changed, 1432 insertions(+), 13 deletions(-) create mode 100644 grafana-alertcheck/internal/gate/classify.go create mode 100644 grafana-alertcheck/internal/gate/classify_test.go diff --git a/grafana-alertcheck/internal/gate/classify.go b/grafana-alertcheck/internal/gate/classify.go new file mode 100644 index 000000000..3f14b6be1 --- /dev/null +++ b/grafana-alertcheck/internal/gate/classify.go @@ -0,0 +1,601 @@ +package gate + +import ( + "fmt" + "slices" + "strings" + "time" +) + +// ReasonNodata is decide's own unobservable reason (§10.1/§10.2): proveCoverage +// (P7) deliberately never sets it — health=nodata is a note there, never fatal, +// because escalating it needs Policy.NodataIsUnobservable, and the pure +// coverage layer has no Policy to consult (coverage.go, check 5). decide is +// the seam that DOES have a Policy, so the escalation lives here. +const ReasonNodata UnobservableReason = "nodata" + +// Outcome is the verdict of one instance's timeline, and — after decide takes +// the worst across a rule's instances — of the rule itself (§9). It is a +// published JSON output (§19.0): the three fail values stay distinct even +// though v1 maps all three to exit 1, because a later reason string cannot +// recover the information a single "fail" value would have thrown away, and +// because splitting them later would break a published interface for no gain. +type Outcome string + +const ( + OutcomeClean Outcome = "clean" + OutcomeNewlyBad Outcome = "newly_bad" + OutcomeRecovered Outcome = "recovered" + OutcomePersistentlyBad Outcome = "persistently_bad" + OutcomeFlapping Outcome = "flapping" + OutcomeSkipped Outcome = "skipped" + OutcomeUnobservable Outcome = "unobservable" +) + +// PreexistingPolicy governs only the ONE ambiguous case in the outcome table: +// an instance that was already bad when the window opened. A newly_bad or +// flapping instance is a fail under every policy (§11.3) — the plan lists +// them among the outcomes that "do not change" — so this type only ever +// changes how `recovered` and `persistently_bad` are judged (isViolation +// below). +type PreexistingPolicy string + +const ( + // PreexistingFailUnlessRecovered is the default (§11.7): a preexisting + // instance that clears and stays clear is a pass (`recovered`); one that + // never clears is still a fail (`persistently_bad`). + PreexistingFailUnlessRecovered PreexistingPolicy = "fail-unless-recovered" + // PreexistingFail makes ANY preexisting instance a fail, even one that + // recovers — for a user who wants no benefit of the doubt for a + // condition this release did not cause. + PreexistingFail PreexistingPolicy = "fail" + // PreexistingIgnore disregards a preexisting instance entirely, whether + // it recovers or stays bad for the whole window: only a genuinely NEW + // bad episode (newly_bad or flapping) can fail the rule. + PreexistingIgnore PreexistingPolicy = "ignore" +) + +// Violation is one instance whose timeline outcome counts against the run, +// after the preexisting policy has been applied (isViolation below). +type Violation struct { + Alert, RuleUID string + Outcome Outcome + State State + Health string // raw, reporting-only, like Poll.Health (P1.2a) + LastError string + // FirstSeen is the episode's onset, in the runner domain (§16): activeAt + // translated by its poll's own skew when the episode opened strictly + // inside the window, or `from` itself when the instance was already bad + // at window-open (preexisting) — never a raw, untranslated Grafana + // timestamp. + FirstSeen time.Time + // ClearedAt is zero unless the episode closed via a genuine Cleared + // event, also translated to the runner domain. + ClearedAt time.Time + InstanceLabels map[string]string + // Note carries an explanation for a Violation that has no instance + // behind it — the synthetic MinObserved shortfall entry decide emits + // when the deficit exceeds what any named paused rule explains (§12). + // LastError is reporting-only rule state from a real poll and must not + // double as a message field for a Violation that never touched one. + Note string +} + +// RuleVerdict is one rule's worst-of outcome (§9), always present for every +// resolved rule — Verdicts includes the passes, not only the failures — so a +// human reading the table sees every alert that was asked for, not only the +// ones that misbehaved. +type RuleVerdict struct { + Alert, RuleUID string + Outcome Outcome + BadFor time.Duration // total wall-clock time any instance was bad inside the window, overlaps merged + PollEvery time.Duration + Note string +} + +// Policy is decide's narrowed, pure-layer view of Config/Cfg (§9's P9 +// comment): the classification knobs and the window, nothing else. No URL, +// no token, no I/O handles — those never reach the pure layer. +type Policy struct { + States []State + Preexisting PreexistingPolicy + MinObserved int + AllowPaused, NodataIsUnobservable bool + From, To time.Time +} + +// Result is decide's whole answer: everything §20.2's table and the action's +// JSON outputs need. Coverage carries one CoverageResult per non-skipped +// rule — no separate Interval type anywhere in the project (§2's +// simplification table). +type Result struct { + From, To time.Time + GrafanaVersion string + ClockSkew time.Duration // the largest |skew| across every poll decide was given, not only the ones a rule's window actually used + Coverage map[string]CoverageResult + Verdicts []RuleVerdict + Violations []Violation +} + +// episode is one contiguous, policy-bad span of one instance's timeline, +// already resolved to the runner domain and clamped to [from, windowEnd]. It +// never crosses a genuine Cleared event (H2): a Vanished marker freezes the +// state instead of closing the episode, which is what keeps a vanish from +// ever reading as a recovery. +type episode struct { + start, end time.Time + closedByRealClear bool +} + +// instanceTimeline accumulates one instance's walk across a rule's in-window +// polls. preexisting is decided once, the first time this key is seen bad: +// by the translated ActiveAt against `from` (§16), never by which poll +// happened to report it first — a poll's own cadence is not evidence of when +// the condition actually began (F1/F2). +type instanceTimeline struct { + labels map[string]string + preexisting bool + seen bool + badOpen bool + episodeStart time.Time + lastState State + lastHealth string + lastError string + episodes []episode +} + +// runnerTime translates a Grafana-domain timestamp recorded on poll p into +// the runner domain, undoing that poll's own measured skew (§16). GrafanaNow +// and ActiveAt come from the same response, so the same poll's skew applies +// to both. This is the single implementation of that translation for the +// package (same drift argument as pollsForRule, F5): coverage.go's window +// membership test and heartbeat boundary segments call it too, rather than +// each keeping its own copy of `p.GrafanaNow.Add(-p.Skew())` that could +// silently diverge from this one. +func runnerTime(p Poll, grafanaDomain time.Time) time.Time { + return grafanaDomain.Add(-p.Skew()) +} + +// classifyRule builds every instance timeline for one rule across +// [from, windowEnd] and reduces them to the rule's worst outcome (§9), its +// merged BadFor, and the Violations the preexisting policy actually charges +// against the run. It is PURE: no I/O, no clock reads (§2) — decide supplies +// windowEnd (to + transitionGrace) rather than this function deriving it, so +// a test can pin the boundary directly. +// +// polls need not be pre-filtered to this rule, matching proveCoverage's own +// contract (§14.5): selection is by def.UID. +func classifyRule(def Definition, polls []Poll, from, windowEnd time.Time, badStates map[State]bool, pol PreexistingPolicy) (Outcome, time.Duration, []Violation) { + rulePolls := pollsForRule(polls, def.UID) + inWindow := inWindowPolls(rulePolls, from, windowEnd) + + timelines := make(map[string]*instanceTimeline) + order := make([]string, 0) + + // get backfills labels the first time a real Instance is seen (F4): a key + // can be created earlier by a bare Cleared/Vanished marker, which carries + // no labels of its own, and the instance later re-firing must not report + // an empty InstanceLabels just because of which event happened to create + // the timeline first. + get := func(key string, labels map[string]string) *instanceTimeline { + tl, ok := timelines[key] + if !ok { + tl = &instanceTimeline{labels: labels} + timelines[key] = tl + order = append(order, key) + return tl + } + if tl.labels == nil && labels != nil { + tl.labels = labels + } + return tl + } + + openEpisode := func(tl *instanceTimeline, start time.Time) { + tl.badOpen = true + tl.episodeStart = start + } + closeEpisode := func(tl *instanceTimeline, end time.Time, real bool) { + // inWindowPolls admits a poll whose translated time is up to its own + // skew bound PAST windowEnd (the membership test widens the boundary + // outward, §16). Without this clamp a genuine Cleared event on such a + // poll would produce an episode.end slightly beyond windowEnd, + // contradicting the episode type's own "clamped to + // [from, windowEnd]" contract. + if end.After(windowEnd) { + end = windowEnd + } + // Different polls can carry different measured skews. In theory a + // closing poll's translated time could land before the opening + // poll's — skew is capped at skewHardLimit (60s), so this is remote, + // not impossible — and a negative span would feed mergeDurations a + // duration that subtracts instead of adds. Clamp rather than trust + // the arithmetic never to invert. + if end.Before(tl.episodeStart) { + end = tl.episodeStart + } + tl.episodes = append(tl.episodes, episode{start: tl.episodeStart, end: end, closedByRealClear: real}) + tl.badOpen = false + } + // onsetOf resolves a fresh episode's start: the instance's own ActiveAt, + // translated to the runner domain by this poll's skew, clamped so it + // never reads as starting before the window opened. + onsetOf := func(p Poll, inst Instance) time.Time { + start := runnerTime(p, inst.ActiveAt) + if start.Before(from) { + start = from + } + return start + } + + for _, p := range inWindow { + byKey := make(map[string]Instance, len(p.Abnormal)) + for _, inst := range p.Abnormal { + byKey[instanceKey(inst.Labels)] = inst + } + + for key, inst := range byKey { + tl := get(key, inst.Labels) + bad := badStates[inst.State] + switch { + case !tl.seen: + tl.seen = true + if bad { + // Fail-closed (§16): only call an onset "preexisting" + // when even the worst-case skew error still puts it at + // or before `from`. An onset that might really have + // landed just inside the window must classify as a new + // episode, never earn the `recovered` benefit of the + // doubt it would get if it later clears (F1/F2). + activeAtRunner := runnerTime(p, inst.ActiveAt) + tl.preexisting = !activeAtRunner.Add(p.SkewBound()).After(from) + if tl.preexisting { + openEpisode(tl, from) + } else { + openEpisode(tl, onsetOf(p, inst)) + } + } + case bad && !tl.badOpen: + openEpisode(tl, onsetOf(p, inst)) + case !bad && tl.badOpen: + closeEpisode(tl, runnerTime(p, p.GrafanaNow), true) + } + tl.lastState, tl.lastHealth, tl.lastError = inst.State, p.Health, p.LastError + } + + for _, key := range p.Cleared { + tl := get(key, nil) + if !tl.seen { + // Cleared on the very first mention means the transition + // happened between the poll just before this one (possibly + // pre-window) and this one: there is no window-internal + // evidence that it was ever bad, so it is neither + // preexisting nor a new episode. + tl.seen = true + continue + } + if tl.badOpen { + closeEpisode(tl, runnerTime(p, p.GrafanaNow), true) + } + tl.lastHealth, tl.lastError = p.Health, p.LastError + } + + // Vanished is a deliberate no-op (H2): freeze whatever badOpen/preexisting + // already holds. An instance that vanishes while bad must stay bad, and + // one that vanishes while never having been bad must stay uninteresting. + for _, key := range p.Vanished { + tl := get(key, nil) + tl.seen = true + tl.lastHealth = p.Health + } + } + + // Multiple instances can appear for the first time within the same poll, + // and map iteration order is nondeterministic; sort so this pure + // function's Violations/BadFor output is stable across runs given the + // same input, like log.go sorts Cleared/Vanished for the same reason. + slices.Sort(order) + + var ( + outcome Outcome = OutcomeClean + badFor []episode + viols []Violation + ) + + for _, key := range order { + tl := timelines[key] + if tl.badOpen { + closeEpisode(tl, windowEnd, false) + } + if len(tl.episodes) == 0 { + continue + } + + var instOutcome Outcome + switch { + case len(tl.episodes) > 1: + instOutcome = OutcomeFlapping + case tl.preexisting: + if tl.episodes[0].closedByRealClear { + instOutcome = OutcomeRecovered + } else { + instOutcome = OutcomePersistentlyBad + } + default: + // A genuinely new onset always fails, whether or not it later + // clears within the window (§11.4 point 3): only a PREEXISTING + // condition earns the benefit of `recovered`. + instOutcome = OutcomeNewlyBad + } + + if outcomeRank(instOutcome) > outcomeRank(outcome) { + outcome = instOutcome + } + badFor = append(badFor, tl.episodes...) + + if isViolation(instOutcome, pol) { + var clearedAt time.Time + last := tl.episodes[len(tl.episodes)-1] + if last.closedByRealClear { + clearedAt = last.end + } + viols = append(viols, Violation{ + Alert: def.Title, + RuleUID: def.UID, + Outcome: instOutcome, + State: tl.lastState, + Health: tl.lastHealth, + LastError: tl.lastError, + FirstSeen: tl.episodes[0].start, + ClearedAt: clearedAt, + InstanceLabels: tl.labels, + }) + } + } + + return outcome, mergeDurations(badFor), viols +} + +// isViolation decides whether one instance's outcome counts against the run, +// once the preexisting policy is applied. newly_bad and flapping always do +// (§11.3): both contain a genuinely new bad episode, so no policy forgives +// them. recovered and persistently_bad are, by classifyRule's construction, +// ALWAYS preexisting (a non-preexisting single episode is newly_bad instead, +// regardless of whether it clears) — so these are the only two policy can +// change, and isViolation needs no separate preexisting flag to know that. +func isViolation(o Outcome, pol PreexistingPolicy) bool { + switch o { + case OutcomeNewlyBad, OutcomeFlapping: + return true + case OutcomePersistentlyBad: + return pol != PreexistingIgnore + case OutcomeRecovered: + return pol == PreexistingFail + default: + return false + } +} + +// outcomeRank orders outcomes for classifyRule's worst-of reduction across a +// rule's instances (§9). The three fail values, and recovered above clean, +// give it exactly the ordering the table requires — +// "unobservable > {flapping, persistently_bad, newly_bad} > recovered > +// skipped > clean" — with unobservable and skipped applied outside this +// function (decide owns both: unobservable from CoverageResult, skipped from +// Definition.IsPaused). The table does not distinguish among the three fail +// values, so their relative order here (flapping above persistently_bad +// above newly_bad) is an arbitrary but fixed and documented tie-break, not a +// claim that one is worse than another. +func outcomeRank(o Outcome) int { + switch o { + case OutcomeFlapping: + return 4 + case OutcomePersistentlyBad: + return 3 + case OutcomeNewlyBad: + return 2 + case OutcomeRecovered: + return 1 + default: // OutcomeClean + return 0 + } +} + +// mergeDurations sums the wall-clock time covered by a set of episodes, +// merging overlaps so a rule with several simultaneously-bad instances is +// not reported as bad for longer than it actually was. +func mergeDurations(eps []episode) time.Duration { + if len(eps) == 0 { + return 0 + } + sorted := slices.Clone(eps) + slices.SortStableFunc(sorted, func(a, b episode) int { return a.start.Compare(b.start) }) + + var total time.Duration + cur := sorted[0] + for _, e := range sorted[1:] { + if e.start.After(cur.end) { + total += cur.end.Sub(cur.start) + cur = e + continue + } + if e.end.After(cur.end) { + cur.end = e.end + } + } + total += cur.end.Sub(cur.start) + return total +} + +// pollsForRule filters polls to one rule and sorts them by GrafanaNow, the +// same selection proveCoverage uses (§14.5: selection is by UID, never by +// title) — stable, because two polls sharing a coarse Date header must not +// reorder nondeterministically in a pure function. This is the single +// filter+sort implementation for the package (F5): proveCoverage calls it +// too, rather than keeping its own copy that could silently drift from this +// one's membership test. +func pollsForRule(polls []Poll, uid string) []Poll { + var out []Poll + for _, p := range polls { + if p.RuleUID == uid { + out = append(out, p) + } + } + slices.SortStableFunc(out, func(a, b Poll) int { return a.GrafanaNow.Compare(b.GrafanaNow) }) + return out +} + +// badStateSet turns Policy.States into a lookup set, defaulting to {firing} +// (§13) when the caller leaves States empty — decide applies the default +// itself so a test can pass a zero-value Policy and get v1's real default, +// rather than relying on a CLI layer that does not exist yet. +func badStateSet(states []State) map[State]bool { + if len(states) == 0 { + states = []State{StateFiring} + } + set := make(map[State]bool, len(states)) + for _, s := range states { + set[s] = true + } + return set +} + +// decide is the pure seam between the collected evidence and the CLI's exit +// code: nearly every §22 test targets this function, not Check (P9). It +// combines proveCoverage's nine checks with classifyRule's timelines under +// one Policy, and OWNS the H6 mapping: any unobservable rule makes decide +// return a non-nil error, which P10's CLI maps to exit 2 unconditionally +// (H7) — never to 0 or 1, and never suppressed by a real violation found +// alongside it. +// +// Result is fully populated even when the returned error is non-nil: H7's +// "err != nil, the violation list is irrelevant" means the CALLER must not +// use Violations to second-guess the error, not that Result stops being +// useful for the human table on exit 2. +func decide(h Header, polls []Poll, sentinel *time.Time, defs []Definition, + rt map[string]ruleTimings, gt globalTimings, pol Policy) (Result, error) { + + badStates := badStateSet(pol.States) + + result := Result{ + From: pol.From, + To: pol.To, + GrafanaVersion: h.GrafanaVersion, + Coverage: make(map[string]CoverageResult), + } + for _, p := range polls { + s := p.Skew() + if s < 0 { + s = -s + } + if s > result.ClockSkew { + result.ClockSkew = s + } + } + + minObserved := pol.MinObserved + if minObserved == 0 { + minObserved = len(defs) + } + + windowEnd := pol.To.Add(gt.transitionGrace) + + var ( + skippedRules []Definition + observedCount int + anyUnobservable bool + unobservableNames []string + ) + + for _, def := range defs { + if def.IsPaused { + skippedRules = append(skippedRules, def) + result.Verdicts = append(result.Verdicts, RuleVerdict{ + Alert: def.Title, RuleUID: def.UID, Outcome: OutcomeSkipped, + PollEvery: rt[def.UID].pollEvery, + Note: "paused before the window opened", + }) + continue + } + observedCount++ + + t := rt[def.UID] + cov := proveCoverage(h, polls, sentinel, t, def, pol.From, pol.To, gt.transitionGrace) + + if pol.NodataIsUnobservable && !cov.Unobservable { + inWindow := inWindowPolls(pollsForRule(polls, def.UID), pol.From, windowEnd) + if runLen, sawAny := longestHealthRun(inWindow, "nodata"); sawAny && runLen > t.healthGrace { + cov.Unobservable = true + cov.Proved = false + if cov.Reason == "" { + cov.Reason = ReasonNodata + } + cov.Notes = append(cov.Notes, fmt.Sprintf( + "rule %q: health=nodata for %s exceeds healthGrace %s and --nodata-is-unobservable is set", + def.Title, runLen, t.healthGrace)) + } + } + result.Coverage[def.UID] = cov + + outcome, badFor, viols := classifyRule(def, polls, pol.From, windowEnd, badStates, pol.Preexisting) + if cov.Unobservable { + outcome = OutcomeUnobservable + anyUnobservable = true + unobservableNames = append(unobservableNames, fmt.Sprintf("%s (%s)", def.Title, cov.Reason)) + } + result.Violations = append(result.Violations, viols...) + result.Verdicts = append(result.Verdicts, RuleVerdict{ + Alert: def.Title, RuleUID: def.UID, Outcome: outcome, BadFor: badFor, + PollEvery: t.pollEvery, Note: strings.Join(cov.Notes, "; "), + }) + } + + // MinObserved (§12): default len(defs) after the collapse (already done + // by Resolve before decide ever sees defs). skipped rules count against + // it unless AllowPaused says otherwise. A shortfall counts toward exit 1 + // (§9.1), never exit 2 — decide never returns an error for this — and H7 + // requires it to surface through Violations like any other fail reason, + // so a shortfall always produces at least one, even when no rule is + // paused at all (an operator-supplied MinObserved that simply exceeds + // what could ever be resolved). + counted := observedCount + var chargeable []Definition + if pol.AllowPaused { + counted += len(skippedRules) + } else { + chargeable = skippedRules + } + if shortfall := minObserved - counted; shortfall > 0 { + charged := 0 + for _, def := range chargeable { + if charged >= shortfall { + break + } + // §12.1 requires the paused rule and --allow-paused both be + // named to the user; naming the rule is this Violation's job, + // the --allow-paused hint is the CLI table/renderer's (P10) — + // tracked here so it is not dropped when that phase is built. + result.Violations = append(result.Violations, Violation{ + Alert: def.Title, RuleUID: def.UID, Outcome: OutcomeSkipped, + Note: "paused before the window opened; counts against --min-observed unless --allow-paused is set", + }) + charged++ + } + for ; charged < shortfall; charged++ { + // No named rule explains this part of the deficit — e.g. an + // operator-supplied --min-observed above what could ever be + // resolved. Note, not LastError: LastError is reporting-only + // rule state read from a real poll, and this Violation never + // touched one. + result.Violations = append(result.Violations, Violation{ + Outcome: OutcomeSkipped, + Note: fmt.Sprintf("min-observed %d exceeds the %d rule(s) counted as observed", minObserved, counted), + }) + } + } + + if anyUnobservable { + return result, fmt.Errorf("gate: %d rule(s) unobservable: %s", len(unobservableNames), strings.Join(unobservableNames, "; ")) + } + return result, nil +} diff --git a/grafana-alertcheck/internal/gate/classify_test.go b/grafana-alertcheck/internal/gate/classify_test.go new file mode 100644 index 000000000..5d773e241 --- /dev/null +++ b/grafana-alertcheck/internal/gate/classify_test.go @@ -0,0 +1,823 @@ +package gate + +import ( + "testing" + "time" +) + +func lbl(name string) map[string]string { return map[string]string{"instance": name} } + +// abnormalPoll builds one Poll carrying a single abnormal instance, with the +// bookkeeping classifyRule needs (RuleUID, GrafanaNow, Health, Abnormal). +func abnormalPoll(uid string, at time.Time, state State, labels map[string]string, activeAt time.Time) Poll { + return Poll{ + RuleUID: uid, + GrafanaNow: at, + Found: true, + Health: "ok", + LastEvaluation: at, + Abnormal: []Instance{{Labels: labels, State: state, ActiveAt: activeAt}}, + } +} + +func clearedPoll(uid string, at time.Time, cleared ...string) Poll { + return Poll{RuleUID: uid, GrafanaNow: at, Found: true, Health: "ok", LastEvaluation: at, Cleared: cleared} +} + +func vanishedPoll(uid string, at time.Time, vanished ...string) Poll { + return Poll{RuleUID: uid, GrafanaNow: at, Found: true, Health: "ok", LastEvaluation: at, Vanished: vanished} +} + +func quietPoll(uid string, at time.Time) Poll { + return Poll{RuleUID: uid, GrafanaNow: at, Found: true, Health: "ok", LastEvaluation: at} +} + +var defaultBad = badStateSet(nil) // {firing} + +// --- clean / newly_bad --- + +func TestClassifyRule_NoEvidenceIsClean(t *testing.T) { + from := time.Date(2026, 1, 1, 0, 0, 0, 0, time.UTC) + to := from.Add(10 * time.Minute) + def := Definition{UID: "r1", Title: "R1"} + + polls := []Poll{quietPoll("r1", from), quietPoll("r1", to)} + outcome, badFor, viols := classifyRule(def, polls, from, to, defaultBad, PreexistingFailUnlessRecovered) + if outcome != OutcomeClean || badFor != 0 || len(viols) != 0 { + t.Fatalf("outcome=%v badFor=%v viols=%v, want clean/0/none", outcome, badFor, viols) + } +} + +func TestClassifyRule_NewOnsetInsideWindowIsNewlyBad(t *testing.T) { + from := time.Date(2026, 1, 1, 0, 0, 0, 0, time.UTC) + to := from.Add(10 * time.Minute) + onset := from.Add(5 * time.Minute) + def := Definition{UID: "r1", Title: "R1"} + + polls := []Poll{ + quietPoll("r1", from), + abnormalPoll("r1", onset, StateFiring, lbl("a"), onset), + abnormalPoll("r1", to, StateFiring, lbl("a"), onset), + } + outcome, badFor, viols := classifyRule(def, polls, from, to, defaultBad, PreexistingFailUnlessRecovered) + if outcome != OutcomeNewlyBad { + t.Fatalf("outcome = %v, want newly_bad", outcome) + } + if want := to.Sub(onset); badFor != want { + t.Fatalf("badFor = %v, want %v", badFor, want) + } + if len(viols) != 1 || viols[0].Outcome != OutcomeNewlyBad { + t.Fatalf("viols = %+v, want exactly one newly_bad violation", viols) + } +} + +// TestClassifyRule_NewOnsetThatClearsStillFails pins §11.4 point 3: a +// genuinely new bad episode fails even if it clears again before the window +// ends — only a PREEXISTING condition earns the benefit of `recovered`. +func TestClassifyRule_NewOnsetThatClearsStillFails(t *testing.T) { + from := time.Date(2026, 1, 1, 0, 0, 0, 0, time.UTC) + to := from.Add(10 * time.Minute) + onset := from.Add(2 * time.Minute) + clearAt := from.Add(3 * time.Minute) + def := Definition{UID: "r1", Title: "R1"} + + polls := []Poll{ + quietPoll("r1", from), + abnormalPoll("r1", onset, StateFiring, lbl("a"), onset), + clearedPoll("r1", clearAt, instanceKey(lbl("a"))), + quietPoll("r1", to), + } + outcome, _, viols := classifyRule(def, polls, from, to, defaultBad, PreexistingFailUnlessRecovered) + if outcome != OutcomeNewlyBad { + t.Fatalf("outcome = %v, want newly_bad even though it cleared", outcome) + } + if len(viols) != 1 { + t.Fatalf("viols = %+v, want one violation", viols) + } +} + +// --- recovered / persistently_bad (preexisting) --- + +func TestClassifyRule_PreexistingThatRecoversIsRecoveredAndNotAViolation(t *testing.T) { + from := time.Date(2026, 1, 1, 0, 0, 0, 0, time.UTC) + to := from.Add(10 * time.Minute) + clearAt := from.Add(8 * time.Minute) + def := Definition{UID: "r1", Title: "R1"} + key := instanceKey(lbl("a")) + + polls := []Poll{ + abnormalPoll("r1", from, StateFiring, lbl("a"), from.Add(-time.Hour)), + clearedPoll("r1", clearAt, key), + quietPoll("r1", to), + } + outcome, badFor, viols := classifyRule(def, polls, from, to, defaultBad, PreexistingFailUnlessRecovered) + if outcome != OutcomeRecovered { + t.Fatalf("outcome = %v, want recovered", outcome) + } + if want := clearAt.Sub(from); badFor != want { + t.Fatalf("badFor = %v, want %v", badFor, want) + } + if len(viols) != 0 { + t.Fatalf("viols = %+v, want none: default policy passes a recovered preexisting instance", viols) + } +} + +func TestClassifyRule_PreexistingStillBadAtWindowEndIsPersistentlyBad(t *testing.T) { + from := time.Date(2026, 1, 1, 0, 0, 0, 0, time.UTC) + to := from.Add(10 * time.Minute) + def := Definition{UID: "r1", Title: "R1"} + + polls := []Poll{ + abnormalPoll("r1", from, StateFiring, lbl("a"), from.Add(-time.Hour)), + abnormalPoll("r1", to, StateFiring, lbl("a"), from.Add(-time.Hour)), + } + outcome, badFor, viols := classifyRule(def, polls, from, to, defaultBad, PreexistingFailUnlessRecovered) + if outcome != OutcomePersistentlyBad { + t.Fatalf("outcome = %v, want persistently_bad", outcome) + } + if badFor != to.Sub(from) { + t.Fatalf("badFor = %v, want the full window %v", badFor, to.Sub(from)) + } + if len(viols) != 1 || viols[0].Outcome != OutcomePersistentlyBad { + t.Fatalf("viols = %+v, want one persistently_bad violation", viols) + } +} + +// --- flapping --- + +func TestClassifyRule_ClearThenBadAgainIsFlapping(t *testing.T) { + from := time.Date(2026, 1, 1, 0, 0, 0, 0, time.UTC) + to := from.Add(10 * time.Minute) + def := Definition{UID: "r1", Title: "R1"} + key := instanceKey(lbl("a")) + + polls := []Poll{ + abnormalPoll("r1", from, StateFiring, lbl("a"), from), + clearedPoll("r1", from.Add(2*time.Minute), key), + abnormalPoll("r1", from.Add(5*time.Minute), StateFiring, lbl("a"), from.Add(5*time.Minute)), + quietPoll("r1", to), + } + outcome, _, viols := classifyRule(def, polls, from, to, defaultBad, PreexistingFailUnlessRecovered) + if outcome != OutcomeFlapping { + t.Fatalf("outcome = %v, want flapping", outcome) + } + if len(viols) != 1 || viols[0].Outcome != OutcomeFlapping { + t.Fatalf("viols = %+v, want one flapping violation, always a fail regardless of policy", viols) + } +} + +// --- H2: vanished is a discontinuity, never a clear --- + +func TestClassifyRule_VanishedWhileBadStaysPersistentlyBad(t *testing.T) { + from := time.Date(2026, 1, 1, 0, 0, 0, 0, time.UTC) + to := from.Add(10 * time.Minute) + def := Definition{UID: "r1", Title: "R1"} + key := instanceKey(lbl("a")) + + polls := []Poll{ + abnormalPoll("r1", from, StateFiring, lbl("a"), from.Add(-time.Minute)), + vanishedPoll("r1", from.Add(5*time.Minute), key), + quietPoll("r1", to), + } + outcome, badFor, viols := classifyRule(def, polls, from, to, defaultBad, PreexistingFailUnlessRecovered) + if outcome != OutcomePersistentlyBad { + t.Fatalf("outcome = %v, want persistently_bad: a vanish must never read as a recovery (H2)", outcome) + } + if badFor != to.Sub(from) { + t.Fatalf("badFor = %v, want the full window %v: the freeze must hold the episode open to windowEnd", badFor, to.Sub(from)) + } + if len(viols) != 1 { + t.Fatalf("viols = %+v, want one violation", viols) + } +} + +func TestClassifyRule_VanishedWhileNeverBadIsUninteresting(t *testing.T) { + from := time.Date(2026, 1, 1, 0, 0, 0, 0, time.UTC) + to := from.Add(10 * time.Minute) + def := Definition{UID: "r1", Title: "R1"} + key := instanceKey(lbl("a")) + + // Pending is abnormal (non-normal) but not in the default {firing} bad + // set, so its vanish must stay uninteresting too. + polls := []Poll{ + abnormalPoll("r1", from, StatePending, lbl("a"), from), + vanishedPoll("r1", from.Add(5*time.Minute), key), + quietPoll("r1", to), + } + outcome, badFor, viols := classifyRule(def, polls, from, to, defaultBad, PreexistingFailUnlessRecovered) + if outcome != OutcomeClean || badFor != 0 || len(viols) != 0 { + t.Fatalf("outcome=%v badFor=%v viols=%v, want clean/0/none", outcome, badFor, viols) + } +} + +// --- preexisting policy --- + +func TestClassifyRule_PreexistingPolicyFailFailsARecoveredInstance(t *testing.T) { + from := time.Date(2026, 1, 1, 0, 0, 0, 0, time.UTC) + to := from.Add(10 * time.Minute) + def := Definition{UID: "r1", Title: "R1"} + key := instanceKey(lbl("a")) + + polls := []Poll{ + abnormalPoll("r1", from, StateFiring, lbl("a"), from.Add(-time.Hour)), + clearedPoll("r1", from.Add(2*time.Minute), key), + quietPoll("r1", to), + } + outcome, _, viols := classifyRule(def, polls, from, to, defaultBad, PreexistingFail) + if outcome != OutcomeRecovered { + t.Fatalf("outcome = %v, want recovered — the descriptive outcome does not change under policy=fail", outcome) + } + if len(viols) != 1 || viols[0].Outcome != OutcomeRecovered { + t.Fatalf("viols = %+v, want one violation: policy=fail gives no benefit of the doubt to a preexisting instance", viols) + } +} + +func TestClassifyRule_PreexistingPolicyIgnoreForgivesPersistentlyBad(t *testing.T) { + from := time.Date(2026, 1, 1, 0, 0, 0, 0, time.UTC) + to := from.Add(10 * time.Minute) + def := Definition{UID: "r1", Title: "R1"} + + polls := []Poll{ + abnormalPoll("r1", from, StateFiring, lbl("a"), from.Add(-time.Hour)), + abnormalPoll("r1", to, StateFiring, lbl("a"), from.Add(-time.Hour)), + } + outcome, _, viols := classifyRule(def, polls, from, to, defaultBad, PreexistingIgnore) + if outcome != OutcomePersistentlyBad { + t.Fatalf("outcome = %v, want persistently_bad — the descriptive outcome does not change under policy=ignore", outcome) + } + if len(viols) != 0 { + t.Fatalf("viols = %+v, want none: policy=ignore disregards a preexisting instance even if it never recovers", viols) + } +} + +func TestClassifyRule_PreexistingPolicyIgnoreStillFailsANewOnset(t *testing.T) { + from := time.Date(2026, 1, 1, 0, 0, 0, 0, time.UTC) + to := from.Add(10 * time.Minute) + onset := from.Add(5 * time.Minute) + def := Definition{UID: "r1", Title: "R1"} + + polls := []Poll{ + quietPoll("r1", from), + abnormalPoll("r1", onset, StateFiring, lbl("a"), onset), + abnormalPoll("r1", to, StateFiring, lbl("a"), onset), + } + outcome, _, viols := classifyRule(def, polls, from, to, defaultBad, PreexistingIgnore) + if outcome != OutcomeNewlyBad || len(viols) != 1 { + t.Fatalf("outcome=%v viols=%v, want newly_bad/1: ignore only forgives PREEXISTING badness", outcome, viols) + } +} + +// --- worst-of across instances --- + +func TestClassifyRule_WorstOfMultipleInstancesWins(t *testing.T) { + from := time.Date(2026, 1, 1, 0, 0, 0, 0, time.UTC) + to := from.Add(10 * time.Minute) + def := Definition{UID: "r1", Title: "R1"} + + polls := []Poll{ + { + RuleUID: "r1", GrafanaNow: from, Found: true, Health: "ok", + Abnormal: []Instance{ + {Labels: lbl("a"), State: StateFiring, ActiveAt: from.Add(-time.Hour)}, // preexisting, will recover + {Labels: lbl("b"), State: StateFiring, ActiveAt: from}, // preexisting, will stay bad + }, + }, + clearedPoll("r1", from.Add(2*time.Minute), instanceKey(lbl("a"))), + { + RuleUID: "r1", GrafanaNow: to, Found: true, Health: "ok", + Abnormal: []Instance{{Labels: lbl("b"), State: StateFiring, ActiveAt: from}}, + }, + } + outcome, _, viols := classifyRule(def, polls, from, to, defaultBad, PreexistingFailUnlessRecovered) + if outcome != OutcomePersistentlyBad { + t.Fatalf("outcome = %v, want persistently_bad: the worse of {recovered, persistently_bad}", outcome) + } + if len(viols) != 1 || viols[0].Outcome != OutcomePersistentlyBad { + t.Fatalf("viols = %+v, want exactly the persistently_bad instance's violation", viols) + } +} + +// --- decide(): skipped rules, unobservable (H6), MinObserved, exit mapping (H7) --- + +func TestDecide_SkippedRuleNeverReachesProveCoverage(t *testing.T) { + from := time.Date(2026, 1, 1, 0, 0, 0, 0, time.UTC) + to := from.Add(10 * time.Minute) + def := Definition{UID: "r1", Title: "R1", IsPaused: true} + defs := []Definition{def} + rt := map[string]ruleTimings{"r1": newRuleTimings(30*time.Second, 60)} + gt := globalTimings{} + pol := Policy{From: from, To: to, AllowPaused: true} + + // No polls, no sentinel at all: a heartbeat_gap/no_sentinel misclassification + // here would mean proveCoverage ran for a skipped rule (§4.3's obligation). + res, err := decide(Header{StartedAt: from.Add(-time.Hour)}, nil, nil, defs, rt, gt, pol) + if err != nil { + t.Fatalf("err = %v, want nil: a rule paused before the window is skipped, not unobservable", err) + } + if len(res.Verdicts) != 1 || res.Verdicts[0].Outcome != OutcomeSkipped { + t.Fatalf("Verdicts = %+v, want exactly one skipped verdict", res.Verdicts) + } + if _, ok := res.Coverage["r1"]; ok { + t.Fatalf("Coverage[r1] present, want absent: a skipped rule has no coverage to prove") + } +} + +func TestDecide_UnobservableRuleAlwaysReturnsAnError(t *testing.T) { + from := time.Date(2026, 1, 1, 0, 0, 0, 0, time.UTC) + to := from.Add(10 * time.Minute) + def := Definition{UID: "r1", Title: "R1"} + defs := []Definition{def} + rt := map[string]ruleTimings{"r1": newRuleTimings(30*time.Second, 60)} + gt := globalTimings{} + pol := Policy{From: from, To: to} + + // No sentinel at all: check 1 fails, so the rule is unobservable + // regardless of anything else. + res, err := decide(Header{StartedAt: from.Add(-time.Hour)}, nil, nil, defs, rt, gt, pol) + if err == nil { + t.Fatalf("err = nil, want non-nil: H6/H7 require an unobservable rule to always fail the run") + } + if len(res.Verdicts) != 1 || res.Verdicts[0].Outcome != OutcomeUnobservable { + t.Fatalf("Verdicts = %+v, want exactly one unobservable verdict", res.Verdicts) + } +} + +// TestDecide_UnobservableWinsEvenAlongsideARealViolation pins H6 exactly: +// "Any unobservable rule -> exit 2, no exception, even alongside a real +// newly_bad." +func TestDecide_UnobservableWinsEvenAlongsideARealViolation(t *testing.T) { + from := time.Date(2026, 1, 1, 0, 0, 0, 0, time.UTC) + to := from.Add(10 * time.Minute) + onset := from.Add(5 * time.Minute) + + defBroken := Definition{UID: "broken", Title: "Broken"} + defBad := Definition{UID: "bad", Title: "Bad"} + defs := []Definition{defBroken, defBad} + rt := map[string]ruleTimings{ + "broken": newRuleTimings(30*time.Second, 60), + "bad": newRuleTimings(30*time.Second, 60), + } + gt := globalTimings{} + pol := Policy{From: from, To: to} + + var polls []Poll + for ts := from; !ts.After(to); ts = ts.Add(30 * time.Second) { + if ts.Equal(onset) || ts.After(onset) { + polls = append(polls, abnormalPoll("bad", ts, StateFiring, lbl("a"), onset)) + } else { + polls = append(polls, quietPoll("bad", ts)) + } + } + // "broken" gets no polls at all: no sentinel, no heartbeats -> unobservable. + sentinel := to + + res, err := decide(Header{StartedAt: from.Add(-time.Hour)}, polls, &sentinel, defs, rt, gt, pol) + if err == nil { + t.Fatalf("err = nil, want non-nil: one rule is unobservable") + } + var gotBroken, gotBad Outcome + for _, v := range res.Verdicts { + switch v.RuleUID { + case "broken": + gotBroken = v.Outcome + case "bad": + gotBad = v.Outcome + } + } + if gotBroken != OutcomeUnobservable { + t.Fatalf("broken.Outcome = %v, want unobservable", gotBroken) + } + if gotBad != OutcomeNewlyBad { + t.Fatalf("bad.Outcome = %v, want newly_bad: classification still runs and is still visible in Verdicts (H5)", gotBad) + } + if len(res.Violations) == 0 { + t.Fatalf("Violations empty, want the newly_bad instance still reported even though the run fails on the unobservable rule") + } +} + +func TestDecide_CleanWindowIsAPass(t *testing.T) { + from := time.Date(2026, 1, 1, 0, 0, 0, 0, time.UTC) + to := from.Add(10 * time.Minute) + def := Definition{UID: "r1", Title: "R1"} + defs := []Definition{def} + rt := map[string]ruleTimings{"r1": newRuleTimings(30*time.Second, 60)} + gt := globalTimings{} + pol := Policy{From: from, To: to} + + var polls []Poll + for ts := from; !ts.After(to); ts = ts.Add(30 * time.Second) { + polls = append(polls, quietPoll("r1", ts)) + } + sentinel := to + + res, err := decide(Header{StartedAt: from.Add(-time.Hour)}, polls, &sentinel, defs, rt, gt, pol) + if err != nil { + t.Fatalf("err = %v, want nil", err) + } + if len(res.Violations) != 0 { + t.Fatalf("Violations = %+v, want none: H7 says a pass is exactly len(Violations)==0 && err==nil", res.Violations) + } + if res.Verdicts[0].Outcome != OutcomeClean { + t.Fatalf("Outcome = %v, want clean", res.Verdicts[0].Outcome) + } +} + +// --- MinObserved shortfall (§12) --- + +func TestDecide_SkippedOnlyShortfallProducesAViolationWithoutAnError(t *testing.T) { + from := time.Date(2026, 1, 1, 0, 0, 0, 0, time.UTC) + to := from.Add(10 * time.Minute) + + watched := Definition{UID: "watched", Title: "Watched"} + paused := Definition{UID: "paused", Title: "Paused", IsPaused: true} + defs := []Definition{watched, paused} + rt := map[string]ruleTimings{ + "watched": newRuleTimings(30*time.Second, 60), + "paused": newRuleTimings(30*time.Second, 60), + } + gt := globalTimings{} + // MinObserved defaults to len(defs) = 2, but only "watched" is observable. + pol := Policy{From: from, To: to} + + var polls []Poll + for ts := from; !ts.After(to); ts = ts.Add(30 * time.Second) { + polls = append(polls, quietPoll("watched", ts)) + } + sentinel := to + + res, err := decide(Header{StartedAt: from.Add(-time.Hour)}, polls, &sentinel, defs, rt, gt, pol) + if err != nil { + t.Fatalf("err = %v, want nil: a shortfall caused only by a skipped rule is exit 1, not exit 2 (§9.1)", err) + } + if len(res.Violations) != 1 { + t.Fatalf("Violations = %+v, want exactly one: H7 needs the shortfall visible through Violations to keep its equivalence", res.Violations) + } + if v := res.Violations[0]; v.Outcome != OutcomeSkipped || v.RuleUID != "paused" || v.Alert != "Paused" { + t.Fatalf("Violations[0] = %+v, want Outcome=skipped naming the paused rule (§12.1: the message names the paused rule)", v) + } + if res.Violations[0].Note == "" { + t.Fatalf("Violations[0].Note is empty, want an explanation: the shortfall reason must not be smuggled into LastError, " + + "which is reporting-only rule state from a real poll this synthetic Violation never touched") + } +} + +// TestDecide_ExplicitMinObservedShortfallWithNoPausedRuleStillProducesAViolation +// pins F3: an operator-supplied MinObserved that exceeds what could ever be +// resolved is still a shortfall, even with zero paused rules to blame it on +// — H7 must not let this silently read as a pass. +func TestDecide_ExplicitMinObservedShortfallWithNoPausedRuleStillProducesAViolation(t *testing.T) { + from := time.Date(2026, 1, 1, 0, 0, 0, 0, time.UTC) + to := from.Add(10 * time.Minute) + + def := Definition{UID: "r1", Title: "R1"} + defs := []Definition{def} + rt := map[string]ruleTimings{"r1": newRuleTimings(30*time.Second, 60)} + gt := globalTimings{} + pol := Policy{From: from, To: to, MinObserved: 3} // only one rule will ever be resolved + + var polls []Poll + for ts := from; !ts.After(to); ts = ts.Add(30 * time.Second) { + polls = append(polls, quietPoll("r1", ts)) + } + sentinel := to + + res, err := decide(Header{StartedAt: from.Add(-time.Hour)}, polls, &sentinel, defs, rt, gt, pol) + if err != nil { + t.Fatalf("err = %v, want nil: an unmet MinObserved is exit 1, never exit 2", err) + } + if len(res.Violations) != 2 { + t.Fatalf("Violations = %+v, want two: the shortfall (3-1=2) is not explained by any paused rule, "+ + "so H7 requires it to surface directly rather than pass silently", res.Violations) + } + for _, v := range res.Violations { + if v.Outcome != OutcomeSkipped { + t.Fatalf("Violations = %+v, want Outcome=skipped on the synthetic shortfall entries", res.Violations) + } + } +} + +func TestDecide_AllowPausedSuppressesTheShortfall(t *testing.T) { + from := time.Date(2026, 1, 1, 0, 0, 0, 0, time.UTC) + to := from.Add(10 * time.Minute) + + watched := Definition{UID: "watched", Title: "Watched"} + paused := Definition{UID: "paused", Title: "Paused", IsPaused: true} + defs := []Definition{watched, paused} + rt := map[string]ruleTimings{ + "watched": newRuleTimings(30*time.Second, 60), + "paused": newRuleTimings(30*time.Second, 60), + } + gt := globalTimings{} + pol := Policy{From: from, To: to, AllowPaused: true} + + var polls []Poll + for ts := from; !ts.After(to); ts = ts.Add(30 * time.Second) { + polls = append(polls, quietPoll("watched", ts)) + } + sentinel := to + + res, err := decide(Header{StartedAt: from.Add(-time.Hour)}, polls, &sentinel, defs, rt, gt, pol) + if err != nil { + t.Fatalf("err = %v, want nil", err) + } + if len(res.Violations) != 0 { + t.Fatalf("Violations = %+v, want none: --allow-paused must suppress the shortfall entirely", res.Violations) + } +} + +// --- nodata escalation (decide's own Policy-driven check) --- + +func TestDecide_NodataIsUnobservableEscalatesASustainedRun(t *testing.T) { + from := time.Date(2026, 1, 1, 0, 0, 0, 0, time.UTC) + to := from.Add(10 * time.Minute) + def := Definition{UID: "r1", Title: "R1"} + defs := []Definition{def} + rt := map[string]ruleTimings{"r1": newRuleTimings(30*time.Second, 60)} // healthGrace = max(60s,60s) = 60s + gt := globalTimings{} + pol := Policy{From: from, To: to, NodataIsUnobservable: true} + + var polls []Poll + for ts := from; !ts.After(to); ts = ts.Add(30 * time.Second) { + polls = append(polls, Poll{RuleUID: "r1", GrafanaNow: ts, Found: true, Health: "nodata", LastEvaluation: ts}) + } + sentinel := to + + res, err := decide(Header{StartedAt: from.Add(-time.Hour)}, polls, &sentinel, defs, rt, gt, pol) + if err == nil { + t.Fatalf("err = nil, want non-nil: a sustained nodata run must be unobservable under --nodata-is-unobservable") + } + if res.Coverage["r1"].Reason != ReasonNodata { + t.Fatalf("Reason = %q, want %q", res.Coverage["r1"].Reason, ReasonNodata) + } +} + +func TestDecide_NodataIsANoteByDefault(t *testing.T) { + from := time.Date(2026, 1, 1, 0, 0, 0, 0, time.UTC) + to := from.Add(10 * time.Minute) + def := Definition{UID: "r1", Title: "R1"} + defs := []Definition{def} + rt := map[string]ruleTimings{"r1": newRuleTimings(30*time.Second, 60)} + gt := globalTimings{} + pol := Policy{From: from, To: to} // NodataIsUnobservable defaults to false + + var polls []Poll + for ts := from; !ts.After(to); ts = ts.Add(30 * time.Second) { + polls = append(polls, Poll{RuleUID: "r1", GrafanaNow: ts, Found: true, Health: "nodata", LastEvaluation: ts}) + } + sentinel := to + + res, err := decide(Header{StartedAt: from.Add(-time.Hour)}, polls, &sentinel, defs, rt, gt, pol) + if err != nil { + t.Fatalf("err = %v, want nil: 96%% of the fleet runs no_data_state:OK and must not fail by default", err) + } + if res.Coverage["r1"].Unobservable { + t.Fatalf("Coverage[r1].Unobservable = true, want false by default") + } +} + +// --- F1/F2 regressions: preexisting is decided by ActiveAt, not poll timing --- + +// TestClassifyRule_OnsetBetweenFromAndFirstPollIsNewlyBadNotRecovered pins +// F1: an instance whose true onset (ActiveAt) falls strictly inside the +// window — even though the first poll that happens to observe it already +// shows it bad — must never be treated as preexisting. If it then clears, +// the plan requires newly_bad (exit 1), not recovered (exit 0). +func TestClassifyRule_OnsetBetweenFromAndFirstPollIsNewlyBadNotRecovered(t *testing.T) { + from := time.Date(2026, 1, 1, 0, 0, 0, 0, time.UTC) + to := from.Add(10 * time.Minute) + onset := from.Add(1 * time.Minute) // the true onset, strictly after `from` + firstPoll := from.Add(2 * time.Minute) // the first poll that happens to observe it + clearAt := from.Add(5 * time.Minute) + def := Definition{UID: "r1", Title: "R1"} + key := instanceKey(lbl("a")) + + polls := []Poll{ + abnormalPoll("r1", firstPoll, StateFiring, lbl("a"), onset), + clearedPoll("r1", clearAt, key), + quietPoll("r1", to), + } + outcome, badFor, viols := classifyRule(def, polls, from, to, defaultBad, PreexistingFailUnlessRecovered) + if outcome != OutcomeNewlyBad { + t.Fatalf("outcome = %v, want newly_bad: the onset is after `from`, so it is not preexisting even though "+ + "the FIRST in-window poll already observes it bad (F1)", outcome) + } + if len(viols) != 1 || viols[0].Outcome != OutcomeNewlyBad { + t.Fatalf("viols = %+v, want one newly_bad violation: a policy=fail-unless-recovered default must still fail this", viols) + } + if want := clearAt.Sub(onset); badFor != want { + t.Fatalf("badFor = %v, want %v: BadFor must count from the true onset, not from `from` (F1's overcount bug)", badFor, want) + } +} + +// TestClassifyRule_OnsetJustBeforeFromIsPreexisting is the mirror check: an +// onset at or before `from` (even if the first poll is later) is genuinely +// preexisting and, if it clears, is `recovered`. +func TestClassifyRule_OnsetJustBeforeFromIsPreexisting(t *testing.T) { + from := time.Date(2026, 1, 1, 0, 0, 0, 0, time.UTC) + to := from.Add(10 * time.Minute) + onset := from.Add(-time.Minute) + firstPoll := from.Add(2 * time.Minute) + clearAt := from.Add(5 * time.Minute) + def := Definition{UID: "r1", Title: "R1"} + key := instanceKey(lbl("a")) + + polls := []Poll{ + abnormalPoll("r1", firstPoll, StateFiring, lbl("a"), onset), + clearedPoll("r1", clearAt, key), + quietPoll("r1", to), + } + outcome, badFor, viols := classifyRule(def, polls, from, to, defaultBad, PreexistingFailUnlessRecovered) + if outcome != OutcomeRecovered { + t.Fatalf("outcome = %v, want recovered: the onset is at/before `from`, genuinely preexisting", outcome) + } + if len(viols) != 0 { + t.Fatalf("viols = %+v, want none: default policy passes a recovered preexisting instance", viols) + } + if want := clearAt.Sub(from); badFor != want { + t.Fatalf("badFor = %v, want %v: a preexisting episode's BadFor is clamped to window-open, not backdated past it", badFor, want) + } +} + +// TestClassifyRule_SkewTranslatesActiveAtAcrossTheWindowBoundary pins F2: a +// poll carrying a nonzero skew must have its ActiveAt (and GrafanaNow) +// translated to the runner domain before comparing against `from` — a raw, +// untranslated comparison would land on the wrong side of the F1 boundary +// check. +func TestClassifyRule_SkewTranslatesActiveAtAcrossTheWindowBoundary(t *testing.T) { + from := time.Date(2026, 1, 1, 0, 0, 0, 0, time.UTC) + to := from.Add(10 * time.Minute) + def := Definition{UID: "r1", Title: "R1"} + + // Grafana's clock reads 90s ahead of the runner's (skew = +90s). The + // poll's raw GrafanaNow/ActiveAt both sit 90s past `from` in Grafana's + // domain, but translate to exactly `from` in the runner domain — genuinely + // preexisting once translated, and wrongly "newly_bad" if the skew is + // ignored. + skew := 90 * time.Second + rawActiveAt := from.Add(skew) + poll := Poll{ + RuleUID: "r1", GrafanaNow: from.Add(skew), Found: true, Health: "ok", + LastEvaluation: from.Add(skew), SkewMS: skew.Milliseconds(), + Abnormal: []Instance{{Labels: lbl("a"), State: StateFiring, ActiveAt: rawActiveAt}}, + } + stillBad := poll + stillBad.GrafanaNow = to.Add(skew) + stillBad.LastEvaluation = to.Add(skew) + + outcome, badFor, _ := classifyRule(def, []Poll{poll, stillBad}, from, to, defaultBad, PreexistingFailUnlessRecovered) + if outcome != OutcomePersistentlyBad { + t.Fatalf("outcome = %v, want persistently_bad: a +90s skew must translate ActiveAt back to exactly `from` (F2)", outcome) + } + if badFor != to.Sub(from) { + t.Fatalf("badFor = %v, want the full window %v", badFor, to.Sub(from)) + } +} + +// --- F4: InstanceLabels must survive a timeline first created by a bare marker --- + +func TestClassifyRule_LabelsSurviveWhenTimelineStartsFromAClearedMarker(t *testing.T) { + from := time.Date(2026, 1, 1, 0, 0, 0, 0, time.UTC) + to := from.Add(10 * time.Minute) + def := Definition{UID: "r1", Title: "R1"} + key := instanceKey(lbl("a")) + newOnset := from.Add(5 * time.Minute) + + polls := []Poll{ + // The very first mention of this key is a bare Cleared marker (its + // prior bad episode, if any, started before the window) — no labels + // travel with a Cleared/Vanished event. + clearedPoll("r1", from.Add(1*time.Minute), key), + abnormalPoll("r1", newOnset, StateFiring, lbl("a"), newOnset), + quietPoll("r1", to), + } + _, _, viols := classifyRule(def, polls, from, to, defaultBad, PreexistingFailUnlessRecovered) + if len(viols) != 1 { + t.Fatalf("viols = %+v, want exactly one newly_bad violation", viols) + } + if viols[0].InstanceLabels == nil || viols[0].InstanceLabels["instance"] != "a" { + t.Fatalf("InstanceLabels = %+v, want {instance: a}: labels must backfill even though the "+ + "timeline was first created by a label-less Cleared marker (F4)", viols[0].InstanceLabels) + } +} + +// TestClassifyRule_ViolationFieldsArePrecise pins FirstSeen/ClearedAt exactly, +// not just that a violation exists (F7). +func TestClassifyRule_ViolationFieldsArePrecise(t *testing.T) { + from := time.Date(2026, 1, 1, 0, 0, 0, 0, time.UTC) + to := from.Add(10 * time.Minute) + onset := from.Add(2 * time.Minute) + clearAt := from.Add(3 * time.Minute) + def := Definition{UID: "r1", Title: "R1"} + + polls := []Poll{ + quietPoll("r1", from), + abnormalPoll("r1", onset, StateFiring, lbl("a"), onset), + clearedPoll("r1", clearAt, instanceKey(lbl("a"))), + quietPoll("r1", to), + } + _, _, viols := classifyRule(def, polls, from, to, defaultBad, PreexistingFailUnlessRecovered) + if len(viols) != 1 { + t.Fatalf("viols = %+v, want exactly one violation", viols) + } + v := viols[0] + if !v.FirstSeen.Equal(onset) { + t.Fatalf("FirstSeen = %v, want %v", v.FirstSeen, onset) + } + if !v.ClearedAt.Equal(clearAt) { + t.Fatalf("ClearedAt = %v, want %v", v.ClearedAt, clearAt) + } + if v.InstanceLabels["instance"] != "a" { + t.Fatalf("InstanceLabels = %+v, want {instance: a}", v.InstanceLabels) + } +} + +// TestClassifyRule_ClearedEventPastWindowEndClampsToWindowEnd pins the +// episode.end clamp: inWindowPolls admits a poll up to its own skew bound +// past windowEnd (§16's widened membership test), so a genuine Cleared event +// on such a poll must not leave the episode extending beyond windowEnd. +func TestClassifyRule_ClearedEventPastWindowEndClampsToWindowEnd(t *testing.T) { + from := time.Date(2026, 1, 1, 0, 0, 0, 0, time.UTC) + to := from.Add(10 * time.Minute) + def := Definition{UID: "r1", Title: "R1"} + key := instanceKey(lbl("a")) + + bound := 30 * time.Second + clearedAt := to.Add(20 * time.Second) // past windowEnd, but within the skew bound + + polls := []Poll{ + abnormalPoll("r1", from, StateFiring, lbl("a"), from.Add(-time.Hour)), + { + RuleUID: "r1", GrafanaNow: clearedAt, Found: true, Health: "ok", + SkewBoundMS: bound.Milliseconds(), Cleared: []string{key}, + }, + } + outcome, badFor, _ := classifyRule(def, polls, from, to, defaultBad, PreexistingFailUnlessRecovered) + if outcome != OutcomeRecovered { + t.Fatalf("outcome = %v, want recovered", outcome) + } + if badFor != to.Sub(from) { + t.Fatalf("badFor = %v, want the window %v exactly: the episode end must clamp to windowEnd, "+ + "not extend to the late Cleared event's raw time", badFor, to.Sub(from)) + } +} + +// TestClassifyRule_CloseBeforeOpenClampsToZeroNotNegative pins the +// end-before-start clamp: two polls with different measured skews can +// translate so that a closing poll's runner-domain time lands before the +// opening poll's, which — unclamped — would feed mergeDurations a negative +// span. +func TestClassifyRule_CloseBeforeOpenClampsToZeroNotNegative(t *testing.T) { + from := time.Date(2026, 1, 1, 0, 0, 0, 0, time.UTC) + to := from.Add(10 * time.Minute) + onsetPoll := from.Add(5 * time.Minute) + closePoll := from.Add(6 * time.Minute) + closeSkew := 2 * time.Minute // translates closePoll back to from+4min, before onsetPoll's from+5min + def := Definition{UID: "r1", Title: "R1"} + key := instanceKey(lbl("a")) + + polls := []Poll{ + quietPoll("r1", from), + abnormalPoll("r1", onsetPoll, StateFiring, lbl("a"), onsetPoll), // skew 0 + { + RuleUID: "r1", GrafanaNow: closePoll, Found: true, Health: "ok", + SkewMS: closeSkew.Milliseconds(), Cleared: []string{key}, + }, + quietPoll("r1", to), + } + outcome, badFor, viols := classifyRule(def, polls, from, to, defaultBad, PreexistingFailUnlessRecovered) + if outcome != OutcomeNewlyBad { + t.Fatalf("outcome = %v, want newly_bad", outcome) + } + if badFor < 0 { + t.Fatalf("badFor = %v, want a non-negative duration even though the closing poll's translated "+ + "time landed before the opening poll's", badFor) + } + if badFor != 0 { + t.Fatalf("badFor = %v, want 0: the clamp collapses the inverted span to a zero-length episode", badFor) + } + if len(viols) != 1 { + t.Fatalf("viols = %+v, want one violation", viols) + } +} + +// --- mergeDurations --- + +func TestMergeDurations_OverlappingEpisodesCountOnce(t *testing.T) { + from := time.Date(2026, 1, 1, 0, 0, 0, 0, time.UTC) + eps := []episode{ + {start: from, end: from.Add(5 * time.Minute)}, + {start: from.Add(2 * time.Minute), end: from.Add(8 * time.Minute)}, // overlaps the first + {start: from.Add(20 * time.Minute), end: from.Add(21 * time.Minute)}, // disjoint + } + got := mergeDurations(eps) + want := 8*time.Minute + 1*time.Minute // [0,8) merged = 8m, plus the disjoint 1m + if got != want { + t.Fatalf("mergeDurations = %v, want %v: two simultaneously-bad instances must not double-count their overlap", got, want) + } +} + +func TestMergeDurations_Empty(t *testing.T) { + if got := mergeDurations(nil); got != 0 { + t.Fatalf("mergeDurations(nil) = %v, want 0", got) + } +} diff --git a/grafana-alertcheck/internal/gate/coverage.go b/grafana-alertcheck/internal/gate/coverage.go index 3d3704b90..008406033 100644 --- a/grafana-alertcheck/internal/gate/coverage.go +++ b/grafana-alertcheck/internal/gate/coverage.go @@ -2,7 +2,6 @@ package gate import ( "fmt" - "slices" "time" ) @@ -83,16 +82,12 @@ func proveCoverage(h Header, polls []Poll, sentinel *time.Time, t ruleTimings, d windowEnd := to.Add(grace) - var rulePolls []Poll - for _, p := range polls { - if p.RuleUID == def.UID { - rulePolls = append(rulePolls, p) - } - } - // Stable, not sort.Slice: two polls sharing a GrafanaNow (a coarse Date - // header, or a corrupted/replayed log) must not reorder nondeterministically - // in a function that promises to be pure. - slices.SortStableFunc(rulePolls, func(a, b Poll) int { return a.GrafanaNow.Compare(b.GrafanaNow) }) + // pollsForRule (classify.go) is the single filter+sort implementation for + // "select one rule's polls, stably ordered by GrafanaNow" — proveCoverage + // and classifyRule must never carry two independent copies of this + // selection, or one drifting from the other becomes exactly the kind of + // silent membership mismatch this file's checks exist to prevent. + rulePolls := pollsForRule(polls, def.UID) var res CoverageResult fail := func(reason UnobservableReason, note string) { @@ -275,7 +270,7 @@ func inWindowPolls(polls []Poll, from, windowEnd time.Time) []Poll { var out []Poll for _, p := range polls { bound := p.SkewBound() - runner := p.GrafanaNow.Add(-p.Skew()) + runner := runnerTime(p, p.GrafanaNow) if runner.Before(from.Add(-bound)) || runner.After(windowEnd.Add(bound)) { continue } @@ -305,7 +300,7 @@ func ruleHeartbeatGap(in []Poll, from, windowEnd time.Time) (largestGap time.Dur return windowEnd.Sub(from), from } - runnerOf := func(p Poll) time.Time { return p.GrafanaNow.Add(-p.Skew()) } + runnerOf := func(p Poll) time.Time { return runnerTime(p, p.GrafanaNow) } first := in[0] if gap := runnerOf(first).Sub(from) + first.SkewBound(); gap > largestGap { From 7d84d247390b5a6ba37e747314fa5ea1aab8be0c Mon Sep 17 00:00:00 2001 From: Bartek Tofel Date: Wed, 2 Sep 2026 13:04:18 +0200 Subject: [PATCH 23/43] chore: rename some vars + add unit tests --- grafana-alertcheck/internal/gate/classify.go | 20 +++++------ .../internal/gate/classify_test.go | 36 +++++++++++++++++++ 2 files changed, 46 insertions(+), 10 deletions(-) diff --git a/grafana-alertcheck/internal/gate/classify.go b/grafana-alertcheck/internal/gate/classify.go index 3f14b6be1..11ddc8ba2 100644 --- a/grafana-alertcheck/internal/gate/classify.go +++ b/grafana-alertcheck/internal/gate/classify.go @@ -502,7 +502,7 @@ func decide(h Header, polls []Poll, sentinel *time.Time, defs []Definition, var ( skippedRules []Definition - observedCount int + watchedCount int anyUnobservable bool unobservableNames []string ) @@ -517,7 +517,7 @@ func decide(h Header, polls []Poll, sentinel *time.Time, defs []Definition, }) continue } - observedCount++ + watchedCount++ t := rt[def.UID] cov := proveCoverage(h, polls, sentinel, t, def, pol.From, pol.To, gt.transitionGrace) @@ -558,17 +558,17 @@ func decide(h Header, polls []Poll, sentinel *time.Time, defs []Definition, // so a shortfall always produces at least one, even when no rule is // paused at all (an operator-supplied MinObserved that simply exceeds // what could ever be resolved). - counted := observedCount - var chargeable []Definition + counted := watchedCount + var attributable []Definition if pol.AllowPaused { counted += len(skippedRules) } else { - chargeable = skippedRules + attributable = skippedRules } if shortfall := minObserved - counted; shortfall > 0 { - charged := 0 - for _, def := range chargeable { - if charged >= shortfall { + attributed := 0 + for _, def := range attributable { + if attributed >= shortfall { break } // §12.1 requires the paused rule and --allow-paused both be @@ -579,9 +579,9 @@ func decide(h Header, polls []Poll, sentinel *time.Time, defs []Definition, Alert: def.Title, RuleUID: def.UID, Outcome: OutcomeSkipped, Note: "paused before the window opened; counts against --min-observed unless --allow-paused is set", }) - charged++ + attributed++ } - for ; charged < shortfall; charged++ { + for ; attributed < shortfall; attributed++ { // No named rule explains this part of the deficit — e.g. an // operator-supplied --min-observed above what could ever be // resolved. Note, not LastError: LastError is reporting-only diff --git a/grafana-alertcheck/internal/gate/classify_test.go b/grafana-alertcheck/internal/gate/classify_test.go index 5d773e241..7176a9167 100644 --- a/grafana-alertcheck/internal/gate/classify_test.go +++ b/grafana-alertcheck/internal/gate/classify_test.go @@ -761,6 +761,42 @@ func TestClassifyRule_ClearedEventPastWindowEndClampsToWindowEnd(t *testing.T) { } } +// TestClassifyRule_OnsetJustPastWindowEndIsNewlyBadNotClean pins the fail-closed +// reading of the upper boundary: an instance whose runner-domain onset lands +// only slightly past windowEnd (to + transitionGrace) is reachable at all only +// because inWindowPolls widens the boundary outward by the skew bound, so the +// gate cannot PROVE it belongs to the next window. It is charged as newly_bad — +// with BadFor truncated to zero — rather than silently forgiven as clean. +func TestClassifyRule_OnsetJustPastWindowEndIsNewlyBadNotClean(t *testing.T) { + from := time.Date(2026, 1, 1, 0, 0, 0, 0, time.UTC) + to := from.Add(10 * time.Minute) + grace := time.Minute + windowEnd := to.Add(grace) + def := Definition{UID: "r1", Title: "R1"} + + // A poll admitted only by its own skew bound: its GrafanaNow sits 20s past + // windowEnd, inside the 30s tolerance. It carries an instance whose onset + // is 10s past windowEnd — still "after the grace", but only by less than + // the measurement's own uncertainty. + bound := 30 * time.Second + poll := Poll{ + RuleUID: "r1", GrafanaNow: windowEnd.Add(20 * time.Second), Found: true, Health: "ok", + LastEvaluation: windowEnd.Add(20 * time.Second), SkewBoundMS: bound.Milliseconds(), + Abnormal: []Instance{{Labels: lbl("a"), State: StateFiring, ActiveAt: windowEnd.Add(10 * time.Second)}}, + } + + outcome, badFor, viols := classifyRule(def, []Poll{poll}, from, windowEnd, defaultBad, PreexistingFailUnlessRecovered) + if outcome != OutcomeNewlyBad { + t.Fatalf("outcome = %v, want newly_bad: an onset past windowEnd seen only via the skew bound must fail closed", outcome) + } + if badFor != 0 { + t.Fatalf("badFor = %v, want 0: the zero-length episode must truncate to the window end", badFor) + } + if len(viols) != 1 { + t.Fatalf("viols = %+v, want exactly one newly_bad violation", viols) + } +} + // TestClassifyRule_CloseBeforeOpenClampsToZeroNotNegative pins the // end-before-start clamp: two polls with different measured skews can // translate so that a closing poll's runner-domain time lands before the From 48393bc010266da09773e889afbb702457b1686c Mon Sep 17 00:00:00 2001 From: Bartek Tofel Date: Mon, 7 Sep 2026 16:43:30 +0200 Subject: [PATCH 24/43] chore: address code review comments --- grafana-alertcheck/internal/gate/classify.go | 9 +++++++-- 1 file changed, 7 insertions(+), 2 deletions(-) diff --git a/grafana-alertcheck/internal/gate/classify.go b/grafana-alertcheck/internal/gate/classify.go index 11ddc8ba2..48a312048 100644 --- a/grafana-alertcheck/internal/gate/classify.go +++ b/grafana-alertcheck/internal/gate/classify.go @@ -218,13 +218,18 @@ func classifyRule(def Definition, polls []Poll, from, windowEnd time.Time, badSt tl.badOpen = false } // onsetOf resolves a fresh episode's start: the instance's own ActiveAt, - // translated to the runner domain by this poll's skew, clamped so it - // never reads as starting before the window opened. + // translated to the runner domain by this poll's skew, clamped to + // [from, windowEnd] so closeEpisode never has to undo its own clamp on a + // start that already overran the window (§16: a poll admitted by the skew + // bound can carry an ActiveAt past windowEnd). onsetOf := func(p Poll, inst Instance) time.Time { start := runnerTime(p, inst.ActiveAt) if start.Before(from) { start = from } + if start.After(windowEnd) { + start = windowEnd + } return start } From 840959ca591d3a3dc1c5a0a88713cd2aca03363c Mon Sep 17 00:00:00 2001 From: Bartek Tofel Date: Mon, 31 Aug 2026 18:24:21 +0200 Subject: [PATCH 25/43] chore: implement phase 9 MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Invariant defended: H5/H7. The one question: can check report a pass over a window it did not prove? Check() is the I/O shell around the pure decide(). Single-step synthesizes the header and its own sentinel, so no mode flag reaches the pure layer. Log mode stops the recorder before the one full read. The header, not a definition re-resolved after the window closed, is the authority for what was paused when the window opened — it decides `skipped`, the drain set, and the transitionGrace max. The flock, not the pidfile, is the authority for whether a writer still exists. --- grafana-alertcheck/internal/gate/check.go | 970 +++++++++++++ .../internal/gate/check_test.go | 1216 +++++++++++++++++ .../internal/gate/check_unix.go | 36 + grafana-alertcheck/internal/gate/classify.go | 11 +- .../internal/gate/classify_test.go | 20 +- grafana-alertcheck/internal/gate/coverage.go | 31 +- grafana-alertcheck/internal/gate/flock.go | 20 + grafana-alertcheck/internal/gate/log.go | 115 +- grafana-alertcheck/internal/gate/schedule.go | 59 +- .../internal/gate/schedule_test.go | 64 + grafana-alertcheck/internal/gate/watch.go | 100 +- .../internal/gate/watch_daemon_test.go | 37 + 12 files changed, 2599 insertions(+), 80 deletions(-) create mode 100644 grafana-alertcheck/internal/gate/check.go create mode 100644 grafana-alertcheck/internal/gate/check_test.go create mode 100644 grafana-alertcheck/internal/gate/check_unix.go diff --git a/grafana-alertcheck/internal/gate/check.go b/grafana-alertcheck/internal/gate/check.go new file mode 100644 index 000000000..6c3bf8805 --- /dev/null +++ b/grafana-alertcheck/internal/gate/check.go @@ -0,0 +1,970 @@ +package gate + +import ( + "context" + "errors" + "fmt" + "io" + "os" + "sort" + "strings" + "time" +) + +// Obligations this phase leaves for P10, carried forward the way P6 and P7 +// carried theirs so a later review has something concrete to check against: +// +// - Exit codes are the CLI's (§20.3, §9.1, H6/H7). Check returns +// (Result, error) and nothing else: err != nil is exit 2 unconditionally, +// never 0 and never 1, even alongside real violations; len(Violations) > 0 +// with err == nil is exit 1; both empty is exit 0. Check deliberately does +// not return a code, because a code is a presentation decision and the +// library must not make it. +// - §12.1 wants the paused rule AND --allow-paused both named to the user. +// decide names the rule in the shortfall Violation's Note (classify.go); +// the flag hint belongs to the CLI's renderer. +// - Config.Notes carries the running commentary (§13.2's planned run time, +// the countdown, the blind-interval warning). §20.2 puts the human output +// on stderr and reserves stdout for --output json, so the CLI must pass +// stderr here. +// - Check never reads the environment. GRAFANA_URL and GRAFANA_TOKEN are +// read by the CLI and passed in as fields, and the token must never reach +// a *flag.FlagSet (§20.2). + +// countdownEvery is how often the collection loop reports what it is waiting +// for (§13.2: "then print a countdown at regular intervals"). A silent wait is +// indistinguishable from a hung process, and the wait after `to` is the +// longest silence in the whole run. +const countdownEvery = 30 * time.Second + +// recorderStopTimeout bounds §4.4 step 3, the wait for the recorder's exit. +// Not in the source plan's table of values — a judgment call, on the same +// reasoning as childReadyTimeout (P6): everything the recorder does after +// SIGTERM is local (finish the in-flight write, append the sentinel, fsync) +// and an in-flight poll aborts through the child's own context, so the real +// figure is milliseconds. Loose enough for an overloaded runner, and a +// timeout is a hard error rather than a longer wait — a log a writer may +// still hold cannot be read at all (§4.4 step 4). +const recorderStopTimeout = 30 * time.Second + +// recorderStopPoll is how often that wait re-checks the pid. There is no +// wait(2) available: the recorder is a detached session leader, not this +// process's child (P6), so its exit can only be observed by polling. +const recorderStopPoll = 100 * time.Millisecond + +// Config is check's whole input. It is the CLI's view of a run, and it is +// deliberately wider than Policy: Policy is the narrowed, pure-layer subset +// that reaches decide (classify.go), and the token is the field that must +// never cross that line. +type Config struct { + // URL and Token are the connection details, read from the environment by + // the CLI and never registered as flags (§20.2). Token never enters the + // pure layer, an error string, or a Result. + URL, Token string + + // Alerts is REQUIRED in single-step mode and must be EMPTY in log mode: + // with a log, the header IS the alert set (§19.1 step 3), and there is + // nothing to compare a second list against. Both directions are encoded, + // resolving the source plan's §19.1 step 1 / step 3 contradiction. + Alerts []string + Folder string + + States []State + Preexisting PreexistingPolicy + MinObserved int + AllowPaused bool + NodataIsUnobservable bool + + // From is the moment the deploy finished and To is the end of the work + // (§7). They are different moments and both come from the work. In + // recorder mode an absent From is a hard error; in single-step mode it + // falls back to the start of this step, with the blind-interval warning + // §4.2 requires. + From, To time.Time + + // Log is the path of a recording made by watch; "" selects single-step + // mode. PidFile defaults to .pid, the convention watch's parent + // writes (P6) and the only way check can reach the recorder it must stop + // before it may read the log (§4.4 steps 1-4). + Log string + PidFile string + + // There is deliberately NO PollEvery here, and `check` has no + // --poll-interval flag. In log mode the cadence comes from the header — + // the cadence the recording actually used — and a second authority would + // let an operator silently widen maxGap over evidence that was recorded at + // a different rate (P5, "two authorities"); in single-step mode the same + // process records and classifies, so §5's default is the only cadence + // there is. + Concurrency int + Clock Clock + + // Notes is where the shell prints what an operator has to see while the + // run is in progress: the planned run time, the grace and its source, the + // countdown, the blind-interval warning. nil discards them. The library + // renders no table — the CLI owns presentation (§20.2). + Notes io.Writer +} + +func (cfg Config) withDefaults() Config { + if cfg.Clock == nil { + cfg.Clock = SystemClock{} + } + if cfg.Notes == nil { + cfg.Notes = io.Discard + } + if cfg.Concurrency < 1 { + cfg.Concurrency = 1 + } + if cfg.PidFile == "" && cfg.Log != "" { + cfg.PidFile = cfg.Log + ".pid" + } + return cfg +} + +// namedAlerts returns the alert names that survive §17.3's trim-and-discard, +// so validation counts what Resolve will actually see rather than what the +// caller happened to pass (a file ending in a newline yields an empty line). +func (cfg Config) namedAlerts() []string { + out := make([]string, 0, len(cfg.Alerts)) + for _, a := range cfg.Alerts { + if strings.TrimSpace(a) != "" { + out = append(out, a) + } + } + return out +} + +// Check is the I/O shell: HTTP, signals, the pidfile, file reads, the +// countdown print. Every correctness question it touches is answered +// elsewhere — by proveCoverage and decide, which are pure — and that split is +// the most important seam in the project (§2). Check therefore needs two +// integration tests; decide carries the suite. +// +// H7 governs the return: a pass is exactly len(Violations) == 0 && err == nil. +// Every error path below leaves err non-nil, and no path anywhere in this file +// converts an error into an empty Result with a nil error. +func Check(ctx context.Context, cfg Config) (Result, error) { + cfg = cfg.withDefaults() + if err := cfg.validate(); err != nil { + return Result{}, err + } + // The Source is built here and injected into check() so every behaviour + // below is testable against a scripted fake — the same seam prepareWatch + // uses (P6), and the reason this file needs no test-only setter. + return check(ctx, cfg, NewHTTPSource(cfg.URL, cfg.Token, cfg.Clock)) +} + +// validate is §19.1 step 1. It runs before any network call, so a +// configuration mistake costs nothing and, more importantly, is never +// discovered after a ten-minute wait. +func (cfg Config) validate() error { + if cfg.URL == "" { + return errors.New("check: no grafana url") + } + if cfg.To.IsZero() { + return errors.New("check: no `to`: the end of the window is required (§7)") + } + + named := cfg.namedAlerts() + if cfg.Log == "" { + // §19.1 step 1: an empty Alerts is an error — but only without a log. + if len(named) == 0 { + return errors.New("check: no alert names given and no recorded log to take them from") + } + } else if len(named) > 0 { + // §19.1 step 3, the other direction: the alert set comes from the log. + // Accepting both would mean reconciling two sets, which is the subset + // arithmetic the source plan removes by making the log the one source. + return fmt.Errorf("check: --alerts is refused with a recorded log: %s already names the alert set it recorded (§19.1 step 3)", cfg.Log) + } + + now := cfg.Clock.Now() + // from mirrors what check() will use, so the two window checks below judge + // the window that will really be classified. The fallback is not written + // back into cfg: check() re-reads the clock at the same point, and one + // authority for that value is better than two that could disagree. + from := cfg.From + switch { + case from.IsZero() && cfg.Log != "": + // §7, and never a warning-and-continue: falling back to the start of + // the check step reinstates exactly the blind interval the recorder + // exists to remove, which is the fail-open shape this design refuses. + return errors.New("check: no `from` in recorder mode: the deploy step must emit a completion timestamp (§7)") + case from.IsZero(): + // Single-step only. The caller sees the resulting blind interval named + // exactly, once the first observation has fixed its end (§4.2). + from = now + } + + if cfg.To.Before(from) { + return fmt.Errorf("check: `to` %s is before `from` %s", cfg.To.Format(time.RFC3339), from.Format(time.RFC3339)) + } + if from.After(now.Add(fromFutureTolerance)) { + return fmt.Errorf("check: `from` %s is more than %s ahead of this runner's clock %s (§7)", + from.Format(time.RFC3339), fromFutureTolerance, now.Format(time.RFC3339)) + } + + // A `to` already in the past is not a special mode WITH a log (§7, §24.3): + // the collection loop's condition is simply already true and the evidence + // is classified immediately. Without one it is a different thing entirely + // — a request to prove a window that nothing observed. Refusing it is not + // pedantry: the coverage window would end before the first observation, + // every heartbeat gap inside it would measure negative, and the run would + // report a proved window it never saw. + if cfg.Log == "" && !cfg.To.After(now) { + return fmt.Errorf("check: `to` %s has already passed and there is no recorded log: a window that ended before check started can only be classified from a recording (§4.2)", + cfg.To.Format(time.RFC3339)) + } + return nil +} + +// check is Check with the Source injected. Its body is §19.1 steps 1-9, one +// commented block each and in that order, so a review can diff it against the +// source plan line by line. +func check(ctx context.Context, cfg Config, src Source) (Result, error) { + // ---- §19.1 step 1 — validate the configuration. ----------------------- + // Done by Check before this function is reached, except for the one part + // that needs a clock reading kept for later: the single-step fallback for + // an absent `from`. + from := cfg.From + if from.IsZero() { + from = cfg.Clock.Now() + fmt.Fprintf(cfg.Notes, "note: no `from` given; the window starts at the start of this step, %s (§4.2)\n", + from.Format(time.RFC3339)) + } + + // ---- §19.1 step 2 — resolve the definitions from the ruler API. ------- + // Unconditional, in BOTH modes. A log's header supplies the alert set as + // UIDs and the recording facts, never the rule facts: `for`, + // intervalSeconds and Kind always come from a fresh ruler read, which is + // why LoggedRule.ForSeconds is never converted back into a Definition. + version, err := src.Version(ctx) + if err != nil { + return Result{}, fmt.Errorf("read grafana version: %w", err) + } + if err := CheckGrafanaVersion(version); err != nil { + return Result{}, err + } + allDefs, err := src.Definitions(ctx) + if err != nil { + // §19.3 case 2: resolution of the definitions failed. + return Result{}, fmt.Errorf("read rule definitions: %w", err) + } + + // ---- §19.1 step 3 — with a log, validate its identity. ---------------- + // The header is read early — line 1 only, the one line a writer can never + // change (ReadLogHeader) — so a wrong URL or a rule that no longer + // resolves fails closed NOW rather than after the whole window has + // elapsed. It is advisory: the authoritative header comes from the single + // full ReadLog in step 6, after the writer has exited, and the identity is + // validated again against that one. + var ( + resolved []Definition + notes []string + earlyHdr Header + logHasHdr bool + rt map[string]ruleTimings + gt globalTimings + timingNote []string + ) + if cfg.Log != "" { + earlyHdr, err = ReadLogHeader(cfg.Log) + if err != nil { + // §19.3 case 3. + return Result{}, fmt.Errorf("log identity: %w", err) + } + logHasHdr = true + resolved, notes, err = resolveFromLog(allDefs, earlyHdr, cfg) + } else { + resolved, notes, err = Resolve(allDefs, cfg.namedAlerts(), cfg.Folder) + } + if err != nil { + return Result{}, err + } + if len(resolved) == 0 { + // Reachable only from a header with an empty rule list. Left to run, + // MinObserved would default to zero, no rule would be judged, and the + // gate would return a pass over nothing at all. + return Result{}, fmt.Errorf("check: no rules to classify") + } + for _, n := range notes { + fmt.Fprintf(cfg.Notes, "note: %s\n", n) + } + + // ---- §19.1 step 4 — derive the timings, print the plan, fit the budget. + if logHasHdr { + // The header is the authority for the cadence actually recorded at; + // re-deriving it from defs would compare gaps recorded at an override + // cadence against thresholds computed from the default — fail-open in + // the faster-override direction (P5). + rt, gt, err = DeriveTimingsFromLog(earlyHdr, resolved) + if err != nil { + return Result{}, fmt.Errorf("log identity: %w", err) + } + } else { + rt, gt, timingNote = DeriveTimings(resolved, 0) + for _, n := range timingNote { + fmt.Fprintf(cfg.Notes, "note: %s\n", n) + } + } + summary, warning := StartupSummary(from, cfg.To, gt) + fmt.Fprintln(cfg.Notes, summary) + if warning != "" { + fmt.Fprintf(cfg.Notes, "warning: %s\n", warning) + } + + // The measurement pass and the budget check belong to single-step mode + // alone (§5.2): in recorder mode watch already took one observation of + // every rule and checked the budget against those measured latencies + // before it detached, and repeating it here would spend a second poll of + // every rule to re-answer a question already answered. + var ( + header Header + initial []Poll + reducer = NewReducer() + ) + if !logHasHdr { + // StartedAt is fixed before the pass rather than after it, so the + // interval it claims to have observed can only be wider than the one + // it really saw — and the first heartbeat's own boundary gap (P7 check + // 3) is what proves that interval, not this timestamp. + startedAt := cfg.Clock.Now() + active := activeRules(resolved) + var measured map[string]time.Duration + initial, measured, err = firstObservations(ctx, src, active, reducer, cfg.Concurrency, cfg.Notes) + if err != nil { + return Result{}, err + } + if err := CheckBudget(activeTimingsOf(active, rt), measured, cfg.Concurrency); err != nil { + return Result{}, err + } + + // Single-step synthesis — how the pure layer stays unconditional (§2). + // The shell builds the Header and later stamps the sentinel itself, so + // P7 checks 1 and 2 run exactly as they do over a recording and no + // mode flag ever reaches proveCoverage or decide. + header = Header{ + SchemaVersion: LogSchemaVersion, + URL: cfg.URL, + GrafanaVersion: version, + StartedAt: startedAt, + Rules: loggedRules(resolved, rt), + } + if from.Before(startedAt) { + // §4.2/§22.4's declared blind interval: in single-step mode this + // is a warning and a pass, and ONLY here. Recorder mode keeps P7 + // check 2 strict (§22.9), because there the recorder was supposed + // to be watching and the gap means it was not. + fmt.Fprintf(cfg.Notes, "warning: cannot see [%s, %s) — %s before the first observation; the window is classified from %s (§4.2)\n", + from.Format(time.RFC3339), startedAt.Format(time.RFC3339), + startedAt.Sub(from).Round(time.Second), startedAt.Format(time.RFC3339)) + from = startedAt + } + } + + // ---- §19.1 step 5 — apply MinObserved. -------------------------------- + // Its default is the resolved rule count AFTER the collapse (§17.3), which + // is len(resolved) by construction. decide defaults it identically; it is + // resolved here as well so the value the run will judge against is printed + // before the wait rather than inferred from the verdict afterwards. + minObserved := cfg.MinObserved + if minObserved == 0 { + minObserved = len(resolved) + } + fmt.Fprintf(cfg.Notes, "min-observed: %d of %d resolved rule(s)\n", minObserved, len(resolved)) + + // ---- §19.1 step 6 — collect the evidence. ----------------------------- + // Collect ONLY. No classification happens here and there is no early exit, + // even once a violation is certain (H5, §19.2): the loop always runs to + // to + transitionGrace, which is what makes "did the early exit lose the + // coverage proof?" a question that cannot be asked. + windowEnd := cfg.To.Add(gt.transitionGrace) + + var poller *livePoller + if !logHasHdr { + poller = newLivePoller(src, reducer, activeRules(resolved), rt, cfg.Concurrency, cfg.Clock.Now()) + } + collected, err := collectUntil(ctx, cfg, windowEnd, poller) + if err != nil { + // §19.3 case 1: the failure limit was exceeded (retryTransport already + // gave every transient failure its backoff), or the context ended. + // Nothing collected is classified — the count is there so an operator + // can tell a run that failed at once from one that failed at minute + // nine. + return Result{}, fmt.Errorf("collect evidence after %d poll(s): %w", len(collected), err) + } + + var ( + polls []Poll + sentinel *time.Time + ) + if logHasHdr { + // §4.4 steps 2-4, in this order and no other: signal the writer, wait + // for its exit, and only THEN read the log once. A log read while a + // writer can still append can only yield a shorter window than the one + // that was actually recorded. + heldLog, err := stopRecorder(ctx, cfg) + if err != nil { + return Result{}, err + } + header, polls, sentinel, err = ReadLog(cfg.Log) + // The lock stays held across the read, so no writer can appear between + // the proof that there was none and the read itself. Released here + // rather than deferred: everything past this point works from bytes + // already in memory, and the drain wait below can take minutes. + _ = heldLog.Close() + if err != nil { + return Result{}, err + } + // The authoritative header, validated the same way the advisory one + // was — and its result is KEPT. Everything from here on judges the + // header ReadLog returned, so nothing downstream rests on the advisory + // read having been right. That read is what it claims to be: a + // fail-fast, and no part of the verdict depends on it. + resolved, _, err = resolveFromLog(allDefs, header, cfg) + if err != nil { + return Result{}, err + } + // rt is re-derived because it depends on the header: PollEverySeconds + // is the one load-bearing value the advisory read supplied. windowEnd + // is deliberately NOT recomputed from the gt this returns: the + // collection loop has already stopped at the earlier value, and moving + // the end of the window afterwards would prove a window this run did + // not collect. + if rt, gt, err = DeriveTimingsFromLog(header, resolved); err != nil { + return Result{}, fmt.Errorf("log identity: %w", err) + } + } else { + polls = make([]Poll, 0, len(initial)+len(collected)) + polls = append(polls, initial...) + polls = append(polls, collected...) + // The shell stamps the sentinel itself, when the collection loop + // exits: by construction that is at or after to + transitionGrace, so + // P7 check 1 passes for the same reason a clean recorder stop does, + // and for no other. + stoppedAt := cfg.Clock.Now() + sentinel = &stoppedAt + } + + // ---- §19.1 step 7 — the drain wait. ----------------------------------- + // The last instance of the liveness check (§14.6): did this rule evaluate + // through the end of the window? It is I/O and it is deliberately NOT part + // of proveCoverage — adding it there would put HTTP inside the pure layer + // and destroy the seam §2 depends on. + drained, err := drainWait(ctx, cfg, src, resolved, header.pausedAtStart(), rt, polls, windowEnd, gt.drainTimeout) + if err != nil { + return Result{}, err + } + + // ---- §19.1 step 8 — classify. ----------------------------------------- + pol := Policy{ + States: cfg.States, + Preexisting: cfg.Preexisting, + MinObserved: minObserved, + AllowPaused: cfg.AllowPaused, + NodataIsUnobservable: cfg.NodataIsUnobservable, + From: from, + To: cfg.To, + } + result, decideErr := decide(header, polls, sentinel, resolved, rt, gt, pol) + result, drainErr := mergeDrainTimeouts(result, drained) + + // ---- §19.1 step 9 — return Result. ------------------------------------ + // Both errors are joined rather than one shadowing the other: each names + // rules the other does not, and on exit 2 that list IS the answer to + // "why". H7 needs only that err be non-nil when either fired. + return result, errors.Join(decideErr, drainErr) +} + +// resolveFromLog turns a log header into the resolved definitions, and is +// §19.1 step 3's identity check in practice. Three things are verified: the +// URL matches, the schema version matches (ReadLog/ReadLogHeader own that), +// and every header UID still resolves against the fresh ruler read. The alert +// set is TAKEN from the log, never compared — with Alerts required empty in +// log mode there is nothing to compare it against, and §22.4's "different +// alert set" refusal is exactly this URL-and-UID failure. +// +// Resolving through Resolve, by uid:, rather than by a private lookup, keeps +// one implementation of §17: a header naming a recording or datasource-managed +// rule gets the same specific refusal an operator would, and a header naming +// the same UID twice collapses with a note (DeriveTimingsFromLog rejects that +// case outright, so the note is belt and braces). +// +// Only the header-to-defs direction needs checking. The opposite direction +// cannot fail here: resolved is BUILT from the header, so no resolved +// definition can be absent from it. +func resolveFromLog(allDefs []Definition, h Header, cfg Config) ([]Definition, []string, error) { + if h.URL != cfg.URL { + return nil, nil, fmt.Errorf("log identity: %s recorded url %q but this run is configured for %q (§19.1 step 3)", + cfg.Log, h.URL, cfg.URL) + } + names := make([]string, 0, len(h.Rules)) + for _, lr := range h.Rules { + names = append(names, "uid:"+lr.UID) + } + resolved, notes, err := Resolve(allDefs, names, "") + if err != nil { + return nil, nil, fmt.Errorf("log identity: %s names a rule that no longer resolves: %w (§19.1 step 3)", cfg.Log, err) + } + return resolved, notes, nil +} + +// activeRules drops the rules whose DEFINITION says paused. They are skipped +// (§12): never polled, never waited for, and reported from the definitions +// alone — a skipped rule has no poll records at all, so it has no heartbeats +// to prove and no IsPaused poll to detect (P6 deviation 4, and the obligation +// it left P7/P8). +func activeRules(defs []Definition) []Definition { + out := make([]Definition, 0, len(defs)) + for _, d := range defs { + if !d.IsPaused { + out = append(out, d) + } + } + return out +} + +// activeTimingsOf narrows the timings map to the rules that will actually be +// polled, which is what §5.2's budget is spent on: a skipped rule consumes +// none of the capacity, so counting it would refuse schedules that fit. +func activeTimingsOf(active []Definition, rt map[string]ruleTimings) map[string]ruleTimings { + out := make(map[string]ruleTimings, len(active)) + for _, d := range active { + out[d.UID] = rt[d.UID] + } + return out +} + +// livePoller is single-step mode's collection engine: the same per-rule +// scheduler and the same Reducer the recorder uses (P4/P6), writing into +// memory instead of a log. Log mode has none — the recorder is doing this +// work in another process — and collectUntil takes a nil poller for it. +type livePoller struct { + src Source + reducer *Reducer + sched *Scheduler + titles map[string]string // uid -> title: poll by title, select by UID (§14.5) + concurrency int +} + +func newLivePoller(src Source, reducer *Reducer, active []Definition, rt map[string]ruleTimings, + concurrency int, now time.Time) *livePoller { + + titles := make(map[string]string, len(active)) + cadence := make(map[string]time.Duration, len(active)) + for _, d := range active { + titles[d.UID] = d.Title + cadence[d.UID] = rt[d.UID].pollEvery + } + return &livePoller{ + src: src, + reducer: reducer, + sched: NewScheduler(cadence, now), + titles: titles, + concurrency: concurrency, + } +} + +// poll runs one round of due rules and returns every poll that succeeded, +// alongside the first failure. +// +// The successes are NOT kept for the reason watchLoopConfig.pollBatch keeps +// its own: those go into a durable log that a later check will read, so +// dropping one would turn a single rule's transport failure into a coverage +// gap for the others. Here there is no later reader. A terminal failure +// during collection is exit 2 (§19.3 case 1) and check discards the whole +// collection, so these come back only to let the error say how far the run +// got before it stopped — which is the one part of it an operator can act on. +func (p *livePoller) poll(ctx context.Context, uids []string) ([]Poll, error) { + observed, obsErr := observeAll(ctx, p.src, p.titles, uids, p.concurrency) + out := make([]Poll, 0, len(uids)) + for _, uid := range uids { + obs, ok := observed[uid] + if !ok { + continue + } + out = append(out, p.reducer.Reduce(uid, obs)) + } + return out, obsErr +} + +// collectUntil is §19.1 step 6's loop, shared by both modes. With a poller it +// polls each rule on its own cadence; with nil it only waits, because in +// recorder mode the evidence is being written by another process. Both print +// the same countdown, because both are the same silence to an operator +// watching a job (§13.2). +// +// It never classifies and never exits early (H5). +func collectUntil(ctx context.Context, cfg Config, deadline time.Time, p *livePoller) ([]Poll, error) { + var ( + polls []Poll + lastPrint time.Time + ) + for { + now := cfg.Clock.Now() + if !now.Before(deadline) { + return polls, nil + } + if lastPrint.IsZero() || now.Sub(lastPrint) >= countdownEvery { + fmt.Fprintf(cfg.Notes, "collecting: %s until the window closes at %s\n", + deadline.Sub(now).Round(time.Second), deadline.Format(time.RFC3339)) + lastPrint = now + } + + wait := min(deadline.Sub(now), countdownEvery) + if p != nil { + // Mark before polling, against the batch's own `now`: the next poll + // is one cadence after this one was DUE, not after it returned, so + // request latency cannot make the heartbeat spacing drift towards + // maxGap (the same rule watchLoop follows). + due := p.sched.Due(now) + for _, uid := range due { + p.sched.Mark(uid, now) + } + if len(due) > 0 { + batch, err := p.poll(ctx, due) + polls = append(polls, batch...) + if err != nil { + return polls, err + } + } + if next, ok := p.sched.earliestDue(); ok { + wait = min(wait, next.Sub(cfg.Clock.Now())) + } + } + + select { + case <-ctx.Done(): + return polls, ctx.Err() + case <-cfg.Clock.After(max(wait, 0)): + } + } +} + +// stopRecorder is §4.4 steps 2 and 3. Nothing here is best-effort: the log may +// not be read until the writer has provably gone, so every failure to reach +// that state is a hard error. +// +// It returns the log held under an exclusive flock. The caller must keep that +// file open across ReadLog and close it afterwards — the lock is the proof +// that no writer exists, and holding it across the read also shuts out a new +// one appearing between the proof and the read. +// +// Two authorities, and only one of them is evidence: +// +// - The PIDFILE says whether a recording was ever started. An absent or +// unparseable one is the load-bearing case, and P6's obligation on this +// phase: it must never read as "there was nothing to stop". The parent +// writes the pidfile only AFTER the child reports that it holds the log +// and is polling, and removes it on every failing path, so a missing one +// means watch failed and this run has no evidence at all. +// - The FLOCK says whether a writer exists RIGHT NOW. Nothing removes the +// pidfile when a recorder exits cleanly — the parent has long returned and +// the child never learns the path — so after a --until run, a supported +// flow, the pidfile names a pid nobody owns. Signalling it would SIGTERM +// whatever same-user process inherited that pid. The kernel releases a +// flock when its holder exits, crash included, so the lock cannot go +// stale that way. +// +// So: read the pidfile to learn that a recording happened, then ask the lock +// whether it is still running, and signal only if it is. +func stopRecorder(ctx context.Context, cfg Config) (*os.File, error) { + pid, err := ReadPidFile(cfg.PidFile) + if err != nil { + return nil, fmt.Errorf("cannot stop the recorder: %w; a pidfile is written only once a recorder reports that it is running, so an unreadable one means the recording never started (§4.4)", err) + } + + log, err := os.Open(cfg.Log) + if err != nil { + return nil, fmt.Errorf("open %s to check for a writer: %w", cfg.Log, err) + } + + held, err := tryLockExclusive(log) + if err != nil { + log.Close() + return nil, err + } + if held { + // No writer. Send no signal, whatever the pidfile says — the pid may + // belong to somebody else entirely by now. Which of --until, a clean + // stop and a death ended the recording is the sentinel's question, + // answered by P7 check 1 over the log this unblocks. + fmt.Fprintf(cfg.Notes, "note: no writer holds %s; the recorder has already finished\n", cfg.Log) + return log, nil + } + + // The lock is held, so a writer is alive and the pidfile's pid cannot be + // stale — the recorder that took the lock is the one the parent recorded. + gone, err := signalRecorder(pid) + if err != nil { + log.Close() + return nil, err + } + if gone { + // A live writer holds the log and the pidfile names a process that + // does not exist. That is a broken contract, not a case to reason + // around: signalling the real holder would mean guessing who it is. + log.Close() + return nil, fmt.Errorf("a writer holds %s but pidfile %s names pid %d, which does not exist: the pidfile does not name the process that holds the log", + cfg.Log, cfg.PidFile, pid) + } + + // Wait on the LOCK, not on the pid: its release is the kernel-guaranteed + // writer-is-gone event, and it carries no pid-reuse hazard. + deadline := cfg.Clock.Now().Add(recorderStopTimeout) + for { + select { + case <-ctx.Done(): + log.Close() + return nil, ctx.Err() + case <-cfg.Clock.After(recorderStopPoll): + } + + held, err := tryLockExclusive(log) + if err != nil { + log.Close() + return nil, err + } + if held { + return log, nil + } + if !cfg.Clock.Now().Before(deadline) { + log.Close() + return nil, fmt.Errorf("recorder pid %d still holds %s %s after SIGTERM; refusing to read a log a writer can still append to (§4.4 step 4)", + pid, cfg.Log, recorderStopTimeout) + } + } +} + +// drainVerdict is what the drain wait concluded about one rule it could not +// clear. It carries the reason as well as the prose because the two outcomes +// are genuinely different faults: drain_timeout means the rule is still there +// and still behind, rule_absent means it is gone. Collapsing both into +// drain_timeout would name the wait instead of the fault, and Reason is a +// published vocabulary that reaches the action's JSON (§19.0). +type drainVerdict struct { + reason UnobservableReason + note string +} + +// drainWait is §19.1 step 7 and §14.6: the final instance of the liveness +// check, asking each rule the last question — did you evaluate through the end +// of the window? A rule that cannot answer within drainTimeout is +// unobservable, never a pass. +// +// It returns one verdict per rule it could not clear, keyed by UID, which the +// caller folds into the Result. It returns an error only for a hard failure of +// the wait itself (§19.3 case 1); a rule that simply never catches up is +// reported, not raised. +// +// Two rules are excluded from the wait before it starts, and both are +// exclusions of work that could not change a verdict: +// +// - a rule the HEADER says was already paused when the recording opened +// (§12): it is skipped, it was not evaluating, and it never was — there is +// no evaluation to wait for. The header and not the definition, for +// decide's reason (Header.pausedAtStart): a rule the header says was +// active must be drained or faulted, because a pause somebody applied +// after the window is not evidence about the window; +// - a rule whose last poll says Found == false: P7 check 8 already makes it +// unobservable, so the only thing draining it could add is drainTimeout of +// waiting before the same answer. +func drainWait(ctx context.Context, cfg Config, src Source, defs []Definition, pausedAtStart map[string]bool, + rt map[string]ruleTimings, polls []Poll, windowEnd time.Time, timeout time.Duration) (map[string]drainVerdict, error) { + + pending := make(map[string]string) // uid -> title, the shape observeAll wants + for _, d := range defs { + if pausedAtStart[d.UID] { + continue + } + rulePolls := pollsForRule(polls, d.UID) + if n := len(rulePolls); n > 0 && !rulePolls[n-1].Found { + continue + } + // Evidence already in hand can satisfy the wait outright: a rule whose + // recorded evaluations already reach past the end of the window has + // answered the question, and polling it again asks nothing new. + if !anyPollEvaluatedThrough(rulePolls, windowEnd) { + pending[d.UID] = d.Title + } + } + if len(pending) == 0 { + return nil, nil + } + + fmt.Fprintf(cfg.Notes, "drain wait: %d rule(s) have not yet evaluated through %s (limit %s)\n", + len(pending), windowEnd.Format(time.RFC3339), timeout) + + deadline := cfg.Clock.Now().Add(timeout) + verdicts := make(map[string]drainVerdict) + for { + uids := make([]string, 0, len(pending)) + for uid := range pending { + uids = append(uids, uid) + } + sort.Strings(uids) // deterministic request order and message order + + observed, err := observeAll(ctx, src, pending, uids, cfg.Concurrency) + if err != nil { + return nil, fmt.Errorf("drain wait: %w", err) + } + for _, uid := range uids { + obs, ok := observed[uid] + if !ok { + continue + } + rule := stateRuleByUID(obs.Rules, uid) + if rule == nil { + // §14.5: a 2xx that parsed and carries no matching rule is an + // authoritative "the rule is gone" — P2 retried every transport + // failure long before this Observation existed. It is knowable + // on the FIRST poll, so waiting the rest of drainTimeout would + // spend two minutes to reach the same verdict under a name that + // describes the wait rather than the fault. + verdicts[uid] = drainVerdict{ + reason: ReasonRuleAbsent, + note: fmt.Sprintf("rule %q: absent from the state endpoint during the drain wait; there is no evaluation to wait for (§14.5)", + pending[uid]), + } + delete(pending, uid) + continue + } + if rule.IsPaused { + // A paused rule does not evaluate, so this one can never catch + // up and the rest of drainTimeout would buy nothing. The reason + // stays drain_timeout: UnobservableReason is a published + // vocabulary that reaches the action's JSON (§19.0), and the + // prose below is where the detail belongs. + verdicts[uid] = drainVerdict{ + reason: ReasonDrainTimeout, + note: fmt.Sprintf("rule %q: paused before it evaluated through %s, so it never will (§14.8)", + pending[uid], windowEnd.Format(time.RFC3339)), + } + delete(pending, uid) + continue + } + if evaluatedThrough(rule.LastEvaluation, obs.Skew, obs.SkewBound, windowEnd) { + delete(pending, uid) + } + } + if len(pending) == 0 { + return verdicts, nil + } + + now := cfg.Clock.Now() + if !now.Before(deadline) { + for uid, title := range pending { + verdicts[uid] = drainVerdict{ + reason: ReasonDrainTimeout, + note: fmt.Sprintf("rule %q: did not evaluate through %s within the %s drain limit (§19.1 step 7)", + title, windowEnd.Format(time.RFC3339), timeout), + } + } + return verdicts, nil + } + + // Re-ask no faster than the tightest cadence among the rules still + // pending: a rule evaluating every 60s cannot answer differently 200ms + // later, and hammering it would spend the request budget §5 accounts + // for on nothing. + wait := deadline.Sub(now) + for uid := range pending { + if every := rt[uid].pollEvery; every > 0 { + wait = min(wait, every) + } + } + select { + case <-ctx.Done(): + return nil, ctx.Err() + case <-cfg.Clock.After(max(wait, 0)): + } + } +} + +// anyPollEvaluatedThrough reports whether any recorded poll of a rule already +// proves it evaluated through windowEnd. +func anyPollEvaluatedThrough(polls []Poll, windowEnd time.Time) bool { + for _, p := range polls { + if !p.Found { + continue + } + if evaluatedThrough(p.LastEvaluation, p.Skew(), p.SkewBound(), windowEnd) { + return true + } + } + return false +} + +// evaluatedThrough is the drain wait's one comparison, and it is cross-domain +// (§16): lastEvaluation is a Grafana timestamp and windowEnd is runner-domain, +// so the Grafana value is translated by its own poll's skew. The skew BOUND is +// then subtracted rather than added — the pessimistic end of the uncertainty — +// so an evaluation that only might have reached the end of the window does not +// count as one that did. Understating it costs a few more seconds of waiting; +// overstating it would pass an unproven window. +// +// A zero lastEvaluation never satisfies the wait: only a paused rule may +// legitimately report it (§2.3), and a paused rule has nothing to drain. +func evaluatedThrough(lastEval time.Time, skew, bound time.Duration, windowEnd time.Time) bool { + if lastEval.IsZero() { + return false + } + return !lastEval.Add(-skew).Add(-bound).Before(windowEnd) +} + +// mergeDrainTimeouts folds the I/O drain wait's verdicts into the pure layer's +// Result. P7 places this merge "before decide runs"; it cannot be, because +// decide owns proveCoverage and therefore builds the Coverage map itself — so +// the merge happens immediately after, which is the same thing from every +// caller's point of view and keeps decide's signature a pure function of its +// arguments. +// +// It returns its own error rather than mutating decide's, so neither hides the +// other: a run with one rule unobservable from the coverage proof and another +// from the drain wait must name both (H6 — inability beats violation, and it +// beats a second inability being dropped from the message too). The error says +// "at the drain wait" for that reason: the two are joined into one message, and +// two counts under one identical phrase read as a contradiction rather than as +// two findings. +func mergeDrainTimeouts(res Result, drained map[string]drainVerdict) (Result, error) { + if len(drained) == 0 { + return res, nil + } + if res.Coverage == nil { + res.Coverage = make(map[string]CoverageResult, len(drained)) + } + + var names []string + for i := range res.Verdicts { + uid := res.Verdicts[i].RuleUID + verdict, ok := drained[uid] + if !ok { + continue + } + cov := res.Coverage[uid] + cov.Unobservable = true + cov.Proved = false + if cov.Reason == "" { + // The FIRST reason wins, as it does inside proveCoverage: a rule + // the coverage proof already faulted keeps the fault it was + // actually caught by. + cov.Reason = verdict.reason + } + cov.Notes = append(cov.Notes, verdict.note) + res.Coverage[uid] = cov + + if res.Verdicts[i].Outcome != OutcomeUnobservable { + names = append(names, fmt.Sprintf("%s (%s)", res.Verdicts[i].Alert, verdict.reason)) + } + res.Verdicts[i].Outcome = OutcomeUnobservable + res.Verdicts[i].Note = strings.Join(cov.Notes, "; ") + } + if len(names) == 0 { + // Every drained rule was already unobservable for an earlier reason, + // so decide's own error already stops the run. Adding a second error + // saying the same thing would only make the message longer. + return res, nil + } + return res, fmt.Errorf("gate: %d rule(s) unobservable at the drain wait: %s", len(names), strings.Join(names, "; ")) +} diff --git a/grafana-alertcheck/internal/gate/check_test.go b/grafana-alertcheck/internal/gate/check_test.go new file mode 100644 index 000000000..2971c30e4 --- /dev/null +++ b/grafana-alertcheck/internal/gate/check_test.go @@ -0,0 +1,1216 @@ +package gate + +import ( + "bufio" + "context" + "errors" + "fmt" + "os" + "os/exec" + "path/filepath" + "strings" + "sync" + "syscall" + "testing" + "time" +) + +// The one rule every test in this file watches, unless it says otherwise: a +// 60s evaluation interval, no `for`, not paused. Every derived value follows +// from those three numbers, and the tests assert against them by name rather +// than by magic constant: +// +// pollEvery 30s (§5: intervalSeconds/2) +// maxGap 60s (2 x pollEvery) +// healthGrace 60s (max(maxGap, interval)) +// evalStaleAfter 120s (2 x interval) +// transitionGrace 60s (for + interval) +// drainTimeout 2m (max(2 x interval, 2m)) +const ( + checkUID = "rule-one" + checkTitle = "Rule One" + + checkPollEvery = 30 * time.Second + checkGrace = 60 * time.Second + checkDrainLimit = 2 * time.Minute +) + +func checkDef() Definition { + return Definition{ + UID: checkUID, Title: checkTitle, Folder: "F", Group: "G", + IntervalSeconds: 60, NoDataState: "OK", ExecErrState: "OK", + Kind: KindGrafanaManaged, + } +} + +// checkSource is a Source whose state answers depend on virtual time and on +// the call count, which is what a collection-loop test needs: fakeSource's +// static script cannot express "healthy for the whole window" without +// scripting every poll, and loopSource (watch_test.go) deliberately refuses +// Version and Definitions because the recorder's child never reads them. +type checkSource struct { + mu sync.Mutex + + version string + versionErr error + defs []Definition + defsErr error + + calls map[string]int + respond func(title string, call int) (Observation, error) +} + +func newCheckSource(respond func(title string, call int) (Observation, error)) *checkSource { + return &checkSource{ + version: "13.1.0", + defs: []Definition{checkDef()}, + calls: map[string]int{}, + respond: respond, + } +} + +func (s *checkSource) Version(context.Context) (string, error) { return s.version, s.versionErr } + +func (s *checkSource) Definitions(context.Context) ([]Definition, error) { return s.defs, s.defsErr } + +// RuleState answers from the responder. A nil responder means the test +// expects no state read at all — it fails with a message rather than a nil +// dereference, because "this path must not poll" is an assertion several tests +// here make on purpose. +func (s *checkSource) RuleState(_ context.Context, title string) (Observation, error) { + s.mu.Lock() + s.calls[title]++ + call := s.calls[title] + s.mu.Unlock() + if s.respond == nil { + return Observation{}, fmt.Errorf("checkSource: this test expects no state read, but %q was polled", title) + } + return s.respond(title, call) +} + +func (s *checkSource) callCount(title string) int { + s.mu.Lock() + defer s.mu.Unlock() + return s.calls[title] +} + +var _ Source = (*checkSource)(nil) + +// checkStateRule builds one state-endpoint rule whose totals agree with the +// instances it carries. That agreement is load-bearing: a totals map claiming +// normal instances that the instance list does not contain fails §3.2's +// verification (VerifyNormalInstancesVisible), which is a different failure +// from the one most of these tests are about. +func checkStateRule(lastEval time.Time, insts ...Instance) StateRule { + totals := map[string]int{} + for _, i := range insts { + totals[string(i.State)]++ + } + return StateRule{ + UID: checkUID, Title: checkTitle, Folder: "F", Group: "G", + Interval: time.Minute, State: "inactive", Health: "ok", + LastEvaluation: lastEval, Totals: totals, Instances: insts, + } +} + +// healthyObservation is a poll of a rule that evaluated at this instant, with +// no skew at all — every test that is not ABOUT skew uses zero so its +// arithmetic reads directly off the timestamps. +func healthyObservation(now time.Time, insts ...Instance) Observation { + return Observation{ + Rules: []StateRule{checkStateRule(now, insts...)}, + GrafanaNow: now, + Latency: 200 * time.Millisecond, + } +} + +// baseConfig is a single-step run over [now, now+5m]: window 5m, grace 60s, so +// the collection loop ends at now+6m. +func baseConfig(t *testing.T, clock Clock) Config { + t.Helper() + now := clock.Now() + return Config{ + URL: "https://grafana.example.com", + Alerts: []string{"uid:" + checkUID}, + From: now, + To: now.Add(5 * time.Minute), + Clock: clock, + Notes: &strings.Builder{}, + }.withDefaults() +} + +func notesOf(cfg Config) string { return cfg.Notes.(*strings.Builder).String() } + +// --------------------------------------------------------------------------- +// §19.1 step 1 — configuration validation +// --------------------------------------------------------------------------- + +func TestCheckValidateRejectsBadConfigurations(t *testing.T) { + clock := newFakeClock(testNow) + base := func() Config { + return Config{ + URL: "https://grafana.example.com", + From: testNow, + To: testNow.Add(5 * time.Minute), + Clock: clock, + } + } + + tests := []struct { + name string + mutate func(*Config) + wantErr string + }{ + { + name: "no url", + mutate: func(c *Config) { c.URL = ""; c.Alerts = []string{"A"} }, + wantErr: "no grafana url", + }, + { + name: "no to", + mutate: func(c *Config) { c.To = time.Time{}; c.Alerts = []string{"A"} }, + wantErr: "no `to`", + }, + { + // §19.1 step 1: an empty Alerts is an error — but only without a log. + name: "single-step without alerts", + mutate: func(c *Config) {}, + wantErr: "no alert names given", + }, + { + // An alerts file ending in a newline must not read as a named alert. + name: "single-step with only blank alert lines", + mutate: func(c *Config) { c.Alerts = []string{"", " "} }, + wantErr: "no alert names given", + }, + { + // §19.1 step 3, the other direction: the log names the alert set. + name: "log mode with alerts", + mutate: func(c *Config) { c.Log = "log.jsonl"; c.Alerts = []string{"A"} }, + wantErr: "--alerts is refused with a recorded log", + }, + { + // §7 — never a warning-and-continue. + name: "log mode without from", + mutate: func(c *Config) { c.Log = "log.jsonl"; c.From = time.Time{} }, + wantErr: "the deploy step must emit a completion timestamp", + }, + { + name: "from beyond the future tolerance", + mutate: func(c *Config) { + c.Alerts = []string{"A"} + c.From = testNow.Add(2 * time.Minute) + c.To = testNow.Add(10 * time.Minute) + }, + wantErr: "ahead of this runner's clock", + }, + { + name: "to before from", + mutate: func(c *Config) { c.Alerts = []string{"A"}; c.To = testNow.Add(-time.Minute) }, + wantErr: "is before `from`", + }, + { + // A past `to` is only "not a special mode" WITH a log: without one + // the coverage window ends before the first observation exists. + name: "single-step with a to already past", + mutate: func(c *Config) { + c.Alerts = []string{"A"} + c.From = testNow.Add(-10 * time.Minute) + c.To = testNow.Add(-time.Minute) + }, + wantErr: "can only be classified from a recording", + }, + } + + for _, tc := range tests { + t.Run(tc.name, func(t *testing.T) { + cfg := base() + tc.mutate(&cfg) + err := cfg.withDefaults().validate() + if err == nil { + t.Fatalf("validate() = nil, want an error containing %q", tc.wantErr) + } + if !strings.Contains(err.Error(), tc.wantErr) { + t.Fatalf("validate() = %q, want it to contain %q", err, tc.wantErr) + } + }) + } +} + +// A past `to` WITH a log is explicitly not a special mode (§7, §24.3): the +// collection loop's condition is already true and the evidence classifies +// immediately. No branch, and no refusal. +func TestCheckValidateAcceptsAPastToWithALog(t *testing.T) { + cfg := Config{ + URL: "https://grafana.example.com", + Log: "log.jsonl", + From: testNow.Add(-10 * time.Minute), + To: testNow.Add(-time.Minute), + Clock: newFakeClock(testNow), + }.withDefaults() + + if err := cfg.validate(); err != nil { + t.Fatalf("validate() = %v, want nil", err) + } + if cfg.PidFile != "log.jsonl.pid" { + t.Errorf("PidFile = %q, want the .pid default", cfg.PidFile) + } +} + +// --------------------------------------------------------------------------- +// Single-step mode +// --------------------------------------------------------------------------- + +func TestCheckSingleStepCleanWindowPasses(t *testing.T) { + clock := newVirtualClock(testNow) + cfg := baseConfig(t, clock) + src := newCheckSource(func(_ string, _ int) (Observation, error) { + return healthyObservation(clock.Now()), nil + }) + + res, err := check(context.Background(), cfg, src) + if err != nil { + t.Fatalf("check() = %v, want nil\nnotes:\n%s", err, notesOf(cfg)) + } + // H7: a pass is exactly this shape. + if len(res.Violations) != 0 { + t.Fatalf("Violations = %+v, want none", res.Violations) + } + if len(res.Verdicts) != 1 || res.Verdicts[0].Outcome != OutcomeClean { + t.Fatalf("Verdicts = %+v, want one clean verdict", res.Verdicts) + } + if cov := res.Coverage[checkUID]; !cov.Proved || cov.Unobservable { + t.Fatalf("Coverage = %+v, want proved", cov) + } + + // The collection loop ran to to+transitionGrace and no further (H5). + windowEnd := cfg.To.Add(checkGrace) + if clock.Now().Before(windowEnd) { + t.Errorf("stopped collecting at %s, before to+grace %s", clock.Now(), windowEnd) + } + // One measurement-pass poll plus one every 30s across the 6-minute + // collection, plus the drain wait's own polls. The exact count depends on + // the scheduler's random stagger, so assert the order of magnitude a full + // window implies rather than an exact number. + if got := src.callCount(checkTitle); got < 12 { + t.Errorf("polled %d times, want at least the ~13 a full 6-minute window at 30s implies", got) + } + if notes := notesOf(cfg); !strings.Contains(notes, "planned run time") { + t.Errorf("§13.2 requires the planned run time at start; notes were:\n%s", notes) + } +} + +// H5: a certain violation does not release the runner early, and it does not +// stop the gate reporting exit-1 shape — violations with a nil error. +func TestCheckSingleStepFiringInstanceReportsWithoutExitingEarly(t *testing.T) { + clock := newVirtualClock(testNow) + cfg := baseConfig(t, clock) + firing := Instance{ + Labels: map[string]string{"alertname": "Rule One", "instance": "a"}, + State: StateFiring, + ActiveAt: testNow.Add(-10 * time.Minute), // bad before the window opened + } + src := newCheckSource(func(_ string, _ int) (Observation, error) { + return healthyObservation(clock.Now(), firing), nil + }) + + res, err := check(context.Background(), cfg, src) + if err != nil { + t.Fatalf("check() = %v, want nil (a violation is exit 1, not an error)", err) + } + if len(res.Violations) != 1 { + t.Fatalf("Violations = %+v, want exactly one", res.Violations) + } + if got := res.Violations[0].Outcome; got != OutcomePersistentlyBad { + t.Errorf("Outcome = %q, want %q", got, OutcomePersistentlyBad) + } + if windowEnd := cfg.To.Add(checkGrace); clock.Now().Before(windowEnd) { + t.Errorf("exited early at %s; H5 requires collecting to %s", clock.Now(), windowEnd) + } +} + +// §4.2/§22.4: in single-step mode an explicit `from` earlier than the first +// observation is a DECLARED blind interval — a warning and a pass, naming the +// exact interval it cannot see. Recorder mode keeps P7 check 2 strict. +func TestCheckSingleStepFromBeforeFirstObservationWarnsAndPasses(t *testing.T) { + clock := newVirtualClock(testNow) + cfg := baseConfig(t, clock) + cfg.From = testNow.Add(-2 * time.Minute) + src := newCheckSource(func(_ string, _ int) (Observation, error) { + return healthyObservation(clock.Now()), nil + }) + + res, err := check(context.Background(), cfg, src) + if err != nil { + t.Fatalf("check() = %v, want a pass with a warning\nnotes:\n%s", err, notesOf(cfg)) + } + notes := notesOf(cfg) + if !strings.Contains(notes, "cannot see [") || !strings.Contains(notes, testNow.Format(time.RFC3339)) { + t.Errorf("want a warning naming the unseen interval; notes were:\n%s", notes) + } + // The classified window is the clamped one, and Result says so rather than + // reporting a window the run never proved. + if !res.From.Equal(testNow) { + t.Errorf("Result.From = %s, want the clamped %s", res.From, testNow) + } +} + +// §19.3 case 1: the failure limit was exceeded. The measurement pass succeeds +// and the collection loop then hits a terminal failure, so this exercises the +// path a live run really takes. +func TestCheckFailClosedOnExhaustedRetries(t *testing.T) { + clock := newVirtualClock(testNow) + cfg := baseConfig(t, clock) + src := newCheckSource(func(_ string, call int) (Observation, error) { + if call > 1 { + return Observation{}, &RetryExhaustedError{Failures: 6, Cause: errors.New("connection refused")} + } + return healthyObservation(clock.Now()), nil + }) + + res, err := check(context.Background(), cfg, src) + if err == nil { + t.Fatalf("check() = nil, want the collection failure to fail closed") + } + if !strings.Contains(err.Error(), "collect evidence") { + t.Errorf("err = %q, want it to name the collection step", err) + } + if len(res.Violations) != 0 { + t.Errorf("Violations = %+v; an error must never be reported as a verdict", res.Violations) + } +} + +// §19.3 case 2: the resolution of the definitions failed. Both shapes — the +// ruler read itself failing, and a name that resolves to nothing. +func TestCheckFailClosedOnDefinitionResolution(t *testing.T) { + t.Run("ruler read fails", func(t *testing.T) { + clock := newVirtualClock(testNow) + cfg := baseConfig(t, clock) + src := newCheckSource(nil) + src.defsErr = errors.New("502 bad gateway") + + if _, err := check(context.Background(), cfg, src); err == nil || + !strings.Contains(err.Error(), "read rule definitions") { + t.Fatalf("check() = %v, want a definitions-read failure", err) + } + }) + + t.Run("unknown alert name", func(t *testing.T) { + clock := newVirtualClock(testNow) + cfg := baseConfig(t, clock) + cfg.Alerts = []string{"No Such Rule"} + src := newCheckSource(nil) + + if _, err := check(context.Background(), cfg, src); err == nil || + !strings.Contains(err.Error(), "no rule matched") { + t.Fatalf("check() = %v, want a no-match failure", err) + } + }) +} + +// The version gate (§2.7 control 2): an unsupported Grafana is exit 2 before +// anything else is attempted. +func TestCheckRefusesUnsupportedGrafanaVersion(t *testing.T) { + clock := newVirtualClock(testNow) + cfg := baseConfig(t, clock) + src := newCheckSource(nil) + src.version = "12.4.0" + + if _, err := check(context.Background(), cfg, src); err == nil || + !strings.Contains(err.Error(), "unsupported grafana version") { + t.Fatalf("check() = %v, want the version gate to refuse 12.4.0", err) + } +} + +// §5.2: the budget is checked against the latencies the measurement pass +// actually measured, and a schedule that cannot fit errors at START rather +// than producing a gap-riddled recording nobody can classify. +func TestCheckSingleStepRefusesAScheduleThatDoesNotFit(t *testing.T) { + clock := newVirtualClock(testNow) + cfg := baseConfig(t, clock) + src := newCheckSource(func(_ string, _ int) (Observation, error) { + obs := healthyObservation(clock.Now()) + obs.Latency = 45 * time.Second // longer than the rule's own 30s cadence + return obs, nil + }) + + _, err := check(context.Background(), cfg, src) + if err == nil { + t.Fatalf("check() = nil, want the budget check to refuse the schedule") + } + for _, want := range []string{"raising concurrency", "raising poll-interval", "watching fewer alerts"} { + if !strings.Contains(err.Error(), want) { + t.Errorf("err = %q, want it to name the control %q (§5.1)", err, want) + } + } +} + +// --------------------------------------------------------------------------- +// Recorder mode +// --------------------------------------------------------------------------- + +// recordedLog writes a log the way watch would have: a header, one poll every +// 30s over [start, end], and a stopped sentinel at sentinelAt. lastEvalLag is +// how far behind each poll's own GrafanaNow its lastEvaluation sits, which is +// what the drain-wait tests vary. +func recordedLog(t *testing.T, dir string, url string, startedAt, start, end, sentinelAt time.Time, lastEvalLag time.Duration) string { + t.Helper() + path := filepath.Join(dir, "log.jsonl") + clock := newFakeClock(sentinelAt) + w, err := NewWriter(path, clock) + if err != nil { + t.Fatalf("NewWriter: %v", err) + } + header := Header{ + URL: url, + GrafanaVersion: "13.1.0", + StartedAt: startedAt, + Rules: []LoggedRule{{ + UID: checkUID, Title: checkTitle, Folder: "F", Group: "G", + IntervalSeconds: 60, NoDataState: "OK", ExecErrState: "OK", + PollEverySeconds: checkPollEvery.Seconds(), + }}, + } + if err := w.WriteHeader(header); err != nil { + t.Fatalf("WriteHeader: %v", err) + } + for at := start; !at.After(end); at = at.Add(checkPollEvery) { + if err := w.WritePoll(Poll{ + RuleUID: checkUID, GrafanaNow: at, Found: true, + State: "inactive", Health: "ok", LastEvaluation: at.Add(-lastEvalLag), + }); err != nil { + t.Fatalf("WritePoll: %v", err) + } + } + if err := w.Stop(); err != nil { + t.Fatalf("Stop: %v", err) + } + return path +} + +// deadPid returns a pid that is guaranteed to have exited — the normal state +// of a recorder by the time check signals it, since a recorder given --until +// (or one that finished cleanly) is already gone. +func deadPid(t *testing.T) int { + t.Helper() + cmd := exec.Command("/bin/sh", "-c", "exit 0") + if err := cmd.Start(); err != nil { + t.Fatalf("start a throwaway process: %v", err) + } + pid := cmd.Process.Pid + if err := cmd.Wait(); err != nil { + t.Fatalf("wait for the throwaway process: %v", err) + } + return pid +} + +func writePid(t *testing.T, path, contents string) { + t.Helper() + if err := os.WriteFile(path, []byte(contents), 0o644); err != nil { + t.Fatalf("write pidfile: %v", err) + } +} + +// recorderConfig points check at a recording of [testNow-1m, windowEnd+30s] +// over the window [testNow, testNow+5m]. +func recorderConfig(t *testing.T, clock Clock, logPath string) Config { + t.Helper() + return Config{ + URL: "https://grafana.example.com", + Log: logPath, + From: testNow, + To: testNow.Add(5 * time.Minute), + Clock: clock, + Notes: &strings.Builder{}, + }.withDefaults() +} + +func TestCheckRecorderModeCleanWindowPasses(t *testing.T) { + dir := t.TempDir() + windowEnd := testNow.Add(5*time.Minute + checkGrace) + logPath := recordedLog(t, dir, "https://grafana.example.com", + testNow.Add(-time.Minute), testNow.Add(-time.Minute), windowEnd.Add(30*time.Second), windowEnd.Add(30*time.Second), 0) + writePid(t, logPath+".pid", fmt.Sprintf("%d\n", deadPid(t))) + + clock := newVirtualClock(testNow.Add(time.Minute)) + cfg := recorderConfig(t, clock, logPath) + // The drain wait is satisfied from the log's own evidence, so the source + // must never be asked for a state — asserted by the nil responder. + src := newCheckSource(func(title string, _ int) (Observation, error) { + t.Errorf("the drain wait polled %q although the log already proves the evaluations", title) + return Observation{}, errors.New("unexpected poll") + }) + + res, err := check(context.Background(), cfg, src) + if err != nil { + t.Fatalf("check() = %v, want nil\nnotes:\n%s", err, notesOf(cfg)) + } + if len(res.Violations) != 0 { + t.Fatalf("Violations = %+v, want none", res.Violations) + } + if len(res.Verdicts) != 1 || res.Verdicts[0].Outcome != OutcomeClean { + t.Fatalf("Verdicts = %+v, want one clean verdict", res.Verdicts) + } + if res.GrafanaVersion != "13.1.0" { + t.Errorf("GrafanaVersion = %q, want the recorded one", res.GrafanaVersion) + } + // The collection loop still waited out to+transitionGrace (H5) even though + // the recorder had already finished. + if clock.Now().Before(windowEnd) { + t.Errorf("returned at %s, before to+grace %s", clock.Now(), windowEnd) + } +} + +// §19.3 case 3: the identity of the log is not correct. The check runs against +// the header read EARLY, so it fails before the window's wait rather than +// after it. +func TestCheckFailClosedOnWrongLogIdentity(t *testing.T) { + t.Run("different url", func(t *testing.T) { + dir := t.TempDir() + windowEnd := testNow.Add(5*time.Minute + checkGrace) + logPath := recordedLog(t, dir, "https://other.example.com", + testNow.Add(-time.Minute), testNow.Add(-time.Minute), windowEnd, windowEnd.Add(30*time.Second), 0) + + clock := newVirtualClock(testNow) + cfg := recorderConfig(t, clock, logPath) + _, err := check(context.Background(), cfg, newCheckSource(nil)) + if err == nil || !strings.Contains(err.Error(), "log identity") { + t.Fatalf("check() = %v, want a log-identity failure", err) + } + if !clock.Now().Equal(testNow) { + t.Errorf("the identity check waited out the window (now %s); it must fail before the wait", clock.Now()) + } + }) + + t.Run("rule no longer resolves", func(t *testing.T) { + dir := t.TempDir() + windowEnd := testNow.Add(5*time.Minute + checkGrace) + logPath := recordedLog(t, dir, "https://grafana.example.com", + testNow.Add(-time.Minute), testNow.Add(-time.Minute), windowEnd, windowEnd.Add(30*time.Second), 0) + + cfg := recorderConfig(t, newVirtualClock(testNow), logPath) + src := newCheckSource(nil) + src.defs = []Definition{{UID: "somebody-else", Title: "Other", Kind: KindGrafanaManaged, IntervalSeconds: 60}} + + _, err := check(context.Background(), cfg, src) + if err == nil || !strings.Contains(err.Error(), "log identity") { + t.Fatalf("check() = %v, want a log-identity failure", err) + } + }) +} + +// §19.3 case 4: the coverage proof failed. A hole in the middle of the +// recording is not saved by healthy data at both ends (§22.4). +func TestCheckFailClosedOnCoverageGap(t *testing.T) { + dir := t.TempDir() + windowEnd := testNow.Add(5*time.Minute + checkGrace) + path := filepath.Join(dir, "log.jsonl") + clock := newFakeClock(windowEnd.Add(30 * time.Second)) + w, err := NewWriter(path, clock) + if err != nil { + t.Fatalf("NewWriter: %v", err) + } + if err := w.WriteHeader(Header{ + URL: "https://grafana.example.com", GrafanaVersion: "13.1.0", StartedAt: testNow.Add(-time.Minute), + Rules: []LoggedRule{{UID: checkUID, Title: checkTitle, IntervalSeconds: 60, PollEverySeconds: checkPollEvery.Seconds()}}, + }); err != nil { + t.Fatalf("WriteHeader: %v", err) + } + for at := testNow.Add(-time.Minute); !at.After(windowEnd.Add(30 * time.Second)); at = at.Add(checkPollEvery) { + // A three-minute hole in the middle of the window. + if at.After(testNow.Add(time.Minute)) && at.Before(testNow.Add(4*time.Minute)) { + continue + } + if err := w.WritePoll(Poll{ + RuleUID: checkUID, GrafanaNow: at, Found: true, State: "inactive", Health: "ok", LastEvaluation: at, + }); err != nil { + t.Fatalf("WritePoll: %v", err) + } + } + if err := w.Stop(); err != nil { + t.Fatalf("Stop: %v", err) + } + writePid(t, path+".pid", fmt.Sprintf("%d\n", deadPid(t))) + + cfg := recorderConfig(t, newVirtualClock(testNow), path) + res, err := check(context.Background(), cfg, newCheckSource(nil)) + if err == nil { + t.Fatalf("check() = nil, want the coverage gap to fail closed") + } + if got := res.Coverage[checkUID].Reason; got != ReasonHeartbeatGap { + t.Errorf("Reason = %q, want %q", got, ReasonHeartbeatGap) + } + if got := res.Verdicts[0].Outcome; got != OutcomeUnobservable { + t.Errorf("Outcome = %q, want %q", got, OutcomeUnobservable) + } +} + +// §19.3 case 5: the drain limit passed. The recording itself is clean, so this +// isolates the drain wait — the rule simply never evaluates through the end of +// the window, and a rule that cannot answer that question is unobservable. +func TestCheckFailClosedOnDrainTimeout(t *testing.T) { + dir := t.TempDir() + windowEnd := testNow.Add(5*time.Minute + checkGrace) + // A 45s lag keeps every poll inside evalStaleAfter (120s), so P7 check 6 + // is silent and only the drain wait can fail. + logPath := recordedLog(t, dir, "https://grafana.example.com", + testNow.Add(-time.Minute), testNow.Add(-time.Minute), windowEnd, windowEnd.Add(30*time.Second), 45*time.Second) + writePid(t, logPath+".pid", fmt.Sprintf("%d\n", deadPid(t))) + + clock := newVirtualClock(testNow) + cfg := recorderConfig(t, clock, logPath) + frozen := windowEnd.Add(-45 * time.Second) + src := newCheckSource(func(_ string, _ int) (Observation, error) { + now := clock.Now() + return Observation{ + Rules: []StateRule{checkStateRule(frozen)}, + GrafanaNow: now, + }, nil + }) + + res, err := check(context.Background(), cfg, src) + if err == nil { + t.Fatalf("check() = nil, want the drain limit to fail closed") + } + if got := res.Coverage[checkUID].Reason; got != ReasonDrainTimeout { + t.Errorf("Reason = %q, want %q", got, ReasonDrainTimeout) + } + if got := res.Verdicts[0].Outcome; got != OutcomeUnobservable { + t.Errorf("Outcome = %q, want %q", got, OutcomeUnobservable) + } + if !strings.Contains(res.Verdicts[0].Note, "drain limit") { + t.Errorf("Note = %q, want it to explain the drain limit", res.Verdicts[0].Note) + } + if waited := clock.Now().Sub(windowEnd); waited < checkDrainLimit { + t.Errorf("gave up after %s of drain wait, want the full %s", waited, checkDrainLimit) + } +} + +// §14.5: a rule the state endpoint no longer serves is knowable on the FIRST +// drain poll, and the answer is rule_absent — the fault — rather than +// drain_timeout, which would only name the wait. It must not spend the whole +// drain limit to reach it. +func TestCheckDrainWaitNamesADeletedRuleAtOnce(t *testing.T) { + dir := t.TempDir() + windowEnd := testNow.Add(5*time.Minute + checkGrace) + logPath := recordedLog(t, dir, "https://grafana.example.com", + testNow.Add(-time.Minute), testNow.Add(-time.Minute), windowEnd, windowEnd.Add(30*time.Second), 45*time.Second) + writePid(t, logPath+".pid", fmt.Sprintf("%d\n", deadPid(t))) + + clock := newVirtualClock(testNow) + cfg := recorderConfig(t, clock, logPath) + // An authoritative 2xx that parsed and carries no matching rule. P2 + // retried every transport failure long before an Observation exists, so + // this is a deletion, not a hiccup. + src := newCheckSource(func(_ string, _ int) (Observation, error) { + return Observation{GrafanaNow: clock.Now()}, nil + }) + + res, err := check(context.Background(), cfg, src) + if err == nil { + t.Fatalf("check() = nil, want a deleted rule to fail closed") + } + if got := res.Coverage[checkUID].Reason; got != ReasonRuleAbsent { + t.Errorf("Reason = %q, want %q — the fault, not the wait", got, ReasonRuleAbsent) + } + if got := src.callCount(checkTitle); got != 1 { + t.Errorf("polled %d times, want exactly 1: the absence is knowable on the first poll", got) + } + if waited := clock.Now().Sub(windowEnd); waited >= checkDrainLimit { + t.Errorf("spent %s in the drain wait, want it to conclude at once", waited) + } +} + +// --------------------------------------------------------------------------- +// `skipped` comes from the header, not from a definition read after the window +// --------------------------------------------------------------------------- + +// pausedAfterWindowLog records a rule that was ACTIVE at record start and that +// fired inside the window. The caller then tells check that the rule's current +// definition says paused — the state somebody set after the fact. +func pausedAfterWindowLog(t *testing.T, dir string, firesAt time.Time, end, sentinelAt time.Time) string { + t.Helper() + path := filepath.Join(dir, "log.jsonl") + w, err := NewWriter(path, newFakeClock(sentinelAt)) + if err != nil { + t.Fatalf("NewWriter: %v", err) + } + if err := w.WriteHeader(Header{ + URL: "https://grafana.example.com", GrafanaVersion: "13.1.0", StartedAt: testNow.Add(-time.Minute), + Rules: []LoggedRule{{ + UID: checkUID, Title: checkTitle, IntervalSeconds: 60, + IsPaused: false, PollEverySeconds: checkPollEvery.Seconds(), + }}, + }); err != nil { + t.Fatalf("WriteHeader: %v", err) + } + firing := Instance{ + Labels: map[string]string{"alertname": checkTitle, "instance": "a"}, + State: StateFiring, + ActiveAt: firesAt, + } + for at := testNow.Add(-time.Minute); !at.After(end); at = at.Add(checkPollEvery) { + p := Poll{RuleUID: checkUID, GrafanaNow: at, Found: true, State: "inactive", Health: "ok", LastEvaluation: at} + if !at.Before(firesAt) { + p.State = "firing" + p.Abnormal = []Instance{firing} + } + if err := w.WritePoll(p); err != nil { + t.Fatalf("WritePoll: %v", err) + } + } + if err := w.Stop(); err != nil { + t.Fatalf("Stop: %v", err) + } + return path +} + +// pausedAfterWindowCheck runs the timeline above. The recording reaches past +// to + transitionGrace, which for this 60s rule is to + 60s: the fresh +// definition says paused, but that no longer shrinks the grace — the header +// does, and the header says the rule was active (deriveGlobalTimings). +func pausedAfterWindowCheck(t *testing.T, allowPaused bool) (Result, error, Config) { + t.Helper() + dir := t.TempDir() + to := testNow.Add(5 * time.Minute) + end := to.Add(checkGrace + 30*time.Second) + logPath := pausedAfterWindowLog(t, dir, testNow.Add(2*time.Minute), end, end) + writePid(t, logPath+".pid", fmt.Sprintf("%d\n", deadPid(t))) + + cfg := recorderConfig(t, newVirtualClock(testNow), logPath) + cfg.AllowPaused = allowPaused + + src := newCheckSource(nil) + paused := checkDef() + paused.IsPaused = true // somebody paused it after the alert started paging + src.defs = []Definition{paused} + + res, err := check(context.Background(), cfg, src) + return res, err, cfg +} + +// The window's own evidence outranks a definition read after it closed: a rule +// that was active at record start is classified, whatever its pause state is +// by the time check resolves the definitions. +func TestCheckPausingARuleAfterTheWindowDoesNotMakeItSkipped(t *testing.T) { + res, err, cfg := pausedAfterWindowCheck(t, false) + if err != nil { + t.Fatalf("check() = %v, want a classified verdict\nnotes:\n%s", err, notesOf(cfg)) + } + if got := res.Verdicts[0].Outcome; got != OutcomeNewlyBad { + t.Fatalf("Outcome = %q, want %q: the rule was active for the whole window and fired inside it", got, OutcomeNewlyBad) + } + if len(res.Violations) != 1 || res.Violations[0].Outcome != OutcomeNewlyBad { + t.Fatalf("Violations = %+v, want the firing reported", res.Violations) + } + if strings.Contains(res.Verdicts[0].Note, "paused before the window opened") { + t.Errorf("Note = %q, which the log's own polls contradict", res.Verdicts[0].Note) + } +} + +// The regression pin for the loophole this fix closed. Reading skipped from +// the post-window definition made the rule skipped; --allow-paused then made +// skipped free; and a window in which the alert fired reported exit 0. The +// default message names --allow-paused, so an operator was led straight to it. +func TestCheckAllowPausedCannotExcuseARulePausedAfterItFired(t *testing.T) { + res, err, cfg := pausedAfterWindowCheck(t, true) + if err != nil { + t.Fatalf("check() = %v, want a classified verdict\nnotes:\n%s", err, notesOf(cfg)) + } + if len(res.Violations) == 0 { + t.Fatalf("Violations = none with --allow-paused: the run passed over a window in which the alert fired") + } +} + +// The other direction, unchanged: a rule the HEADER says was paused when the +// recording opened is genuinely skipped. It has no polls, so no coverage is +// attempted for it, and --allow-paused behaves as it always did. +func TestCheckHeaderPausedRuleStaysSkipped(t *testing.T) { + dir := t.TempDir() + windowEnd := testNow.Add(5*time.Minute + checkGrace) + path := filepath.Join(dir, "log.jsonl") + w, err := NewWriter(path, newFakeClock(windowEnd.Add(30*time.Second))) + if err != nil { + t.Fatalf("NewWriter: %v", err) + } + // Named in the header, is_paused true, and no poll records at all — the + // shape watch writes for a rule paused before the window opened (P6 + // deviation 4). + if err := w.WriteHeader(Header{ + URL: "https://grafana.example.com", GrafanaVersion: "13.1.0", StartedAt: testNow.Add(-time.Minute), + Rules: []LoggedRule{{ + UID: checkUID, Title: checkTitle, IntervalSeconds: 60, + IsPaused: true, PollEverySeconds: checkPollEvery.Seconds(), + }}, + }); err != nil { + t.Fatalf("WriteHeader: %v", err) + } + if err := w.Stop(); err != nil { + t.Fatalf("Stop: %v", err) + } + writePid(t, path+".pid", fmt.Sprintf("%d\n", deadPid(t))) + + run := func(allowPaused bool) (Result, error) { + cfg := recorderConfig(t, newVirtualClock(testNow), path) + cfg.AllowPaused = allowPaused + // The definition is unpaused now; the header still decides. + src := newCheckSource(func(title string, _ int) (Observation, error) { + t.Errorf("the drain wait polled skipped rule %q", title) + return Observation{}, errors.New("unexpected poll") + }) + return check(context.Background(), cfg, src) + } + + res, err := run(false) + if err != nil { + t.Fatalf("check() = %v, want exit-1 shape: a skipped rule is a known condition, not an inability", err) + } + if got := res.Verdicts[0].Outcome; got != OutcomeSkipped { + t.Fatalf("Outcome = %q, want %q", got, OutcomeSkipped) + } + if _, ok := res.Coverage[checkUID]; ok { + t.Errorf("Coverage[%s] present, want absent: a skipped rule has no coverage to prove", checkUID) + } + if len(res.Violations) != 1 { + t.Errorf("Violations = %+v, want the MinObserved shortfall (§12.1)", res.Violations) + } + + res, err = run(true) + if err != nil || len(res.Violations) != 0 { + t.Errorf("with --allow-paused: err = %v, Violations = %+v, want a pass", err, res.Violations) + } +} + +// A paused rule does not evaluate, so it can never catch up: the drain wait +// must conclude on the first poll instead of spending the whole limit. +func TestCheckDrainWaitConcludesAtOnceOnAPausedRule(t *testing.T) { + dir := t.TempDir() + windowEnd := testNow.Add(5*time.Minute + checkGrace) + logPath := recordedLog(t, dir, "https://grafana.example.com", + testNow.Add(-time.Minute), testNow.Add(-time.Minute), windowEnd, windowEnd.Add(30*time.Second), 45*time.Second) + writePid(t, logPath+".pid", fmt.Sprintf("%d\n", deadPid(t))) + + clock := newVirtualClock(testNow) + cfg := recorderConfig(t, clock, logPath) + src := newCheckSource(func(_ string, _ int) (Observation, error) { + now := clock.Now() + rule := checkStateRule(windowEnd.Add(-45 * time.Second)) + rule.IsPaused = true + return Observation{Rules: []StateRule{rule}, GrafanaNow: now}, nil + }) + + res, err := check(context.Background(), cfg, src) + if err == nil { + t.Fatalf("check() = nil, want a rule that stopped evaluating to fail closed") + } + if got := src.callCount(checkTitle); got != 1 { + t.Errorf("polled %d times, want exactly 1: a paused rule can never catch up", got) + } + if got := res.Coverage[checkUID].Reason; got != ReasonDrainTimeout { + t.Errorf("Reason = %q, want %q — the vocabulary is published (§19.0), so the detail goes in the note", got, ReasonDrainTimeout) + } + if !strings.Contains(res.Verdicts[0].Note, "paused before it evaluated through") { + t.Errorf("Note = %q, want it to say the rule was paused", res.Verdicts[0].Note) + } + if waited := clock.Now().Sub(windowEnd); waited >= checkDrainLimit { + t.Errorf("spent %s in the drain wait, want it to conclude at once", waited) + } +} + +// P6's obligation on this phase: an absent or unparseable pidfile is never +// "there was nothing to stop". The parent writes the pidfile only once the +// child reports that it is recording, so a missing one means the recording +// never started — and the log must not be read at all. +func TestCheckRefusesToReadALogItCannotStop(t *testing.T) { + windowEnd := testNow.Add(5*time.Minute + checkGrace) + + tests := []struct { + name string + pidfile string // "" = do not create one + }{ + {name: "missing pidfile"}, + {name: "unparseable pidfile", pidfile: "not-a-pid\n"}, + {name: "empty pidfile", pidfile: ""}, + } + + for i, tc := range tests { + t.Run(tc.name, func(t *testing.T) { + dir := t.TempDir() + logPath := recordedLog(t, dir, "https://grafana.example.com", + testNow.Add(-time.Minute), testNow.Add(-time.Minute), windowEnd, windowEnd.Add(30*time.Second), 0) + if i != 0 { + writePid(t, logPath+".pid", tc.pidfile) + } + + cfg := recorderConfig(t, newVirtualClock(testNow), logPath) + _, err := check(context.Background(), cfg, newCheckSource(nil)) + if err == nil || !strings.Contains(err.Error(), "cannot stop the recorder") { + t.Fatalf("check() = %v, want a refusal to stop the recorder", err) + } + }) + } +} + +// startLockHolder re-execs this test binary as a process that holds the log's +// flock and ignores SIGTERM, and returns its pid once the lock is genuinely +// held. See lockHolderEnv (watch_daemon_test.go) for why it must be a separate +// real process rather than a shell one-liner. +func startLockHolder(t *testing.T, logPath string) int { + t.Helper() + cmd := exec.Command(os.Args[0]) + cmd.Env = append(os.Environ(), lockHolderEnv+"="+logPath) + cmd.Stderr = os.Stderr + stdout, err := cmd.StdoutPipe() + if err != nil { + t.Fatalf("pipe: %v", err) + } + if err := cmd.Start(); err != nil { + t.Fatalf("start the lock holder: %v", err) + } + t.Cleanup(func() { + _ = cmd.Process.Kill() + _ = cmd.Wait() + }) + if _, err := bufio.NewReader(stdout).ReadString('\n'); err != nil { + t.Fatalf("the lock holder never reported holding the lock: %v", err) + } + return cmd.Process.Pid +} + +// §4.4 step 4: a recorder that will not let go of the log means the log may +// still be appended to, and a log a writer can change cannot be read at all. +func TestCheckFailsWhenTheRecorderWillNotExit(t *testing.T) { + dir := t.TempDir() + windowEnd := testNow.Add(5*time.Minute + checkGrace) + logPath := recordedLog(t, dir, "https://grafana.example.com", + testNow.Add(-time.Minute), testNow.Add(-time.Minute), windowEnd, windowEnd.Add(30*time.Second), 0) + + writePid(t, logPath+".pid", fmt.Sprintf("%d\n", startLockHolder(t, logPath))) + + cfg := recorderConfig(t, newVirtualClock(testNow), logPath) + _, err := check(context.Background(), cfg, newCheckSource(nil)) + if err == nil || !strings.Contains(err.Error(), "still holds") { + t.Fatalf("check() = %v, want the stop wait to time out on the lock", err) + } +} + +// The regression pin for a stray SIGTERM. Nothing removes the pidfile when a +// recorder exits cleanly — the parent has returned and the child never learns +// the path — so after a --until run, a supported flow, the pidfile names a pid +// the operating system is free to hand to somebody else. The flock, not the +// pid, is what says whether a writer exists. +func TestCheckDoesNotSignalABystanderHoldingAReusedPid(t *testing.T) { + dir := t.TempDir() + windowEnd := testNow.Add(5*time.Minute + checkGrace) + logPath := recordedLog(t, dir, "https://grafana.example.com", + testNow.Add(-time.Minute), testNow.Add(-time.Minute), windowEnd.Add(30*time.Second), windowEnd.Add(30*time.Second), 0) + + // An innocent process that happens to hold the pid the finished recorder + // left behind. It does not hold the log's lock, because it is not a + // recorder. + bystander := exec.Command("sleep", "30") + if err := bystander.Start(); err != nil { + t.Fatalf("start the bystander: %v", err) + } + t.Cleanup(func() { + _ = bystander.Process.Kill() + _ = bystander.Wait() + }) + writePid(t, logPath+".pid", fmt.Sprintf("%d\n", bystander.Process.Pid)) + + cfg := recorderConfig(t, newVirtualClock(testNow), logPath) + if _, err := check(context.Background(), cfg, newCheckSource(nil)); err != nil { + t.Fatalf("check() = %v, want nil\nnotes:\n%s", err, notesOf(cfg)) + } + if err := syscall.Kill(bystander.Process.Pid, 0); err != nil { + t.Fatalf("the bystander is gone (%v): check signalled a process that was not the recorder", err) + } +} + +// P5's "two authorities", from check's side: maxGap comes from the cadence the +// header records, never from a re-derivation off intervalSeconds. The +// fail-open direction is the one asserted — a log recorded at 5s on a 60s rule +// must still fail on a hole a re-derived 30s maxGap would have forgiven. +func TestCheckDerivesMaxGapFromTheRecordedCadence(t *testing.T) { + dir := t.TempDir() + windowEnd := testNow.Add(5*time.Minute + checkGrace) + path := filepath.Join(dir, "log.jsonl") + w, err := NewWriter(path, newFakeClock(windowEnd.Add(30*time.Second))) + if err != nil { + t.Fatalf("NewWriter: %v", err) + } + if err := w.WriteHeader(Header{ + URL: "https://grafana.example.com", GrafanaVersion: "13.1.0", StartedAt: testNow.Add(-time.Minute), + Rules: []LoggedRule{{UID: checkUID, Title: checkTitle, IntervalSeconds: 60, PollEverySeconds: 5}}, + }); err != nil { + t.Fatalf("WriteHeader: %v", err) + } + for at := testNow.Add(-time.Minute); !at.After(windowEnd.Add(30 * time.Second)); at = at.Add(5 * time.Second) { + // A 20s hole: under the recorded 5s cadence maxGap is 10s and this + // fails; under a cadence re-derived from intervalSeconds it would be + // 60s and the hole would pass unseen. + if at.After(testNow.Add(time.Minute)) && at.Before(testNow.Add(80*time.Second)) { + continue + } + if err := w.WritePoll(Poll{ + RuleUID: checkUID, GrafanaNow: at, Found: true, State: "inactive", Health: "ok", LastEvaluation: at, + }); err != nil { + t.Fatalf("WritePoll: %v", err) + } + } + if err := w.Stop(); err != nil { + t.Fatalf("Stop: %v", err) + } + writePid(t, path+".pid", fmt.Sprintf("%d\n", deadPid(t))) + + cfg := recorderConfig(t, newVirtualClock(testNow), path) + res, err := check(context.Background(), cfg, newCheckSource(nil)) + if err == nil { + t.Fatalf("check() = nil; a 20s hole exceeds the 10s maxGap the recorded 5s cadence implies") + } + if got := res.Coverage[checkUID].Reason; got != ReasonHeartbeatGap { + t.Errorf("Reason = %q, want %q", got, ReasonHeartbeatGap) + } +} + +// --------------------------------------------------------------------------- +// The pieces, in isolation +// --------------------------------------------------------------------------- + +// The drain wait's one comparison is cross-domain (§16), and its uncertainty +// is spent in the fail-closed direction: an evaluation that only MIGHT have +// reached the end of the window does not count as one that did. +func TestEvaluatedThroughSpendsItsUncertaintyFailingClosed(t *testing.T) { + end := testNow + + tests := []struct { + name string + lastEval time.Time + skew, bound time.Duration + wantSatisfied bool + }{ + {name: "zero lastEvaluation never satisfies", lastEval: time.Time{}, wantSatisfied: false}, + {name: "exactly at the end, no skew", lastEval: end, wantSatisfied: true}, + {name: "one second short", lastEval: end.Add(-time.Second), wantSatisfied: false}, + { + name: "far enough past the end to absorb the bound", + // Grafana runs 10s fast; the reading translates back to end+5s and + // the 1s bound still leaves it past the end. + lastEval: end.Add(16 * time.Second), skew: 10 * time.Second, bound: time.Second, + wantSatisfied: true, + }, + { + name: "inside the bound is not proof", + // Translated it lands exactly on the end, so the bound can put it + // either side — which is not an evaluation THROUGH the end. + lastEval: end.Add(10 * time.Second), skew: 10 * time.Second, bound: time.Second, + wantSatisfied: false, + }, + } + + for _, tc := range tests { + t.Run(tc.name, func(t *testing.T) { + if got := evaluatedThrough(tc.lastEval, tc.skew, tc.bound, end); got != tc.wantSatisfied { + t.Errorf("evaluatedThrough() = %v, want %v", got, tc.wantSatisfied) + } + }) + } +} + +// H6 through the merge: a drain timeout on one rule and a coverage failure on +// another must both reach the message. Neither error may shadow the other. +func TestMergeDrainTimeoutsNamesEveryUnobservableRule(t *testing.T) { + res := Result{ + Coverage: map[string]CoverageResult{ + "a": {Proved: true}, + "b": {Unobservable: true, Reason: ReasonHeartbeatGap, Notes: []string{"rule \"B\": gap"}}, + }, + Verdicts: []RuleVerdict{ + {Alert: "A", RuleUID: "a", Outcome: OutcomeClean}, + {Alert: "B", RuleUID: "b", Outcome: OutcomeUnobservable}, + }, + } + + merged, err := mergeDrainTimeouts(res, map[string]drainVerdict{ + "a": {reason: ReasonDrainTimeout, note: "rule \"A\": did not evaluate through the end within the drain limit"}, + "b": {reason: ReasonDrainTimeout, note: "rule \"B\": did not evaluate through the end within the drain limit"}, + }) + if err == nil { + t.Fatal("mergeDrainTimeouts() = nil, want an error naming the newly unobservable rule") + } + // Its own shape: joined with decide's, two counts under one identical + // phrase would read as a contradiction rather than as two findings. + if !strings.Contains(err.Error(), "unobservable at the drain wait") { + t.Errorf("err = %q, want the drain wait's own error shape", err) + } + // Only A is newly unobservable; B was already, so naming it twice would + // only lengthen the message. + if !strings.Contains(err.Error(), "A ("+string(ReasonDrainTimeout)+")") { + t.Errorf("err = %q, want it to name A's drain timeout", err) + } + if strings.Contains(err.Error(), "B (") { + t.Errorf("err = %q, want it not to re-report B, which decide already reported", err) + } + if got := merged.Coverage["a"].Reason; got != ReasonDrainTimeout { + t.Errorf("Coverage[a].Reason = %q, want %q", got, ReasonDrainTimeout) + } + // B keeps the reason the coverage proof gave it — the FIRST reason wins, + // as it does inside proveCoverage. + if got := merged.Coverage["b"].Reason; got != ReasonHeartbeatGap { + t.Errorf("Coverage[b].Reason = %q, want the earlier %q", got, ReasonHeartbeatGap) + } + if merged.Verdicts[0].Outcome != OutcomeUnobservable { + t.Errorf("Verdicts[0].Outcome = %q, want %q", merged.Verdicts[0].Outcome, OutcomeUnobservable) + } +} + +// ReadLogHeader is the one read of a log a writer may still hold, so its +// refusals matter as much as its successes. +func TestReadLogHeader(t *testing.T) { + dir := t.TempDir() + + t.Run("reads line 1 while the log keeps growing", func(t *testing.T) { + path := filepath.Join(dir, "growing.jsonl") + w, err := NewWriter(path, newFakeClock(testNow)) + if err != nil { + t.Fatalf("NewWriter: %v", err) + } + defer w.Close() + if err := w.WriteHeader(testHeader()); err != nil { + t.Fatalf("WriteHeader: %v", err) + } + if err := w.WritePoll(Poll{RuleUID: "rule1", GrafanaNow: testNow, Found: true}); err != nil { + t.Fatalf("WritePoll: %v", err) + } + + h, err := ReadLogHeader(path) + if err != nil { + t.Fatalf("ReadLogHeader: %v", err) + } + if h.URL != testHeader().URL || len(h.Rules) != 1 { + t.Errorf("header = %+v, want the written one", h) + } + }) + + t.Run("a half-written header is not a header", func(t *testing.T) { + path := filepath.Join(dir, "torn.jsonl") + if err := os.WriteFile(path, []byte(`{"type":"header","url":"htt`), 0o644); err != nil { + t.Fatalf("write: %v", err) + } + if _, err := ReadLogHeader(path); err == nil || + !strings.Contains(err.Error(), "no complete header") { + t.Fatalf("ReadLogHeader() = %v, want a refusal", err) + } + }) + + t.Run("a wrong schema version is refused", func(t *testing.T) { + path := filepath.Join(dir, "old.jsonl") + if err := os.WriteFile(path, []byte(`{"type":"header","schema_version":99,"url":"u"}`+"\n"), 0o644); err != nil { + t.Fatalf("write: %v", err) + } + if _, err := ReadLogHeader(path); err == nil || + !strings.Contains(err.Error(), "schema version 99") { + t.Fatalf("ReadLogHeader() = %v, want a schema refusal", err) + } + }) +} diff --git a/grafana-alertcheck/internal/gate/check_unix.go b/grafana-alertcheck/internal/gate/check_unix.go new file mode 100644 index 000000000..40ffa2ae3 --- /dev/null +++ b/grafana-alertcheck/internal/gate/check_unix.go @@ -0,0 +1,36 @@ +//go:build unix + +package gate + +import ( + "errors" + "fmt" + "syscall" +) + +// There is deliberately no Windows counterpart, for the same reason +// flock_unix.go and watch_unix.go have none: runners are Linux, goreleaser +// builds linux+darwin only (P12), and a package that does not build there is +// safer than one that silently skips the stop protocol. + +// signalRecorder asks the recorder to stop (§4.4 step 2). +// +// The caller must have established that a writer is alive — by taking the +// log's flock and being refused — before it calls this. Nothing removes the +// pidfile when a recorder exits cleanly, so a pid read without that proof can +// name any same-user process that has since inherited it. +// +// gone reports ESRCH. Given the lock proof, that is a broken contract rather +// than a clean stop, and stopRecorder treats it as one; the value is reported +// instead of raised here because this function knows the errno and not what +// it means. +func signalRecorder(pid int) (gone bool, err error) { + switch err := syscall.Kill(pid, syscall.SIGTERM); { + case err == nil: + return false, nil + case errors.Is(err, syscall.ESRCH): + return true, nil + default: + return false, fmt.Errorf("signal recorder pid %d: %w", pid, err) + } +} diff --git a/grafana-alertcheck/internal/gate/classify.go b/grafana-alertcheck/internal/gate/classify.go index 48a312048..5bfa237a4 100644 --- a/grafana-alertcheck/internal/gate/classify.go +++ b/grafana-alertcheck/internal/gate/classify.go @@ -512,8 +512,17 @@ func decide(h Header, polls []Poll, sentinel *time.Time, defs []Definition, unobservableNames []string ) + // `skipped` is decided from the header, never from defs (§12). defs are + // resolved after the window has closed, so Definition.IsPaused describes + // the present; Header.pausedAtStart describes the moment the recording + // opened, which is the only moment "paused before the window opened" can + // mean. Reading the late definition instead let a rule that fired and was + // then paused report as skipped, with its firing never classified — and + // under AllowPaused that was a pass. + pausedAtStart := h.pausedAtStart() + for _, def := range defs { - if def.IsPaused { + if pausedAtStart[def.UID] { skippedRules = append(skippedRules, def) result.Verdicts = append(result.Verdicts, RuleVerdict{ Alert: def.Title, RuleUID: def.UID, Outcome: OutcomeSkipped, diff --git a/grafana-alertcheck/internal/gate/classify_test.go b/grafana-alertcheck/internal/gate/classify_test.go index 7176a9167..25d766b6c 100644 --- a/grafana-alertcheck/internal/gate/classify_test.go +++ b/grafana-alertcheck/internal/gate/classify_test.go @@ -34,6 +34,18 @@ func quietPoll(uid string, at time.Time) Poll { var defaultBad = badStateSet(nil) // {firing} +// pausedHeader builds the header decide reads `skipped` from: the pause state +// as of record start. Definition.IsPaused is deliberately NOT that authority +// — it comes from a ruler read taken after the window closed — so a test that +// wants a rule treated as skipped must say so HERE (Header.pausedAtStart). +func pausedHeader(startedAt time.Time, pausedUIDs ...string) Header { + h := Header{SchemaVersion: LogSchemaVersion, StartedAt: startedAt} + for _, uid := range pausedUIDs { + h.Rules = append(h.Rules, LoggedRule{UID: uid, IsPaused: true}) + } + return h +} + // --- clean / newly_bad --- func TestClassifyRule_NoEvidenceIsClean(t *testing.T) { @@ -310,7 +322,9 @@ func TestDecide_SkippedRuleNeverReachesProveCoverage(t *testing.T) { // No polls, no sentinel at all: a heartbeat_gap/no_sentinel misclassification // here would mean proveCoverage ran for a skipped rule (§4.3's obligation). - res, err := decide(Header{StartedAt: from.Add(-time.Hour)}, nil, nil, defs, rt, gt, pol) + // The HEADER is what says paused — decide reads skipped from there, not + // from def.IsPaused, which is a post-window reading (Header.pausedAtStart). + res, err := decide(pausedHeader(from.Add(-time.Hour), "r1"), nil, nil, defs, rt, gt, pol) if err != nil { t.Fatalf("err = %v, want nil: a rule paused before the window is skipped, not unobservable", err) } @@ -445,7 +459,7 @@ func TestDecide_SkippedOnlyShortfallProducesAViolationWithoutAnError(t *testing. } sentinel := to - res, err := decide(Header{StartedAt: from.Add(-time.Hour)}, polls, &sentinel, defs, rt, gt, pol) + res, err := decide(pausedHeader(from.Add(-time.Hour), "paused"), polls, &sentinel, defs, rt, gt, pol) if err != nil { t.Fatalf("err = %v, want nil: a shortfall caused only by a skipped rule is exit 1, not exit 2 (§9.1)", err) } @@ -516,7 +530,7 @@ func TestDecide_AllowPausedSuppressesTheShortfall(t *testing.T) { } sentinel := to - res, err := decide(Header{StartedAt: from.Add(-time.Hour)}, polls, &sentinel, defs, rt, gt, pol) + res, err := decide(pausedHeader(from.Add(-time.Hour), "paused"), polls, &sentinel, defs, rt, gt, pol) if err != nil { t.Fatalf("err = %v, want nil", err) } diff --git a/grafana-alertcheck/internal/gate/coverage.go b/grafana-alertcheck/internal/gate/coverage.go index 008406033..03b1e1a75 100644 --- a/grafana-alertcheck/internal/gate/coverage.go +++ b/grafana-alertcheck/internal/gate/coverage.go @@ -8,22 +8,25 @@ import ( // keepLastReason is the instance Reason that check 9 watches for (§10.2). const keepLastReason = "KeepLast" -// Obligations this phase leaves for later ones — carried forward the same -// way P6's own deviations list did, so a later review has something concrete -// to check against: +// Two things this file deliberately does not do, and where they are done +// instead — both were open obligations when P7 was written, and both are now +// discharged: // -// - fromFutureTolerance (§5: 60s) has no constant and no hard-error check -// anywhere yet. Check 2 below implements only "from < StartedAt"; the -// second clause — from more than fromFutureTolerance ahead is a hard -// error — is once-per-run input validation, not a per-rule coverage -// check, and belongs to Check's construction in a later phase (P9). -// - decide (P8) must read a rule's skipped status from the definitions -// (LoggedRule.IsPaused / Definition.IsPaused), never from the polls, and -// must do so BEFORE calling proveCoverage for that rule: a rule paused -// before the window opened is never scheduled or polled (§4.3), so it -// reaches this function with zero polls and today reads as one large -// heartbeat_gap, not skipped (pinned by +// - §7's second clause, "from more than fromFutureTolerance ahead is a hard +// error", is once-per-run input validation rather than a per-rule +// coverage check, and this function has no error return. Discharged by +// P9: the constant is fromFutureTolerance (schedule.go) and Config.validate +// (check.go) applies it. Check 2 below still owns the first clause, +// "from < StartedAt". +// - A rule paused before the window opened is never scheduled or polled +// (§4.3), so it reaches this function with zero polls and reads as one +// large heartbeat_gap, not as skipped (pinned by // TestProveCoverage_SkippedRuleWithZeroPollsPinnedAsHeartbeatGap). +// Discharged by P8: decide returns before it ever calls proveCoverage for +// such a rule (classify.go). It reads skipped from the log header +// (Header.pausedAtStart), NOT from Definition.IsPaused — the definitions +// are re-resolved after the window closed, so they cannot answer what was +// paused when it opened. // UnobservableReason names why proveCoverage could not prove a rule's window. // It is machine-readable — this reaches the action's JSON outputs, so it is a diff --git a/grafana-alertcheck/internal/gate/flock.go b/grafana-alertcheck/internal/gate/flock.go index af4cf99fe..f2a3dbb24 100644 --- a/grafana-alertcheck/internal/gate/flock.go +++ b/grafana-alertcheck/internal/gate/flock.go @@ -23,3 +23,23 @@ func lockExclusive(f *os.File) error { func isLockContention(err error) bool { return errors.Is(err, syscall.EWOULDBLOCK) || errors.Is(err, syscall.EAGAIN) } + +// tryLockExclusive is the same call read as a question rather than as a +// demand: held is false when another process holds the lock, and err is +// non-nil only for a failure that is not contention. +// +// check needs that distinction where NewWriter does not. NewWriter is entitled +// to treat any refusal as "another writer has it", because it wants the lock; +// check only wants to know whether a writer EXISTS (§4.4). The lock answers +// that directly, where a pid can only infer it — the kernel releases a flock +// when the holder exits, crash included, and pids get reused. +func tryLockExclusive(f *os.File) (held bool, err error) { + switch err := syscall.Flock(int(f.Fd()), syscall.LOCK_EX|syscall.LOCK_NB); { + case err == nil: + return true, nil + case errors.Is(err, syscall.EWOULDBLOCK): + return false, nil + default: + return false, fmt.Errorf("flock %s: %w", f.Name(), err) + } +} diff --git a/grafana-alertcheck/internal/gate/log.go b/grafana-alertcheck/internal/gate/log.go index a66b508e1..243060535 100644 --- a/grafana-alertcheck/internal/gate/log.go +++ b/grafana-alertcheck/internal/gate/log.go @@ -1,6 +1,7 @@ package gate import ( + "bufio" "encoding/json" "fmt" "os" @@ -39,16 +40,22 @@ type LoggedRule struct { Title string `json:"title"` Folder string `json:"folder"` Group string `json:"group"` - // ForSeconds, IntervalSeconds, IsPaused, NoDataState and ExecErrState are - // purely forensic: a resolve-time snapshot that makes the uploaded - // artifact self-describing to a human reading it after the runner is gone - // (§21.3). check never converts them back into a Definition — it always - // re-resolves definitions from the ruler API (§19.1 step 2). + // ForSeconds, IntervalSeconds, NoDataState and ExecErrState are purely + // forensic: a resolve-time snapshot that makes the uploaded artifact + // self-describing to a human reading it after the runner is gone (§21.3). + // check never converts them back into a Definition — it always re-resolves + // definitions from the ruler API (§19.1 step 2). ForSeconds float64 `json:"for_seconds"` IntervalSeconds int `json:"interval_seconds"` - IsPaused bool `json:"is_paused"` - NoDataState string `json:"no_data_state"` - ExecErrState string `json:"exec_err_state"` + // IsPaused is NOT forensic, and is the second load-bearing field here + // beside PollEverySeconds. It is the pause state at record start, which is + // the only moment `skipped` can honestly mean (§12), and decide reads it + // through Header.pausedAtStart rather than reading Definition.IsPaused off + // a ruler read taken after the window had already closed. See that method + // for what goes wrong the other way. + IsPaused bool `json:"is_paused"` + NoDataState string `json:"no_data_state"` + ExecErrState string `json:"exec_err_state"` // PollEverySeconds is the cadence this recording ACTUALLY used, after any // --poll-interval override. Load-bearing, not forensic: check derives // maxGap from it and never re-derives it from the definitions. Getting @@ -69,6 +76,30 @@ type Header struct { Rules []LoggedRule `json:"rules"` // THE alert set (§19.1 step 3) } +// pausedAtStart reports, per rule UID, whether the rule was paused when the +// recording opened. That instant — and no other — is what `skipped` means +// (§12): a rule nobody was watching on purpose. +// +// It is the authority for `skipped` in BOTH modes, and the reason is that no +// other source knows the right moment. `check` re-resolves the definitions +// AFTER the window closed (§19.1 step 2), so Definition.IsPaused there +// describes the present, not the window: a rule that fired and was then +// paused would read as skipped, its firing would never be classified, and +// under --allow-paused the run would pass. The header cannot drift that way, +// because watch stamps it before the deploy step runs and single-step check +// stamps it from definitions resolved at the start of its own step. +// +// A UID the header does not name is reported NOT paused, which is the safe +// direction: it then reaches proveCoverage with no polls and fails closed as +// a heartbeat gap, rather than being waved through as legitimately unwatched. +func (h Header) pausedAtStart() map[string]bool { + paused := make(map[string]bool, len(h.Rules)) + for _, lr := range h.Rules { + paused[lr.UID] = lr.IsPaused + } + return paused +} + // Poll is one reduced observation of one rule — the log's heartbeat and the // only input the pure coverage and classification layers ever see. type Poll struct { @@ -167,13 +198,7 @@ func (r *Reducer) Reduce(uid string, obs Observation) Poll { LatencyMS: obs.Latency.Milliseconds(), } - var rule *StateRule - for i := range obs.Rules { - if obs.Rules[i].UID == uid { - rule = &obs.Rules[i] - break - } - } + rule := stateRuleByUID(obs.Rules, uid) if rule == nil { // An authoritative "the rule is absent". No markers are computed and // the previous abnormal set is kept untouched: if the rule comes back @@ -265,6 +290,25 @@ func (r *Reducer) seedFrom(polls []Poll) { } } +// stateRuleByUID picks one rule out of a state-endpoint response BY UID, and +// nil means the response is an authoritative "the rule is absent" (§14.5). +// +// Never by title: the ?rule_name= filter is a title filter, and a filtered +// response can carry several rules sharing one title (the known 2-way +// collision), so picking the first would silently watch the wrong rule. This +// is the single implementation of that selection for the package — Reduce +// above and the drain wait (check.go) both call it, for the same reason +// pollsForRule (classify.go) is shared between proveCoverage and classifyRule: +// two copies of a membership test are two chances for one to drift. +func stateRuleByUID(rules []StateRule, uid string) *StateRule { + for i := range rules { + if rules[i].UID == uid { + return &rules[i] + } + } + return nil +} + // reasonNames reports whether reason names want. Newer Grafana versions // comma-join several reasons into one string, so this tests membership rather // than equality (P7 check 9 needs the same test for KeepLast). @@ -463,6 +507,47 @@ func (w *Writer) Close() error { return nil } +// ReadLogHeader reads ONLY line 1 and is the one read of a log that a writer +// may still hold. That is safe for exactly one line and for no other: the +// header is written once, by watch's parent, before any child appends a byte, +// the file is opened O_APPEND and never O_TRUNC (§8), so line 1 is complete +// and immutable for the whole life of the recording. +// +// It exists so check can fail closed EARLY (§19.1 steps 3-4): the log's +// identity, the rule set and the cadences are all knowable at the start, and +// discovering a wrong URL or an unresolvable rule after a ten-minute wait +// helps nobody. It is advisory only — the authoritative read is still ReadLog, +// once, after the writer has exited (§4.4 step 4), and check re-validates the +// identity against that header rather than trusting this one. +func ReadLogHeader(path string) (Header, error) { + f, err := os.Open(path) + if err != nil { + return Header{}, fmt.Errorf("read log header %s: %w", path, err) + } + defer f.Close() + + line, err := bufio.NewReader(f).ReadString('\n') + if err != nil { + // io.EOF included: a log whose first line has no terminating newline is + // a log whose header was never fully written, which is not a header. + return Header{}, fmt.Errorf("log %s: no complete header on line 1: %w", path, err) + } + + var rec headerRecord + if err := json.Unmarshal([]byte(line), &rec); err != nil { + return Header{}, fmt.Errorf("log %s line 1: unparseable header: %w", path, err) + } + if rec.Type != RecordHeader { + return Header{}, fmt.Errorf("log %s line 1: got record type %q; the header must be line 1", path, rec.Type) + } + if rec.SchemaVersion != LogSchemaVersion { + return Header{}, fmt.Errorf( + "log %s: schema version %d is not %d — this log was written by a different version of the gate", + path, rec.SchemaVersion, LogSchemaVersion) + } + return rec.Header, nil +} + // ReadLog reads the whole log once and returns its header, its polls in // recorded order, and the sentinel time when one is present (nil when the // recording never finished — check turns that into unobservable, never a diff --git a/grafana-alertcheck/internal/gate/schedule.go b/grafana-alertcheck/internal/gate/schedule.go index ec89dcb13..fda33cacc 100644 --- a/grafana-alertcheck/internal/gate/schedule.go +++ b/grafana-alertcheck/internal/gate/schedule.go @@ -14,6 +14,16 @@ import ( // P2 needed it before this file did, so it started there. const skewHardLimit = 60 * time.Second +// fromFutureTolerance is how far ahead of the runner's own clock a supplied +// `from` may sit before check refuses it (§7: "from in the future, more than +// the skew tolerance — error"). §7 names no number, so this is the judgment +// call §5's table records: the same 60s as skewHardLimit, because the only +// legitimate reason for a `from` in the future is clock disagreement between +// the deploy step and the check step, and that is bounded by the same figure. +// It is once-per-run input validation, not a per-rule coverage check, so +// Check applies it (P9) and proveCoverage does not. +const fromFutureTolerance = 60 * time.Second + // minDrainTimeout is §5's floor on drainTimeout: max(2 x max(intervalSeconds), // 2m). Without the floor, a fleet of very tight rules would derive a // drainTimeout too short to let a healthy in-flight poll land. @@ -94,7 +104,22 @@ func DeriveTimings(defs []Definition, override time.Duration) (rules map[string] } rules[d.UID] = newRuleTimings(pollEvery, d.IntervalSeconds) } - return rules, deriveGlobalTimings(defs), notes + // In this mode the defs ARE the start-of-step snapshot — watch resolves + // them before it detaches, and single-step check before its first + // observation — so they can answer what was paused when the window opened. + // Only the log-mode counterpart below has to look elsewhere. + return rules, deriveGlobalTimings(defs, pausedSet(defs)), notes +} + +// pausedSet is Header.pausedAtStart's counterpart for a set of definitions +// resolved at the start of the step, which is the one moment a definition can +// answer "was this paused when the window opened". +func pausedSet(defs []Definition) map[string]bool { + paused := make(map[string]bool, len(defs)) + for _, d := range defs { + paused[d.UID] = d.IsPaused + } + return paused } // DeriveTimingsFromLog is DeriveTimings' log-mode counterpart, and the two @@ -155,17 +180,31 @@ func DeriveTimingsFromLog(h Header, defs []Definition) (rules map[string]ruleTim pollEvery := time.Duration(lr.PollEverySeconds * float64(time.Second)) rules[lr.UID] = newRuleTimings(pollEvery, def.IntervalSeconds) } - return rules, deriveGlobalTimings(defs), nil + // The header, not defs, decides which rules are excluded from the grace: + // defs were resolved after the window closed. See deriveGlobalTimings. + return rules, deriveGlobalTimings(defs, h.pausedAtStart()), nil } // deriveGlobalTimings computes transitionGrace and drainTimeout over defs -// (§5, §13.1, §19). A rule paused before the window opened — skipped, §12 — -// is excluded from the transitionGrace max: its `for` value can never fire -// during the window, so counting it would only inflate the wait past what any -// watched rule actually needs (a judgment call the v2 plan makes explicitly -// for this formula; §19's drainTimeout carries no such exclusion, so it still -// runs over every resolved rule). -func deriveGlobalTimings(defs []Definition) globalTimings { +// (§5, §13.1, §19). +// +// A rule paused before the window opened — skipped, §12 — is excluded from the +// transitionGrace max: its `for` value can never fire during the window, so +// counting it would only inflate the wait past what any watched rule actually +// needs (a judgment call the v2 plan makes explicitly for this formula; §19's +// drainTimeout carries no such exclusion, so it still runs over every resolved +// rule). +// +// "Before the window opened" is the whole content of that exclusion, so the +// authority is pausedAtStart and NEVER Definition.IsPaused: in log mode the +// definitions are re-resolved after the window closed. Reading them instead +// was a fail-open, and a quiet one. transitionGrace is what lets a condition +// arising just before `to` be seen when it surfaces at to + `for`, and +// windowEnd is BOTH the classification bound and the collection deadline — so +// a rule somebody paused after `to` dropped out of the max, the grace +// collapsed, the surfacing poll was never even recorded, and the run reported +// clean. With one watched rule the shrink is total. +func deriveGlobalTimings(defs []Definition, pausedAtStart map[string]bool) globalTimings { var g globalTimings var maxInterval time.Duration for _, d := range defs { @@ -173,7 +212,7 @@ func deriveGlobalTimings(defs []Definition) globalTimings { if interval > maxInterval { maxInterval = interval } - if d.IsPaused { + if pausedAtStart[d.UID] { continue } if candidate := d.For + interval; candidate > g.transitionGrace { diff --git a/grafana-alertcheck/internal/gate/schedule_test.go b/grafana-alertcheck/internal/gate/schedule_test.go index e897414bc..b1871bb3d 100644 --- a/grafana-alertcheck/internal/gate/schedule_test.go +++ b/grafana-alertcheck/internal/gate/schedule_test.go @@ -77,6 +77,70 @@ func TestDeriveTimings_TransitionGraceZeroWhenAllSkipped(t *testing.T) { } } +// TestDeriveTimingsFromLog_TransitionGraceFollowsTheHeaderNotTheDefinition +// pins the log-mode authority for the grace exclusion. Definitions are +// re-resolved AFTER the window closed, so "paused" in a definition says +// nothing about whether the rule was watched during it. +// +// The fail-open direction is the first case. transitionGrace exists so a +// condition arising just before `to` is still seen when it surfaces at +// to + `for`, and windowEnd is both the classification bound and the +// collection deadline — so a rule somebody paused after `to` dropping out of +// the max collapses the grace, the surfacing poll is never recorded, and the +// run reports clean. +func TestDeriveTimingsFromLog_TransitionGraceFollowsTheHeaderNotTheDefinition(t *testing.T) { + loggedRule := func(uid string, pausedAtStart bool) LoggedRule { + return LoggedRule{UID: uid, Title: uid, IntervalSeconds: 60, PollEverySeconds: 30, IsPaused: pausedAtStart} + } + // The definition says paused in BOTH cases: it is the post-window reading, + // and it must change nothing. + defs := []Definition{{UID: "r1", Title: "R1", IntervalSeconds: 60, For: 5 * time.Minute, IsPaused: true}} + want := 5*time.Minute + 60*time.Second + + t.Run("header says active: the rule stays in the max", func(t *testing.T) { + h := Header{Rules: []LoggedRule{loggedRule("r1", false)}} + _, global, err := DeriveTimingsFromLog(h, defs) + if err != nil { + t.Fatalf("DeriveTimingsFromLog: %v", err) + } + if global.transitionGrace != want { + t.Fatalf("transitionGrace = %s, want %s: the rule was active when the recording opened, "+ + "so a pause applied afterwards must not shrink the window", global.transitionGrace, want) + } + if !strings.Contains(global.graceSource, "R1") { + t.Errorf("graceSource = %q, want it to name R1", global.graceSource) + } + }) + + t.Run("header says paused: the rule stays out", func(t *testing.T) { + h := Header{Rules: []LoggedRule{loggedRule("r1", true)}} + _, global, err := DeriveTimingsFromLog(h, defs) + if err != nil { + t.Fatalf("DeriveTimingsFromLog: %v", err) + } + if global.transitionGrace != 0 { + t.Fatalf("transitionGrace = %s, want 0: a rule paused before the window opened can never fire during it", + global.transitionGrace) + } + }) + + t.Run("drainTimeout counts every rule either way", func(t *testing.T) { + // §19 puts no pause exclusion on drainTimeout, so both headers give the + // same floor-bound value. + for _, pausedAtStart := range []bool{false, true} { + h := Header{Rules: []LoggedRule{loggedRule("r1", pausedAtStart)}} + _, global, err := DeriveTimingsFromLog(h, defs) + if err != nil { + t.Fatalf("DeriveTimingsFromLog: %v", err) + } + if global.drainTimeout != minDrainTimeout { + t.Fatalf("drainTimeout = %s with pausedAtStart=%v, want the %s floor", + global.drainTimeout, pausedAtStart, minDrainTimeout) + } + } + }) +} + func TestDeriveTimings_DrainTimeoutIncludesPaused(t *testing.T) { defs := []Definition{{UID: "r1", Title: "R1", IntervalSeconds: 10}} _, global, _ := DeriveTimings(defs, 0) diff --git a/grafana-alertcheck/internal/gate/watch.go b/grafana-alertcheck/internal/gate/watch.go index f0f24f247..f8726fec6 100644 --- a/grafana-alertcheck/internal/gate/watch.go +++ b/grafana-alertcheck/internal/gate/watch.go @@ -372,7 +372,6 @@ func openRecording(ctx context.Context, cfg WatchConfig, src Source, writer *Wri // it as skipped from the definitions. var active []Definition activeTimings := make(map[string]ruleTimings, len(resolved)) - titles := make(map[string]string, len(resolved)) for _, d := range resolved { if d.IsPaused { fmt.Fprintf(cfg.Notes, "note: rule %q (%s) is paused: recorded as skipped, not waited for (§4.3)\n", d.Title, d.UID) @@ -380,48 +379,15 @@ func openRecording(ctx context.Context, cfg WatchConfig, src Source, writer *Wri } active = append(active, d) activeTimings[d.UID] = rt[d.UID] - titles[d.UID] = d.Title fmt.Fprintf(cfg.Notes, "recording %q (%s) every %s (maxGap %s)\n", d.Title, d.UID, rt[d.UID].pollEvery, rt[d.UID].maxGap) } - uids := make([]string, 0, len(active)) - for _, d := range active { - uids = append(uids, d.UID) - } - observed, err := observeAll(ctx, src, titles, uids, cfg.Concurrency) + polls, measured, err := firstObservations(ctx, src, active, NewReducer(), cfg.Concurrency, cfg.Notes) if err != nil { return nil, err } - - // Verify §3.2 before anything downstream relies on it: if the state - // endpoint ever stops returning normal instances, the reduction's "keep - // the non-normal ones" silently becomes "keep everything it happened to - // send" and the transition markers lose their ground truth. - for _, d := range active { - if err := VerifyNormalInstancesVisible(observed[d.UID].Rules); err != nil { - return nil, err - } - } - - // One poll record per rule, in resolve order so the log is byte-stable for - // a given set of observations. These ARE the log's first heartbeats: they - // predate the deploy step, which is the whole point of §4.3. - reducer := NewReducer() - measured := make(map[string]time.Duration, len(active)) - for _, d := range active { - obs := observed[d.UID] - measured[d.UID] = obs.Latency - poll := reducer.Reduce(d.UID, obs) - if !poll.Found { - // Authoritative, not transient (P2 already retried transport - // failures): the rule resolved in the ruler API but the state - // endpoint does not serve it. Recorded as Found=false, which P7 - // turns into unobservable — a note rather than an error here, - // because the state endpoint can lag a freshly created rule and - // check re-resolves and fails closed either way. - fmt.Fprintf(cfg.Notes, "warning: rule %q (%s) is absent from the state endpoint; recorded as not found\n", d.Title, d.UID) - } - if err := writer.WritePoll(poll); err != nil { + for _, p := range polls { + if err := writer.WritePoll(p); err != nil { return nil, err } } @@ -460,6 +426,66 @@ func loggedRules(defs []Definition, rt map[string]ruleTimings) []LoggedRule { return out } +// firstObservations takes one observation of every rule in active, verifies +// §3.2 against those very responses, and reduces each into the poll record +// that IS the window's first heartbeat — plus the measured latency of each, +// which is the only honest input to §5.2's budget check (a fixed estimate is +// worthless when one rule's payload is ~230x another's). +// +// Both entry paths share it: watch's parent, before it detaches (§4.3), and +// single-step check's measurement pass, which keeps the polls as evidence +// rather than writing them to a log (P9). Keeping one implementation is the +// point — the §3.2 verification and the "absent is a warning, not an error" +// rule are exactly the places where two copies would silently drift, and a +// drift in either direction is fail-open. +// +// polls come back in `active` order, so a log written from them is byte-stable +// for a given set of observations. +func firstObservations(ctx context.Context, src Source, active []Definition, reducer *Reducer, + concurrency int, notes io.Writer) ([]Poll, map[string]time.Duration, error) { + + titles := make(map[string]string, len(active)) + uids := make([]string, 0, len(active)) + for _, d := range active { + titles[d.UID] = d.Title + uids = append(uids, d.UID) + } + + observed, err := observeAll(ctx, src, titles, uids, concurrency) + if err != nil { + return nil, nil, err + } + + // Verify §3.2 before anything downstream relies on it: if the state + // endpoint ever stops returning normal instances, the reduction's "keep + // the non-normal ones" silently becomes "keep everything it happened to + // send" and the transition markers lose their ground truth. + for _, d := range active { + if err := VerifyNormalInstancesVisible(observed[d.UID].Rules); err != nil { + return nil, nil, err + } + } + + polls := make([]Poll, 0, len(active)) + measured := make(map[string]time.Duration, len(active)) + for _, d := range active { + obs := observed[d.UID] + measured[d.UID] = obs.Latency + poll := reducer.Reduce(d.UID, obs) + if !poll.Found { + // Authoritative, not transient (P2 already retried transport + // failures): the rule resolved in the ruler API but the state + // endpoint does not serve it. Recorded as Found=false, which P7 + // check 8 turns into unobservable — a note rather than an error + // here, because the state endpoint can lag a freshly created rule + // and the coverage proof fails closed either way. + fmt.Fprintf(notes, "warning: rule %q (%s) is absent from the state endpoint; recorded as not found\n", d.Title, d.UID) + } + polls = append(polls, poll) + } + return polls, measured, nil +} + // observeAll polls every rule in uids concurrently, bounded by concurrency, // and returns one Observation per rule that answered. Every rule is polled by // TITLE (the ?rule_name= filter, §2.8) and selected out of the response by diff --git a/grafana-alertcheck/internal/gate/watch_daemon_test.go b/grafana-alertcheck/internal/gate/watch_daemon_test.go index 35e9c424a..db35c505b 100644 --- a/grafana-alertcheck/internal/gate/watch_daemon_test.go +++ b/grafana-alertcheck/internal/gate/watch_daemon_test.go @@ -7,6 +7,7 @@ import ( "net/http" "net/http/httptest" "os" + "os/signal" "path/filepath" "slices" "strconv" @@ -22,12 +23,48 @@ import ( // inherited environment, a real SIGTERM — with this function standing in for // the CLI's `watch --daemon-child` dispatch, which lands in P10. func TestMain(m *testing.M) { + if path := os.Getenv(lockHolderEnv); path != "" { + os.Exit(runTestLockHolder(path)) + } if slices.Contains(os.Args, DaemonChildFlag) { os.Exit(runTestDaemonChild(os.Args[1:])) } os.Exit(m.Run()) } +// lockHolderEnv turns this test binary into a stand-in recorder that holds the +// log's flock and refuses to die: a process check's stop protocol must wait +// for and, on a timeout, refuse to read around. +// +// It has to be a real second process. flock is what stopRecorder probes, and +// there is no flock(1) on darwin, so a shell one-liner cannot take the lock — +// while a lock taken in the test process itself would be granted to the probe +// on some platforms and prove nothing. +const lockHolderEnv = "GRAFANA_ALERTCHECK_TEST_LOCK_HOLDER" + +// runTestLockHolder takes the log's exclusive lock, reports that it has it on +// stdout, ignores SIGTERM, and waits to be killed. The report is what lets the +// test start only once the lock is genuinely held, rather than racing it. +func runTestLockHolder(path string) int { + signal.Ignore(syscall.SIGTERM) + + f, err := os.OpenFile(path, os.O_APPEND|os.O_CREATE|os.O_WRONLY, 0o644) + if err != nil { + fmt.Fprintln(os.Stderr, err) + return 1 + } + if err := lockExclusive(f); err != nil { + fmt.Fprintln(os.Stderr, err) + return 1 + } + fmt.Println("locked") + + // Long enough to outlive any test that starts it; the test kills it, and + // SIGKILL is not ignorable. + time.Sleep(5 * time.Minute) + return 0 +} + // runTestDaemonChild parses the child argv childArgs() writes, and reads the // connection details from the environment — never from argv (§20.2). P10's // `watch` FlagSet does the same four flags. From 7c3f20693a2ec693b6fde4a778dde4934694471b Mon Sep 17 00:00:00 2001 From: Bartek Tofel Date: Wed, 2 Sep 2026 13:22:33 +0200 Subject: [PATCH 26/43] chore: remove unix build tag --- .../internal/gate/{check_unix.go => check_process.go} | 7 ------- 1 file changed, 7 deletions(-) rename grafana-alertcheck/internal/gate/{check_unix.go => check_process.go} (76%) diff --git a/grafana-alertcheck/internal/gate/check_unix.go b/grafana-alertcheck/internal/gate/check_process.go similarity index 76% rename from grafana-alertcheck/internal/gate/check_unix.go rename to grafana-alertcheck/internal/gate/check_process.go index 40ffa2ae3..580a2a143 100644 --- a/grafana-alertcheck/internal/gate/check_unix.go +++ b/grafana-alertcheck/internal/gate/check_process.go @@ -1,5 +1,3 @@ -//go:build unix - package gate import ( @@ -8,11 +6,6 @@ import ( "syscall" ) -// There is deliberately no Windows counterpart, for the same reason -// flock_unix.go and watch_unix.go have none: runners are Linux, goreleaser -// builds linux+darwin only (P12), and a package that does not build there is -// safer than one that silently skips the stop protocol. - // signalRecorder asks the recorder to stop (§4.4 step 2). // // The caller must have established that a writer is alive — by taking the From df1d58a95bf746069e71a44b9b6c53690512a4a3 Mon Sep 17 00:00:00 2001 From: Bartek Tofel Date: Mon, 31 Aug 2026 18:53:07 +0200 Subject: [PATCH 27/43] Wire watch/check subcommands to the gate library, with a table+JSON renderer and H6/H7 exit-code mapping. Extend Result with per-rule/global thresholds and a real skew bound; export SkewHardLimit; reject --states normal. --- .../cmd/grafana-alertcheck/check.go | 148 ++++++++++++++++++ .../cmd/grafana-alertcheck/check_test.go | 147 +++++++++++++++++ .../cmd/grafana-alertcheck/common.go | 108 +++++++++++++ .../cmd/grafana-alertcheck/main.go | 13 +- .../cmd/grafana-alertcheck/table.go | 136 ++++++++++++++++ .../cmd/grafana-alertcheck/table_test.go | 114 ++++++++++++++ .../cmd/grafana-alertcheck/watch.go | 148 ++++++++++++++++++ .../cmd/grafana-alertcheck/watch_test.go | 99 ++++++++++++ grafana-alertcheck/internal/gate/check.go | 5 +- grafana-alertcheck/internal/gate/classify.go | 69 +++++++- grafana-alertcheck/internal/gate/schedule.go | 10 +- grafana-alertcheck/internal/gate/source.go | 6 +- 12 files changed, 981 insertions(+), 22 deletions(-) create mode 100644 grafana-alertcheck/cmd/grafana-alertcheck/check.go create mode 100644 grafana-alertcheck/cmd/grafana-alertcheck/check_test.go create mode 100644 grafana-alertcheck/cmd/grafana-alertcheck/common.go create mode 100644 grafana-alertcheck/cmd/grafana-alertcheck/table.go create mode 100644 grafana-alertcheck/cmd/grafana-alertcheck/table_test.go create mode 100644 grafana-alertcheck/cmd/grafana-alertcheck/watch.go create mode 100644 grafana-alertcheck/cmd/grafana-alertcheck/watch_test.go diff --git a/grafana-alertcheck/cmd/grafana-alertcheck/check.go b/grafana-alertcheck/cmd/grafana-alertcheck/check.go new file mode 100644 index 000000000..f4e9dab26 --- /dev/null +++ b/grafana-alertcheck/cmd/grafana-alertcheck/check.go @@ -0,0 +1,148 @@ +package main + +import ( + "context" + "encoding/json" + "flag" + "fmt" + "io" + "os/signal" + "syscall" + "time" + + "github.com/smartcontractkit/chainlink-testing-framework/grafana-alertcheck/internal/gate" +) + +const checkUsage = "usage: grafana-alertcheck check [--in ] [--pidfile F] --from RFC3339 --to RFC3339 " + + "[--alerts ...] [--folder F] [--states ...] [--preexisting ...] [--min-observed N] [--allow-paused] " + + "[--nodata-is-unobservable] [--concurrency N] [--output json]" + +// runCheck is the classify step's CLI surface: parse flags into a +// gate.Config, run gate.Check, and translate its (Result, error) into +// §20.2/§20.3's output and exit code. All of the correctness lives in +// gate.Check (P9) and decide (P8) — this file's only job is presentation and +// the H6/H7 exit-code mapping, which exitCode below keeps as one pure +// function so it can be tested without a network. +func runCheck(args []string, stdin io.Reader, stdout, stderr io.Writer) int { + fs := flag.NewFlagSet("check", flag.ContinueOnError) + fs.SetOutput(stderr) + fs.Usage = func() { fmt.Fprintln(stderr, checkUsage) } + + common := registerCommon(fs) + in := fs.String("in", "", "path of a log recorded by watch; empty selects single-step mode (§9)") + pidfile := fs.String("pidfile", "", "pidfile of the recorder to stop before reading --in (default .pid)") + from := fs.String("from", "", "the moment the deploy finished, RFC3339 (required in recorder mode, §7)") + to := fs.String("to", "", "the end of the window to classify, RFC3339 (required)") + states := fs.String("states", "", "comma-separated bad states to classify against (default: firing, §13)") + preexisting := fs.String("preexisting", "", "how to judge an instance already bad at `from` (default: fail-unless-recovered, §11.7)") + minObserved := fs.Int("min-observed", 0, "minimum rules that must be observed (default: every resolved rule, §12)") + allowPaused := fs.Bool("allow-paused", false, "do not count a rule paused before the window against --min-observed (§12.1)") + nodataIsUnobservable := fs.Bool("nodata-is-unobservable", false, "treat a sustained health=nodata as unobservable rather than a note (§10.2)") + output := fs.String("output", "", `"json" writes the machine-readable Result to stdout in addition to the table (§20.2); default is the table alone`) + + if err := fs.Parse(args); err != nil { + return 2 + } + if *output != "" && *output != "json" { + fmt.Fprintf(stderr, "--output: unknown value %q (only \"json\" is supported)\n", *output) + return 2 + } + + url, token, err := grafanaEnv() + if err != nil { + fmt.Fprintln(stderr, err) + return 2 + } + alerts, err := readAlerts(stdin, *common.alerts) + if err != nil { + fmt.Fprintln(stderr, err) + return 2 + } + stateList, err := parseStates(*states) + if err != nil { + fmt.Fprintln(stderr, err) + return 2 + } + preexistingPolicy, err := parsePreexisting(*preexisting) + if err != nil { + fmt.Fprintln(stderr, err) + return 2 + } + + cfg := gate.Config{ + URL: url, + Token: token, + Alerts: alerts, + Folder: *common.folder, + States: stateList, + Preexisting: preexistingPolicy, + MinObserved: *minObserved, + AllowPaused: *allowPaused, + NodataIsUnobservable: *nodataIsUnobservable, + Log: *in, + PidFile: *pidfile, + Concurrency: *common.concurrency, + Clock: gate.SystemClock{}, + Notes: stderr, + } + if *to == "" { + fmt.Fprintln(stderr, "check: --to is required (§7)") + return 2 + } + t, err := time.Parse(time.RFC3339, *to) + if err != nil { + fmt.Fprintf(stderr, "--to: %v\n", err) + return 2 + } + cfg.To = t + if *from != "" { + f, err := time.Parse(time.RFC3339, *from) + if err != nil { + fmt.Fprintf(stderr, "--from: %v\n", err) + return 2 + } + cfg.From = f + } + + // SIGINT/SIGTERM cancel the run cleanly rather than leaving an operator's + // Ctrl-C to kill the process mid-collection: Check's collection loop and + // drain wait both already select on ctx.Done() (check.go), so this makes + // an interrupted run fail the way every other could-not-check path does + // — exit 2, never a silently truncated pass. + ctx, stop := signal.NotifyContext(context.Background(), syscall.SIGINT, syscall.SIGTERM) + defer stop() + + result, checkErr := gate.Check(ctx, cfg) + + if err := renderTable(stderr, result); err != nil { + fmt.Fprintln(stderr, err) + } + if checkErr != nil { + fmt.Fprintln(stderr, checkErr) + } + if *output == "json" { + enc := json.NewEncoder(stdout) + enc.SetIndent("", " ") + if err := enc.Encode(result); err != nil { + fmt.Fprintf(stderr, "encode --output json: %v\n", err) + return 2 + } + } + return exitCode(result, checkErr) +} + +// exitCode is §20.3/H6/H7's whole mapping, kept as one pure function of +// exactly what Check returns so it is testable without a network: err != nil +// is exit 2 UNCONDITIONALLY — never 0 and never 1, even alongside real +// violations, because inability beats violation (H6) and an error is never a +// pass (H7). Violations without an error is exit 1. Neither is exit 0. +func exitCode(res gate.Result, err error) int { + switch { + case err != nil: + return 2 + case len(res.Violations) > 0: + return 1 + default: + return 0 + } +} diff --git a/grafana-alertcheck/cmd/grafana-alertcheck/check_test.go b/grafana-alertcheck/cmd/grafana-alertcheck/check_test.go new file mode 100644 index 000000000..dbf81d3a9 --- /dev/null +++ b/grafana-alertcheck/cmd/grafana-alertcheck/check_test.go @@ -0,0 +1,147 @@ +package main + +import ( + "bytes" + "errors" + "os" + "strings" + "testing" + + "github.com/smartcontractkit/chainlink-testing-framework/grafana-alertcheck/internal/gate" +) + +// TestExitCode pins §20.3/H6/H7's mapping directly against exitCode, with no +// network involved: err != nil is exit 2 even alongside violations (H6 — +// inability beats violation), violations alone are exit 1, and neither is 0. +func TestExitCode(t *testing.T) { + tests := []struct { + name string + res gate.Result + err error + want int + }{ + {"pass", gate.Result{}, nil, 0}, + {"violation", gate.Result{Violations: []gate.Violation{{}}}, nil, 1}, + {"error alone", gate.Result{}, errors.New("boom"), 2}, + {"error beats violation", gate.Result{Violations: []gate.Violation{{}}}, errors.New("boom"), 2}, + } + for _, tt := range tests { + t.Run(tt.name, func(t *testing.T) { + if got := exitCode(tt.res, tt.err); got != tt.want { + t.Fatalf("exitCode(...) = %d, want %d", got, tt.want) + } + }) + } +} + +func writeTempAlerts(t *testing.T) string { + t.Helper() + path := t.TempDir() + "/alerts.txt" + if err := os.WriteFile(path, []byte("Some Alert\n"), 0o644); err != nil { + t.Fatal(err) + } + return path +} + +// TestRunCheck_FlagValidation is the flag-validation matrix: every one of +// these must fail before any network call, because Config.validate() (P9) +// runs first — an unreachable GRAFANA_URL succeeding or timing out is a +// different test than these, which check pure input validation. +func TestRunCheck_FlagValidation(t *testing.T) { + tests := []struct { + name string + env bool + args func(t *testing.T) []string + wantErr string + }{ + {"missing env", false, func(t *testing.T) []string { + return []string{"--to", "2026-01-01T00:00:00Z", "--alerts", writeTempAlerts(t)} + }, "GRAFANA_URL"}, + {"missing to", true, func(t *testing.T) []string { + return []string{"--alerts", writeTempAlerts(t)} + }, "--to"}, + {"bad to", true, func(t *testing.T) []string { + return []string{"--to", "not-a-time", "--alerts", writeTempAlerts(t)} + }, "--to"}, + {"bad from", true, func(t *testing.T) []string { + return []string{"--to", "2026-01-01T00:00:00Z", "--from", "not-a-time", "--alerts", writeTempAlerts(t)} + }, "--from"}, + {"bad output", true, func(t *testing.T) []string { + return []string{"--to", "2026-01-01T00:00:00Z", "--output", "xml", "--alerts", writeTempAlerts(t)} + }, "--output"}, + {"bad states", true, func(t *testing.T) []string { + return []string{"--to", "2026-01-01T00:00:00Z", "--states", "bogus", "--alerts", writeTempAlerts(t)} + }, "--states"}, + {"states normal is rejected", true, func(t *testing.T) []string { + // normal is the good state, never a state to classify AS bad + // (R1): accepting it would make --states normal fail every + // healthy instance, the fail-open shape H7 exists to prevent. + return []string{"--to", "2026-01-01T00:00:00Z", "--states", "normal", "--alerts", writeTempAlerts(t)} + }, "--states"}, + {"bad preexisting", true, func(t *testing.T) []string { + return []string{"--to", "2026-01-01T00:00:00Z", "--preexisting", "bogus", "--alerts", writeTempAlerts(t)} + }, "--preexisting"}, + {"alerts with in", true, func(t *testing.T) []string { + return []string{"--to", "2026-01-01T00:00:00Z", "--in", "some.jsonl", "--alerts", writeTempAlerts(t)} + }, "refused"}, + {"no alerts no in", true, func(t *testing.T) []string { + return []string{"--to", "2026-01-01T00:00:00Z"} + }, "no alert names"}, + } + for _, tt := range tests { + t.Run(tt.name, func(t *testing.T) { + if tt.env { + t.Setenv("GRAFANA_URL", "http://example.invalid") + t.Setenv("GRAFANA_TOKEN", "test-token") + } else { + t.Setenv("GRAFANA_URL", "") + t.Setenv("GRAFANA_TOKEN", "") + } + var stdout, stderr bytes.Buffer + args := append([]string{"check"}, tt.args(t)...) + code := run(args, &stdout, &stderr) + if code != 2 { + t.Fatalf("code = %d, want 2; stderr = %q", code, stderr.String()) + } + if !strings.Contains(stderr.String(), tt.wantErr) { + t.Fatalf("stderr = %q, want it to contain %q", stderr.String(), tt.wantErr) + } + }) + } +} + +// TestRunCheck_ToInPastNoLog pins §4.2's refusal: a `to` already in the past +// with no recorded log cannot be classified from anything, because nothing +// ever observed the window. +func TestRunCheck_ToInPastNoLog(t *testing.T) { + t.Setenv("GRAFANA_URL", "http://example.invalid") + t.Setenv("GRAFANA_TOKEN", "test-token") + + var stdout, stderr bytes.Buffer + code := run([]string{"check", + "--from", "1999-01-01T00:00:00Z", "--to", "2000-01-01T00:00:00Z", + "--alerts", writeTempAlerts(t), + }, &stdout, &stderr) + if code != 2 { + t.Fatalf("code = %d, want 2; stderr = %q", code, stderr.String()) + } + if !strings.Contains(stderr.String(), "already passed") { + t.Fatalf("stderr = %q, want the §4.2 refusal", stderr.String()) + } +} + +// TestRunCheck_NoResultOnConfigError pins §20.2: --output json never writes +// to stdout when Check was never reached, because there is no Result to +// encode — only the table (on stderr) can report a configuration failure. +func TestRunCheck_NoResultOnConfigError(t *testing.T) { + t.Setenv("GRAFANA_URL", "") + t.Setenv("GRAFANA_TOKEN", "") + var stdout, stderr bytes.Buffer + code := run([]string{"check", "--to", "2026-01-01T00:00:00Z", "--output", "json"}, &stdout, &stderr) + if code != 2 { + t.Fatalf("code = %d, want 2", code) + } + if stdout.Len() != 0 { + t.Fatalf("stdout = %q, want empty", stdout.String()) + } +} diff --git a/grafana-alertcheck/cmd/grafana-alertcheck/common.go b/grafana-alertcheck/cmd/grafana-alertcheck/common.go new file mode 100644 index 000000000..e605c029b --- /dev/null +++ b/grafana-alertcheck/cmd/grafana-alertcheck/common.go @@ -0,0 +1,108 @@ +package main + +import ( + "bufio" + "flag" + "fmt" + "io" + "os" + "strings" + + "github.com/smartcontractkit/chainlink-testing-framework/grafana-alertcheck/internal/gate" +) + +// commonFlags is registerCommon's result: the exactly three flags watch and +// check share (§20). Connection details are never flags (§20.2) and states / +// poll-interval are deliberately NOT here — states is check-only because +// recording is unfiltered (P6), and poll-interval is watch-only because check +// reads the cadence from the log header (P5). Putting either here would +// silently reinstate a knob this plan removed. +type commonFlags struct { + folder *string + concurrency *int + alerts *string +} + +func registerCommon(fs *flag.FlagSet) *commonFlags { + return &commonFlags{ + folder: fs.String("folder", "", "default folder to scope an unqualified alert name to (§17)"), + concurrency: fs.Int("concurrency", 1, "maximum concurrent requests to Grafana"), + alerts: fs.String("alerts", "", "path to a file of alert names, one per line, or - for stdin"), + } +} + +// readAlerts reads §17's alert names, one per line, from a file or from +// stdin when path is "-". An empty path is not an error here — watch and +// check each decide for themselves whether an empty list is allowed +// (log mode never wants one; single-step / record mode always does). +func readAlerts(stdin io.Reader, path string) ([]string, error) { + if path == "" { + return nil, nil + } + var r io.Reader + if path == "-" { + r = stdin + } else { + f, err := os.Open(path) + if err != nil { + return nil, fmt.Errorf("read --alerts %s: %w", path, err) + } + defer f.Close() + r = f + } + var lines []string + sc := bufio.NewScanner(r) + for sc.Scan() { + lines = append(lines, sc.Text()) + } + if err := sc.Err(); err != nil { + return nil, fmt.Errorf("read --alerts %s: %w", path, err) + } + return lines, nil +} + +// parseStates parses check's --states flag: a comma-separated list of the +// "bad" state vocabulary Config.States matches against (§13, classify.go's +// badStateSet). An empty string is not resolved here — it means "use the +// library default of {firing}" — so this returns nil, nil for "" rather than +// an error. +// +// normal is deliberately NOT accepted: the v2 plan fixes this vocabulary to +// firing | pending | nodata | error (line 378) precisely because "normal" is +// the good state, never a bad one to classify against. Accepting it here +// would let --states normal turn every healthy instance into a violation and +// fail every healthy fleet — the exact fail-open shape H7 exists to prevent. +func parseStates(s string) ([]gate.State, error) { + if strings.TrimSpace(s) == "" { + return nil, nil + } + var out []gate.State + for _, part := range strings.Split(s, ",") { + part = strings.TrimSpace(part) + if part == "" { + continue + } + switch gate.State(part) { + case gate.StateFiring, gate.StatePending, gate.StateNodata, gate.StateError: + out = append(out, gate.State(part)) + default: + return nil, fmt.Errorf("--states: unknown state %q (want any of: firing, pending, nodata, error)", part) + } + } + if len(out) == 0 { + return nil, fmt.Errorf("--states: %q named no state", s) + } + return out, nil +} + +// parsePreexisting parses check's --preexisting flag (§11.7). +func parsePreexisting(s string) (gate.PreexistingPolicy, error) { + switch gate.PreexistingPolicy(s) { + case "": + return gate.PreexistingFailUnlessRecovered, nil + case gate.PreexistingFailUnlessRecovered, gate.PreexistingFail, gate.PreexistingIgnore: + return gate.PreexistingPolicy(s), nil + default: + return "", fmt.Errorf("--preexisting: unknown policy %q (want one of: fail-unless-recovered, fail, ignore)", s) + } +} diff --git a/grafana-alertcheck/cmd/grafana-alertcheck/main.go b/grafana-alertcheck/cmd/grafana-alertcheck/main.go index d7968f5a8..73d30f3f3 100644 --- a/grafana-alertcheck/cmd/grafana-alertcheck/main.go +++ b/grafana-alertcheck/cmd/grafana-alertcheck/main.go @@ -1,8 +1,5 @@ -// Command grafana-alertcheck is the CLI entry point for the gate. P3 wires -// only the `list` subcommand — enough to validate auth, the ruler parse, and -// resolution against a real Grafana before any coverage logic exists (§9 rule -// 4, "reach runnable at PR 5"). P10 extends this file with `watch` and -// `check`. +// Command grafana-alertcheck is the CLI entry point for the gate: `list` +// (P3), `watch` (record, P10) and `check` (classify, P10). package main import ( @@ -15,7 +12,7 @@ func main() { os.Exit(run(os.Args[1:], os.Stdout, os.Stderr)) } -const usage = "usage: grafana-alertcheck " +const usage = "usage: grafana-alertcheck " // run is the whole of main's testable surface: parse the subcommand, dispatch, // return the process exit code. Exit codes below 2 (pass/violations) belong to @@ -36,6 +33,10 @@ func run(args []string, stdout, stderr io.Writer) int { switch args[0] { case "list": return runList(args[1:], stdout, stderr) + case "watch": + return runWatch(args[1:], os.Stdin, stdout, stderr) + case "check": + return runCheck(args[1:], os.Stdin, stdout, stderr) case "-h", "-help", "--help": fmt.Fprintln(stdout, usage) return 0 diff --git a/grafana-alertcheck/cmd/grafana-alertcheck/table.go b/grafana-alertcheck/cmd/grafana-alertcheck/table.go new file mode 100644 index 000000000..0e4c93c67 --- /dev/null +++ b/grafana-alertcheck/cmd/grafana-alertcheck/table.go @@ -0,0 +1,136 @@ +package main + +import ( + "fmt" + "io" + "sort" + "text/tabwriter" + "time" + + "github.com/smartcontractkit/chainlink-testing-framework/grafana-alertcheck/internal/gate" +) + +// renderTable is §20.2's required human table. It always writes to the +// writer it is given, which the caller (runCheck) always points at +// stderr — the human table is not the machine output §20.2 reserves stdout +// for. +// +// Three sections, in order: +// +// 1. one line per rule: outcome, BadFor, pollEvery, proved-or-not with the +// largest gap; +// 2. one line per Violation (R2): a rule's worst-of outcome does not carry +// the State/Health of the instance that actually caused it — Violation +// does — so this is also where those two columns appear, sorted after +// the rule table rather than folded into it, and it is the only place an +// operator running WITHOUT --output json sees the §12.1 --allow-paused +// hint that Violation.Note already carries (classify.go); +// 3. a footer with the per-rule thresholds and the run-wide numbers §20.2 +// says are the answer to "why" on exit 2: each non-skipped rule's +// maxGap/healthGrace/evalStaleAfter, the global transitionGrace and +// drainTimeout, and the largest measured clock skew alongside its own +// error bound (RTT/2) — SkewHardLimit is a separate, fixed input +// threshold and is reported next to it, never as if it were that bound. +func renderTable(w io.Writer, res gate.Result) error { + alertOf := make(map[string]string, len(res.Verdicts)) + for _, v := range res.Verdicts { + alertOf[v.RuleUID] = v.Alert + } + + tw := tabwriter.NewWriter(w, 0, 4, 2, ' ', 0) + fmt.Fprintln(tw, "ALERT\tOUTCOME\tBADFOR\tPOLLEVERY\tPROVED\tNOTE") + for _, v := range sortedVerdicts(res.Verdicts) { + fmt.Fprintf(tw, "%s\t%s\t%s\t%s\t%s\t%s\n", + v.Alert, v.Outcome, v.BadFor.Round(time.Second), v.PollEvery.Round(time.Second), + provedLabel(res.Coverage[v.RuleUID]), v.Note) + } + if err := tw.Flush(); err != nil { + return fmt.Errorf("render table: %w", err) + } + + if len(res.Violations) > 0 { + fmt.Fprintln(w, "\nVIOLATIONS") + vtw := tabwriter.NewWriter(w, 0, 4, 2, ' ', 0) + fmt.Fprintln(vtw, "ALERT\tOUTCOME\tSTATE\tHEALTH\tNOTE") + for _, v := range sortedViolations(res.Violations) { + fmt.Fprintf(vtw, "%s\t%s\t%s\t%s\t%s\n", alertLabel(v, alertOf), v.Outcome, v.State, v.Health, v.Note) + } + if err := vtw.Flush(); err != nil { + return fmt.Errorf("render table: %w", err) + } + } + + fmt.Fprintln(w) + for _, uid := range sortedThresholdUIDs(res.Thresholds, alertOf) { + t := res.Thresholds[uid] + fmt.Fprintf(w, "rule %s: maxGap=%s healthGrace=%s evalStaleAfter=%s\n", + alertOr(uid, alertOf), t.MaxGap, t.HealthGrace, t.EvalStaleAfter) + } + fmt.Fprintf(w, "global: transitionGrace=%s (source: %s) drainTimeout=%s\n", + res.Global.TransitionGrace, res.Global.GraceSource, res.Global.DrainTimeout) + fmt.Fprintf(w, "violations: %d, largest measured clock skew: %s (bound ±%s, hard limit %s), grafana %s\n", + len(res.Violations), res.ClockSkew.Round(time.Millisecond), res.ClockSkewBound.Round(time.Millisecond), + gate.SkewHardLimit, res.GrafanaVersion) + return nil +} + +// provedLabel is the table's PROVED column: "yes" for a clean coverage +// proof, "no" with the reason and largest gap for an unobservable rule, and +// "-" for a rule decide never asked proveCoverage about at all (skipped — +// paused before the window opened, §12). +func provedLabel(cov gate.CoverageResult) string { + if cov.Reason == "" && !cov.Unobservable && !cov.Proved { + return "-" + } + if cov.Unobservable { + if cov.LargestGap > 0 { + return fmt.Sprintf("no (%s; largest gap %s at %s)", cov.Reason, + cov.LargestGap.Round(time.Second), cov.LargestGapAt.Format(time.RFC3339)) + } + return fmt.Sprintf("no (%s)", cov.Reason) + } + return "yes" +} + +// alertLabel resolves a Violation's alert name. Most violations already +// carry it directly; the synthetic MinObserved-shortfall entry with no named +// rule (classify.go) has an empty Alert and an empty RuleUID, so alertOf +// cannot resolve it either — "-" says plainly that this row is not about a +// specific rule. +func alertLabel(v gate.Violation, alertOf map[string]string) string { + if v.Alert != "" { + return v.Alert + } + if a, ok := alertOf[v.RuleUID]; ok { + return a + } + return "-" +} + +func alertOr(uid string, alertOf map[string]string) string { + if a, ok := alertOf[uid]; ok { + return a + } + return uid +} + +func sortedVerdicts(in []gate.RuleVerdict) []gate.RuleVerdict { + out := append([]gate.RuleVerdict(nil), in...) + sort.Slice(out, func(i, j int) bool { return out[i].Alert < out[j].Alert }) + return out +} + +func sortedViolations(in []gate.Violation) []gate.Violation { + out := append([]gate.Violation(nil), in...) + sort.SliceStable(out, func(i, j int) bool { return out[i].Alert < out[j].Alert }) + return out +} + +func sortedThresholdUIDs(thresholds map[string]gate.RuleThresholds, alertOf map[string]string) []string { + uids := make([]string, 0, len(thresholds)) + for uid := range thresholds { + uids = append(uids, uid) + } + sort.Slice(uids, func(i, j int) bool { return alertOr(uids[i], alertOf) < alertOr(uids[j], alertOf) }) + return uids +} diff --git a/grafana-alertcheck/cmd/grafana-alertcheck/table_test.go b/grafana-alertcheck/cmd/grafana-alertcheck/table_test.go new file mode 100644 index 000000000..106760fbb --- /dev/null +++ b/grafana-alertcheck/cmd/grafana-alertcheck/table_test.go @@ -0,0 +1,114 @@ +package main + +import ( + "bytes" + "strings" + "testing" + "time" + + "github.com/smartcontractkit/chainlink-testing-framework/grafana-alertcheck/internal/gate" +) + +// TestRenderTable is the golden table test: a fixed Result renders a +// deterministic, ordered rule table, a violations section (R2) and a footer +// carrying the per-rule and global thresholds plus the skew and its bound +// (R3) — with no live Check involved. +func TestRenderTable(t *testing.T) { + gapAt := time.Date(2026, 1, 1, 12, 0, 0, 0, time.UTC) + res := gate.Result{ + GrafanaVersion: "13.1.0", + ClockSkew: 1500 * time.Millisecond, + ClockSkewBound: 250 * time.Millisecond, + Verdicts: []gate.RuleVerdict{ + {Alert: "Zebra Alert", RuleUID: "uid-z", Outcome: gate.OutcomeClean, PollEvery: 30 * time.Second}, + {Alert: "Ape Alert", RuleUID: "uid-a", Outcome: gate.OutcomeUnobservable, + PollEvery: 30 * time.Second, Note: "gap of 5m0s starting at 2026-01-01T12:00:00Z exceeds maxGap 1m0s"}, + {Alert: "Paused Alert", RuleUID: "uid-p", Outcome: gate.OutcomeSkipped, + Note: "paused before the window opened; counts against --min-observed unless --allow-paused is set"}, + }, + Violations: []gate.Violation{ + {Alert: "Ape Alert", RuleUID: "uid-a", Outcome: gate.OutcomeUnobservable, State: gate.StateFiring, Health: "error", Note: "unobservable"}, + {Alert: "Paused Alert", RuleUID: "uid-p", Outcome: gate.OutcomeSkipped, + Note: "paused before the window opened; counts against --min-observed unless --allow-paused is set"}, + }, + Coverage: map[string]gate.CoverageResult{ + "uid-z": {Proved: true}, + "uid-a": {Unobservable: true, Reason: gate.ReasonHeartbeatGap, LargestGap: 5 * time.Minute, LargestGapAt: gapAt}, + }, + Thresholds: map[string]gate.RuleThresholds{ + "uid-z": {MaxGap: time.Minute, HealthGrace: time.Minute, EvalStaleAfter: time.Minute}, + "uid-a": {MaxGap: time.Minute, HealthGrace: 2 * time.Minute, EvalStaleAfter: time.Minute}, + }, + Global: gate.GlobalThresholds{ + TransitionGrace: 5 * time.Minute, + GraceSource: `Ape Alert (for=5m)`, + DrainTimeout: 2 * time.Minute, + }, + } + + var buf bytes.Buffer + if err := renderTable(&buf, res); err != nil { + t.Fatalf("renderTable: %v", err) + } + out := buf.String() + + // Rule table: Ape sorts before Zebra sorts before... Paused is skipped and + // carries no coverage entry, so it renders "-" for PROVED. + if !strings.Contains(out, "Ape Alert") || !strings.Contains(out, "unobservable") { + t.Fatalf("out = %q, want Ape's unobservable row", out) + } + if !strings.Contains(out, "heartbeat_gap") || !strings.Contains(out, "largest gap 5m0s") { + t.Fatalf("out = %q, want the coverage reason and largest gap", out) + } + if !strings.Contains(out, "Zebra Alert") || !strings.Contains(out, "clean") { + t.Fatalf("out = %q, want Zebra's clean row", out) + } + + // Violations section (R2): must show up even without --output json, and + // must carry the §12.1 --allow-paused hint text verbatim. + if !strings.Contains(out, "VIOLATIONS") { + t.Fatalf("out = %q, want a VIOLATIONS section", out) + } + if !strings.Contains(out, "--allow-paused") { + t.Fatalf("out = %q, want the §12.1 --allow-paused hint in the human table", out) + } + if !strings.Contains(out, "STATE") || !strings.Contains(out, "HEALTH") { + t.Fatalf("out = %q, want the violations table to have STATE and HEALTH columns", out) + } + if !strings.Contains(out, string(gate.StateFiring)) || !strings.Contains(out, "error") { + t.Fatalf("out = %q, want Ape's violation State/Health", out) + } + + // Footer (R3): per-rule thresholds, global thresholds, and skew with its + // own bound rather than the fixed hard limit. + if !strings.Contains(out, "Ape Alert: maxGap=1m0s healthGrace=2m0s evalStaleAfter=1m0s") { + t.Fatalf("out = %q, want Ape's per-rule thresholds", out) + } + if !strings.Contains(out, "Zebra Alert: maxGap=1m0s healthGrace=1m0s evalStaleAfter=1m0s") { + t.Fatalf("out = %q, want Zebra's per-rule thresholds", out) + } + if strings.Contains(out, "Paused Alert: maxGap") { + t.Fatalf("out = %q, a skipped rule must not report thresholds it never had (§12)", out) + } + if !strings.Contains(out, "global: transitionGrace=5m0s (source: Ape Alert (for=5m)) drainTimeout=2m0s") { + t.Fatalf("out = %q, want the global thresholds line", out) + } + if !strings.Contains(out, "largest measured clock skew: 1.5s (bound ±250ms, hard limit 1m0s)") { + t.Fatalf("out = %q, want the skew and its own bound, not the hard limit misused as one", out) + } + if !strings.Contains(out, "violations: 2") { + t.Fatalf("out = %q, want the violation count", out) + } + if !strings.Contains(out, "13.1.0") { + t.Fatalf("out = %q, want the grafana version", out) + } +} + +// TestProvedLabel_Skipped pins the "-" case: a rule decide never asked +// proveCoverage about (paused before the window opened, §12) has an empty +// CoverageResult and must not be reported as either proved or unobservable. +func TestProvedLabel_Skipped(t *testing.T) { + if got := provedLabel(gate.CoverageResult{}); got != "-" { + t.Fatalf("provedLabel(zero value) = %q, want \"-\"", got) + } +} diff --git a/grafana-alertcheck/cmd/grafana-alertcheck/watch.go b/grafana-alertcheck/cmd/grafana-alertcheck/watch.go new file mode 100644 index 000000000..f3ad57d92 --- /dev/null +++ b/grafana-alertcheck/cmd/grafana-alertcheck/watch.go @@ -0,0 +1,148 @@ +package main + +import ( + "context" + "flag" + "fmt" + "io" + "time" + + "github.com/smartcontractkit/chainlink-testing-framework/grafana-alertcheck/internal/gate" +) + +const watchUsage = "usage: grafana-alertcheck watch --out [--pidfile F] [--daemon-log F] " + + "--alerts [--folder F] [--poll-interval D] [--concurrency N] [--until RFC3339]" + +// runWatch is the record step's entire CLI surface, split in two by one flag +// set — gate.DaemonChildFlag ("--daemon-child") and gate.ReadyFDFlag +// ("--ready-fd") select which side of P6's parent/child split this +// invocation is: +// +// - without them: the record command an operator types. It parses --out, +// --alerts and the rest, builds a gate.WatchConfig and calls gate.Watch, +// which resolves, records the first observation of every rule, and +// detaches the recorder before returning (§4.3). +// - with them: the detached recorder itself. gate.Watch's own childArgs +// (watch_unix.go) is the only thing that ever sets them — an operator +// never types "--daemon-child" and it does not appear in watchUsage — +// and this dispatches straight to gate.RunDaemonChild. P6's integration +// test already covers the spawn; this is the one new test P10 owns: that +// seeing the flag reaches RunDaemonChild. +// +// Both flags live in the SAME flag set as the operator-facing ones rather +// than a second, hidden set: the child is started with childArgs' exact +// argv, e.g. "watch --daemon-child --out log.jsonl --ready-fd 3 +// [--until ...] [--concurrency ...]", and a second parser would have to stay +// byte-for-byte in sync with that slice to accept it. +func runWatch(args []string, stdin io.Reader, stdout, stderr io.Writer) int { + fs := flag.NewFlagSet("watch", flag.ContinueOnError) + fs.SetOutput(stderr) + fs.Usage = func() { fmt.Fprintln(stderr, watchUsage) } + + common := registerCommon(fs) + out := fs.String("out", "", "JSONL log path to record to") + pidfile := fs.String("pidfile", "", "pidfile path (default .pid)") + daemonLog := fs.String("daemon-log", "", "stdout/stderr sink for the detached recorder (default .daemon.log)") + until := fs.String("until", "", "optional hard stop, RFC3339 (default: run until check stops it)") + pollInterval := fs.String("poll-interval", "", "override every rule's poll cadence (default: half its own evaluation interval)") + + // Hidden: never in watchUsage, never typed by an operator (see doc comment). + daemonChild := fs.Bool(gate.DaemonChildFlag[2:], false, "") + readyFD := fs.Int(gate.ReadyFDFlag[2:], 0, "") + + if err := fs.Parse(args); err != nil { + return 2 + } + + if *daemonChild { + return runDaemonChild(*out, *until, *common.concurrency, *readyFD, stderr) + } + + url, token, err := grafanaEnv() + if err != nil { + fmt.Fprintln(stderr, err) + return 2 + } + + alerts, err := readAlerts(stdin, *common.alerts) + if err != nil { + fmt.Fprintln(stderr, err) + return 2 + } + + cfg := gate.WatchConfig{ + URL: url, + Token: token, + Alerts: alerts, + Folder: *common.folder, + Out: *out, + PidFile: *pidfile, + DaemonLog: *daemonLog, + Concurrency: *common.concurrency, + Clock: gate.SystemClock{}, + Notes: stderr, + } + if *until != "" { + t, err := time.Parse(time.RFC3339, *until) + if err != nil { + fmt.Fprintf(stderr, "--until: %v\n", err) + return 2 + } + cfg.Until = t + } + if *pollInterval != "" { + d, err := time.ParseDuration(*pollInterval) + if err != nil { + fmt.Fprintf(stderr, "--poll-interval: %v\n", err) + return 2 + } + cfg.PollEvery = d + } + + if err := gate.Watch(context.Background(), cfg); err != nil { + fmt.Fprintln(stderr, err) + return 2 + } + return 0 +} + +// runDaemonChild is the detached recorder's whole entry point (P6's +// obligation on this phase). Its stdout and stderr are already the daemon +// log file — spawnChild (watch_unix.go) redirects both before Start — so +// writing to stderr here lands exactly where waitForChildReady's failure +// path quotes from. +func runDaemonChild(out, until string, concurrency, readyFD int, stderr io.Writer) int { + url, token, err := grafanaEnv() + if err != nil { + fmt.Fprintln(stderr, err) + return 2 + } + cfg := gate.DaemonChildConfig{ + URL: url, + Token: token, + Out: out, + Concurrency: concurrency, + Clock: gate.SystemClock{}, + ReadyFD: readyFD, + } + if until != "" { + t, err := time.Parse(time.RFC3339, until) + if err != nil { + fmt.Fprintf(stderr, "--until: %v\n", err) + return 2 + } + cfg.Until = t + } + if err := gate.RunDaemonChild(context.Background(), cfg); err != nil { + fmt.Fprintln(stderr, err) + return 2 + } + return 0 +} + +// The daemon-child and ready-fd flags registered above must keep matching +// watch_unix.go's childArgs, which names exactly --daemon-child, --out, +// --ready-fd, --until and --concurrency and nothing else: that function +// builds this process's own argv when it re-execs itself as the detached +// recorder, so a flag added to one side without the other means the child +// fails on its very first flag.Parse. diff --git a/grafana-alertcheck/cmd/grafana-alertcheck/watch_test.go b/grafana-alertcheck/cmd/grafana-alertcheck/watch_test.go new file mode 100644 index 000000000..267a312a2 --- /dev/null +++ b/grafana-alertcheck/cmd/grafana-alertcheck/watch_test.go @@ -0,0 +1,99 @@ +package main + +import ( + "bytes" + "os" + "strings" + "testing" +) + +// TestRunWatch_FlagValidation is the record step's flag-validation matrix. +// Every case fails inside gate.WatchConfig.validate() (P6) or before it, so +// none needs a reachable Grafana. +func TestRunWatch_FlagValidation(t *testing.T) { + tests := []struct { + name string + env bool + args func(t *testing.T) []string + wantErr string + }{ + {"missing env", false, func(t *testing.T) []string { + return []string{"--out", t.TempDir() + "/log.jsonl", "--alerts", writeTempAlerts(t)} + }, "GRAFANA_URL"}, + {"missing out", true, func(t *testing.T) []string { + return []string{"--alerts", writeTempAlerts(t)} + }, "no log path"}, + {"missing alerts", true, func(t *testing.T) []string { + return []string{"--out", t.TempDir() + "/log.jsonl"} + }, "no alert names"}, + {"bad until format", true, func(t *testing.T) []string { + return []string{"--out", t.TempDir() + "/log.jsonl", "--alerts", writeTempAlerts(t), "--until", "not-a-time"} + }, "--until"}, + {"until in the past", true, func(t *testing.T) []string { + return []string{"--out", t.TempDir() + "/log.jsonl", "--alerts", writeTempAlerts(t), "--until", "2000-01-01T00:00:00Z"} + }, "not in the future"}, + {"bad poll-interval", true, func(t *testing.T) []string { + return []string{"--out", t.TempDir() + "/log.jsonl", "--alerts", writeTempAlerts(t), "--poll-interval", "not-a-duration"} + }, "--poll-interval"}, + } + for _, tt := range tests { + t.Run(tt.name, func(t *testing.T) { + if tt.env { + t.Setenv("GRAFANA_URL", "http://example.invalid") + t.Setenv("GRAFANA_TOKEN", "test-token") + } else { + t.Setenv("GRAFANA_URL", "") + t.Setenv("GRAFANA_TOKEN", "") + } + var stdout, stderr bytes.Buffer + args := append([]string{"watch"}, tt.args(t)...) + code := run(args, &stdout, &stderr) + if code != 2 { + t.Fatalf("code = %d, want 2; stderr = %q", code, stderr.String()) + } + if !strings.Contains(stderr.String(), tt.wantErr) { + t.Fatalf("stderr = %q, want it to contain %q", stderr.String(), tt.wantErr) + } + }) + } +} + +// TestRunWatch_DaemonChildDispatch pins P6's obligation on this phase: seeing +// gate.DaemonChildFlag must dispatch to gate.RunDaemonChild, and the flag +// must never appear in watchUsage (an operator never types it). +func TestRunWatch_DaemonChildDispatch(t *testing.T) { + t.Setenv("GRAFANA_URL", "http://example.invalid") + t.Setenv("GRAFANA_TOKEN", "test-token") + + var stdout, stderr bytes.Buffer + // No log at this path: RunDaemonChild fails trying to read it, which is + // enough to prove dispatch happened without needing a real recording. + missing := os.DevNull + ".missing" + code := run([]string{"watch", "--daemon-child", "--out", missing, "--ready-fd", "0"}, &stdout, &stderr) + if code != 2 { + t.Fatalf("code = %d, want 2; stderr = %q", code, stderr.String()) + } + if !strings.Contains(stderr.String(), missing) { + t.Fatalf("stderr = %q, want RunDaemonChild's read failure naming %q", stderr.String(), missing) + } + if strings.Contains(watchUsage, "daemon-child") { + t.Fatalf("watchUsage = %q, must never name --daemon-child", watchUsage) + } + if strings.Contains(watchUsage, "ready-fd") { + t.Fatalf("watchUsage = %q, must never name --ready-fd", watchUsage) + } +} + +func TestRunWatch_DaemonChild_MissingEnv(t *testing.T) { + t.Setenv("GRAFANA_URL", "") + t.Setenv("GRAFANA_TOKEN", "") + + var stdout, stderr bytes.Buffer + code := run([]string{"watch", "--daemon-child", "--out", "log.jsonl"}, &stdout, &stderr) + if code != 2 { + t.Fatalf("code = %d, want 2; stderr = %q", code, stderr.String()) + } + if !strings.Contains(stderr.String(), "GRAFANA_URL") { + t.Fatalf("stderr = %q, want it to name the missing env var", stderr.String()) + } +} diff --git a/grafana-alertcheck/internal/gate/check.go b/grafana-alertcheck/internal/gate/check.go index 6c3bf8805..710bedef9 100644 --- a/grafana-alertcheck/internal/gate/check.go +++ b/grafana-alertcheck/internal/gate/check.go @@ -21,8 +21,9 @@ import ( // not return a code, because a code is a presentation decision and the // library must not make it. // - §12.1 wants the paused rule AND --allow-paused both named to the user. -// decide names the rule in the shortfall Violation's Note (classify.go); -// the flag hint belongs to the CLI's renderer. +// decide names both in the shortfall Violation's Note (classify.go), and +// P10's renderer prints every Violation, including Note, in the human +// table — not only in --output json. // - Config.Notes carries the running commentary (§13.2's planned run time, // the countdown, the blind-interval warning). §20.2 puts the human output // on stderr and reserves stdout for --output json, so the CLI must pass diff --git a/grafana-alertcheck/internal/gate/classify.go b/grafana-alertcheck/internal/gate/classify.go index 5bfa237a4..c4bda0a54 100644 --- a/grafana-alertcheck/internal/gate/classify.go +++ b/grafana-alertcheck/internal/gate/classify.go @@ -104,6 +104,28 @@ type Policy struct { From, To time.Time } +// RuleThresholds is one non-skipped rule's resolved coverage thresholds +// (§5/§10.1/§14.1), carried on Result so the CLI's table (P10, §20.2) can +// print the numbers that answer "why" on exit 2 without decide exposing the +// unexported ruleTimings type itself. +type RuleThresholds struct { + MaxGap time.Duration + HealthGrace time.Duration + EvalStaleAfter time.Duration +} + +// GlobalThresholds is the run-wide half of the same information (§13.1, +// §19): transitionGrace and drainTimeout apply once, across every +// non-skipped watched rule, not per rule (globalTimings). +type GlobalThresholds struct { + TransitionGrace time.Duration + // GraceSource names, and already carries the `for` value of, the rule + // that set TransitionGrace (§13.2 requires printing both). "none" when no + // rule contributed (TransitionGrace is then 0). + GraceSource string + DrainTimeout time.Duration +} + // Result is decide's whole answer: everything §20.2's table and the action's // JSON outputs need. Coverage carries one CoverageResult per non-skipped // rule — no separate Interval type anywhere in the project (§2's @@ -112,9 +134,22 @@ type Result struct { From, To time.Time GrafanaVersion string ClockSkew time.Duration // the largest |skew| across every poll decide was given, not only the ones a rule's window actually used + // ClockSkewBound is the skew BOUND (RTT/2, §16) of that SAME poll — not + // the largest bound seen overall, which would pair a wide bound from an + // unrelated slow request with the worst skew and misstate how tightly + // that skew is actually known. SkewHardLimit is a separate, fixed input + // validation threshold (source.go) and is not an error bound on this + // value; the CLI prints both, but must not conflate them. + ClockSkewBound time.Duration Coverage map[string]CoverageResult - Verdicts []RuleVerdict - Violations []Violation + // Thresholds carries one RuleThresholds per rule Coverage also covers — + // every non-skipped rule, keyed by UID. A skipped rule has neither: it + // was never scheduled, so it has no maxGap/healthGrace/evalStaleAfter to + // report (§12). + Thresholds map[string]RuleThresholds + Global GlobalThresholds + Verdicts []RuleVerdict + Violations []Violation } // episode is one contiguous, policy-bad span of one instance's timeline, @@ -207,7 +242,7 @@ func classifyRule(def Definition, polls []Poll, from, windowEnd time.Time, badSt } // Different polls can carry different measured skews. In theory a // closing poll's translated time could land before the opening - // poll's — skew is capped at skewHardLimit (60s), so this is remote, + // poll's — skew is capped at SkewHardLimit (60s), so this is remote, // not impossible — and a negative span would feed mergeDurations a // duration that subtracts instead of adds. Clamp rather than trust // the arithmetic never to invert. @@ -487,14 +522,29 @@ func decide(h Header, polls []Poll, sentinel *time.Time, defs []Definition, To: pol.To, GrafanaVersion: h.GrafanaVersion, Coverage: make(map[string]CoverageResult), + Thresholds: make(map[string]RuleThresholds), + Global: GlobalThresholds{ + TransitionGrace: gt.transitionGrace, + GraceSource: gt.graceSource, + DrainTimeout: gt.drainTimeout, + }, } + skewSeen := false for _, p := range polls { s := p.Skew() if s < 0 { s = -s } - if s > result.ClockSkew { + // The bound travels with ITS OWN poll's skew, never the largest bound + // seen overall (Result.ClockSkewBound's doc comment) — so it is only + // ever overwritten in lockstep with ClockSkew, on the same poll. >= + // rather than > on top of skewSeen: a strict > would never assign the + // bound at all when every poll's skew is exactly 0, understating the + // real measurement uncertainty as an unearned "bound ±0s". + if !skewSeen || s > result.ClockSkew { result.ClockSkew = s + result.ClockSkewBound = p.SkewBound() + skewSeen = true } } @@ -550,6 +600,11 @@ func decide(h Header, polls []Poll, sentinel *time.Time, defs []Definition, } } result.Coverage[def.UID] = cov + result.Thresholds[def.UID] = RuleThresholds{ + MaxGap: t.maxGap, + HealthGrace: t.healthGrace, + EvalStaleAfter: t.evalStaleAfter, + } outcome, badFor, viols := classifyRule(def, polls, pol.From, windowEnd, badStates, pol.Preexisting) if cov.Unobservable { @@ -586,9 +641,9 @@ func decide(h Header, polls []Poll, sentinel *time.Time, defs []Definition, break } // §12.1 requires the paused rule and --allow-paused both be - // named to the user; naming the rule is this Violation's job, - // the --allow-paused hint is the CLI table/renderer's (P10) — - // tracked here so it is not dropped when that phase is built. + // named to the user; both live in this one Violation, in Note — + // P10's renderer prints Note verbatim rather than re-deriving + // the hint, so the exact wording here is what an operator reads. result.Violations = append(result.Violations, Violation{ Alert: def.Title, RuleUID: def.UID, Outcome: OutcomeSkipped, Note: "paused before the window opened; counts against --min-observed unless --allow-paused is set", diff --git a/grafana-alertcheck/internal/gate/schedule.go b/grafana-alertcheck/internal/gate/schedule.go index fda33cacc..0417af6a9 100644 --- a/grafana-alertcheck/internal/gate/schedule.go +++ b/grafana-alertcheck/internal/gate/schedule.go @@ -8,16 +8,18 @@ import ( "time" ) -// skewHardLimit is one of §5's filled-in values (basis: §16; §22.11 asserts +// SkewHardLimit is one of §5's filled-in values (basis: §16; §22.11 asserts // 120s errors, 30s does not). Defined here, in schedule.go's named-constants // block, per §5's instruction — it moved out of source.go now that P4 exists; -// P2 needed it before this file did, so it started there. -const skewHardLimit = 60 * time.Second +// P2 needed it before this file did, so it started there. Exported (P10) so +// the CLI can report it verbatim next to a measured skew instead of keeping +// its own mirrored copy. +const SkewHardLimit = 60 * time.Second // fromFutureTolerance is how far ahead of the runner's own clock a supplied // `from` may sit before check refuses it (§7: "from in the future, more than // the skew tolerance — error"). §7 names no number, so this is the judgment -// call §5's table records: the same 60s as skewHardLimit, because the only +// call §5's table records: the same 60s as SkewHardLimit, because the only // legitimate reason for a `from` in the future is clock disagreement between // the deploy step and the check step, and that is bounded by the same figure. // It is once-per-run input validation, not a per-rule coverage check, so diff --git a/grafana-alertcheck/internal/gate/source.go b/grafana-alertcheck/internal/gate/source.go index 6e587b580..d440d4938 100644 --- a/grafana-alertcheck/internal/gate/source.go +++ b/grafana-alertcheck/internal/gate/source.go @@ -271,7 +271,7 @@ type requestResult struct { // doRequest performs one HTTP GET and classifies the outcome (§14.5, §16): // a network failure, a non-2xx status, or a body-read failure is retryable // (*TransportError); a missing or unparseable Date header, or a skew beyond -// skewHardLimit, is a hard error — retrying can never fix either, so neither +// SkewHardLimit, is a hard error — retrying can never fix either, so neither // may enter the backoff loop (H4). // // The Date-header/skew check runs for every endpoint this hits, including @@ -331,8 +331,8 @@ func (s *httpSource) doRequest(ctx context.Context, path string) (requestResult, if absSkew < 0 { absSkew = -absSkew } - if absSkew > skewHardLimit { - return requestResult{}, fmt.Errorf("%s: clock skew %s exceeds hard limit %s (§16)", path, absSkew, skewHardLimit) + if absSkew > SkewHardLimit { + return requestResult{}, fmt.Errorf("%s: clock skew %s exceeds hard limit %s (§16)", path, absSkew, SkewHardLimit) } return requestResult{Body: b, ServerDate: serverDate, Skew: signedSkew, SkewBound: bound, Latency: latency}, nil From a6afb463f4c5208e936b1ec084433169763a4f14 Mon Sep 17 00:00:00 2001 From: Bartek Tofel Date: Mon, 7 Sep 2026 11:45:23 +0200 Subject: [PATCH 28/43] chore: fix goreleaser.yaml and add version command --- grafana-alertcheck/.goreleaser.yaml | 8 ++--- .../cmd/grafana-alertcheck/main.go | 4 ++- .../cmd/grafana-alertcheck/version.go | 29 +++++++++++++++++ .../cmd/grafana-alertcheck/version_test.go | 32 +++++++++++++++++++ 4 files changed, 68 insertions(+), 5 deletions(-) create mode 100644 grafana-alertcheck/cmd/grafana-alertcheck/version.go create mode 100644 grafana-alertcheck/cmd/grafana-alertcheck/version_test.go diff --git a/grafana-alertcheck/.goreleaser.yaml b/grafana-alertcheck/.goreleaser.yaml index 09e6032cb..0890d9f82 100644 --- a/grafana-alertcheck/.goreleaser.yaml +++ b/grafana-alertcheck/.goreleaser.yaml @@ -14,10 +14,10 @@ builds: ldflags: - -s - -w - - -X github.com/smartcontractkit/chainlink-testing-framework/grafana-alertcheck.version={{.Version}} - - -X github.com/smartcontractkit/chainlink-testing-framework/grafana-alertcheck.commit={{.ShortCommit}} - - -X github.com/smartcontractkit/chainlink-testing-framework/grafana-alertcheck.date={{.CommitDate}} - - -X github.com/smartcontractkit/chainlink-testing-framework/grafana-alertcheck.builtBy=goreleaser + - -X github.com/smartcontractkit/chainlink-testing-framework/grafana-alertcheck/cmd/grafana-alertcheck.version={{.Version}} + - -X github.com/smartcontractkit/chainlink-testing-framework/grafana-alertcheck/cmd/grafana-alertcheck.commit={{.ShortCommit}} + - -X github.com/smartcontractkit/chainlink-testing-framework/grafana-alertcheck/cmd/grafana-alertcheck.date={{.CommitDate}} + - -X github.com/smartcontractkit/chainlink-testing-framework/grafana-alertcheck/cmd/grafana-alertcheck.builtBy=goreleaser goos: - linux - darwin diff --git a/grafana-alertcheck/cmd/grafana-alertcheck/main.go b/grafana-alertcheck/cmd/grafana-alertcheck/main.go index 73d30f3f3..db5f93197 100644 --- a/grafana-alertcheck/cmd/grafana-alertcheck/main.go +++ b/grafana-alertcheck/cmd/grafana-alertcheck/main.go @@ -12,7 +12,7 @@ func main() { os.Exit(run(os.Args[1:], os.Stdout, os.Stderr)) } -const usage = "usage: grafana-alertcheck " +const usage = "usage: grafana-alertcheck " // run is the whole of main's testable surface: parse the subcommand, dispatch, // return the process exit code. Exit codes below 2 (pass/violations) belong to @@ -37,6 +37,8 @@ func run(args []string, stdout, stderr io.Writer) int { return runWatch(args[1:], os.Stdin, stdout, stderr) case "check": return runCheck(args[1:], os.Stdin, stdout, stderr) + case "version": + return runVersion(args[1:], stdout, stderr) case "-h", "-help", "--help": fmt.Fprintln(stdout, usage) return 0 diff --git a/grafana-alertcheck/cmd/grafana-alertcheck/version.go b/grafana-alertcheck/cmd/grafana-alertcheck/version.go new file mode 100644 index 000000000..5cf68b6af --- /dev/null +++ b/grafana-alertcheck/cmd/grafana-alertcheck/version.go @@ -0,0 +1,29 @@ +package main + +import ( + "fmt" + "io" +) + +// Build metadata. These are package-level variables so goreleaser's ldflags +// (-X) can stamp them at build time; left unstamped they fall back to the +// "dev" defaults below, which is what a plain `go build` produces. +var ( + version = "dev" + commit = "unknown" + date = "unknown" + builtBy = "unknown" +) + +// runVersion prints the build metadata to stdout. Unlike list/watch/check it +// needs no Grafana connection, so it never touches the environment or the +// network; it exists purely so operators can answer "what am I running?" +// against a deployed binary. +func runVersion(args []string, stdout, stderr io.Writer) int { + if len(args) != 0 { + fmt.Fprintf(stderr, "version takes no arguments, got %v\n", args) + return 2 + } + fmt.Fprintf(stdout, "version: %s\ncommit: %s\ndate: %s\nbuiltBy: %s\n", version, commit, date, builtBy) + return 0 +} diff --git a/grafana-alertcheck/cmd/grafana-alertcheck/version_test.go b/grafana-alertcheck/cmd/grafana-alertcheck/version_test.go new file mode 100644 index 000000000..4122c2f13 --- /dev/null +++ b/grafana-alertcheck/cmd/grafana-alertcheck/version_test.go @@ -0,0 +1,32 @@ +package main + +import ( + "bytes" + "strings" + "testing" +) + +func TestRunVersion(t *testing.T) { + var stdout, stderr bytes.Buffer + code := run([]string{"version"}, &stdout, &stderr) + if code != 0 { + t.Fatalf("code = %d, want 0", code) + } + out := stdout.String() + for _, want := range []string{"version:", "commit:", "date:", "builtBy:"} { + if !strings.Contains(out, want) { + t.Errorf("stdout = %q, want it to contain %q", out, want) + } + } + if stderr.String() != "" { + t.Errorf("stderr = %q, want empty", stderr.String()) + } +} + +func TestRunVersion_RejectsArgs(t *testing.T) { + var stdout, stderr bytes.Buffer + code := run([]string{"version", "extra"}, &stdout, &stderr) + if code != 2 { + t.Fatalf("code = %d, want 2", code) + } +} From 95e0c34fc1e0fe77d7859861a5dcd5f123504fb7 Mon Sep 17 00:00:00 2001 From: Bartek Tofel Date: Mon, 31 Aug 2026 20:15:47 +0200 Subject: [PATCH 29/43] chore: implement phase 11 MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Add coverage.go's declared-KeepLast check (no_data_state/exec_err_state, not just an observed reason) and close the remaining §22 gaps: newly_bad's no-early-exit clock assertion, a recorder-mode gap right after the deploy, a rule's own coverage gap overriding its own recovery, a genuinely skew-discriminating staleness test, exit-2 consequences on two Reason-only coverage tests, and end-to-end checks for log-name collapse, a truncated log, and the real watch-written state histogram. --- .../internal/gate/check_test.go | 264 ++++++++++++++++ .../internal/gate/classify_test.go | 285 ++++++++++++++++++ grafana-alertcheck/internal/gate/coverage.go | 19 +- .../internal/gate/coverage_test.go | 68 ++++- .../internal/gate/parse_state_test.go | 53 ++++ .../internal/gate/resolve_test.go | 20 ++ .../internal/gate/schedule_test.go | 51 ++++ .../internal/gate/source_test.go | 53 ++++ .../internal/gate/watch_test.go | 7 + 9 files changed, 815 insertions(+), 5 deletions(-) diff --git a/grafana-alertcheck/internal/gate/check_test.go b/grafana-alertcheck/internal/gate/check_test.go index 2971c30e4..8cd94468f 100644 --- a/grafana-alertcheck/internal/gate/check_test.go +++ b/grafana-alertcheck/internal/gate/check_test.go @@ -3,6 +3,7 @@ package gate import ( "bufio" "context" + "encoding/json" "errors" "fmt" "os" @@ -300,6 +301,83 @@ func TestCheckSingleStepCleanWindowPasses(t *testing.T) { } } +// §22.2: the collapse-note-plus-satisfied-MinObserved path (resolve_test.go's +// TestResolve_CollapseByUIDGivesNoteNotError and +// TestResolve_MinObservedCountIsPostCollapse) is proven only at Resolve() +// directly; this drives the same shape through check() end to end — the two +// input names must collapse to one verdict, the run must pass, and the +// collapse note must reach the run's own notes, not just Resolve()'s return +// value. +func TestCheckSingleStepDuplicateAlertNamesCollapseWithNote(t *testing.T) { + clock := newVirtualClock(testNow) + cfg := baseConfig(t, clock) + cfg.Alerts = []string{"uid:" + checkUID, checkTitle} // the same rule, named two different ways + src := newCheckSource(func(_ string, _ int) (Observation, error) { + return healthyObservation(clock.Now()), nil + }) + + res, err := check(context.Background(), cfg, src) + if err != nil { + t.Fatalf("check() = %v, want nil\nnotes:\n%s", err, notesOf(cfg)) + } + if len(res.Verdicts) != 1 { + t.Fatalf("Verdicts = %+v, want exactly one — the duplicate must collapse to a single rule", res.Verdicts) + } + if len(res.Violations) != 0 { + t.Fatalf("Violations = %+v, want none: MinObserved must be satisfied by the post-collapse count of 1", res.Violations) + } + if notes := notesOf(cfg); !strings.Contains(notes, "counted once") { + t.Errorf("want the collapse note in the run's own notes; got:\n%s", notes) + } +} + +// §22.1's highest-priority regression: a rule with health=error for the +// whole window is unobservable, exit 2 — using the real "[JD] No Job +// Proposals" capture (testdata/README.md), not a synthetic Poll table, so a +// change in how the real payload shapes health/lastError cannot slip past a +// hand-built fixture that happens to still look right. +func TestCheckSingleStepContinuousHealthErrorIsUnobservable(t *testing.T) { + body := readFixture(t, "state_health_error.json") + rules, err := ParseState(body) + if err != nil { + t.Fatalf("ParseState: %v", err) + } + base := rules[0] + def := Definition{ + UID: base.UID, Title: base.Title, Folder: base.Folder, Group: base.Group, + IntervalSeconds: int(base.Interval / time.Second), NoDataState: "OK", ExecErrState: "OK", + Kind: KindGrafanaManaged, + } + + clock := newVirtualClock(testNow) + cfg := Config{ + URL: "https://grafana.example.com", Alerts: []string{"uid:" + def.UID}, + From: testNow, To: testNow.Add(5 * time.Minute), Clock: clock, Notes: &strings.Builder{}, + }.withDefaults() + + src := newCheckSource(func(_ string, _ int) (Observation, error) { + // Every field but LastEvaluation stays exactly as the real capture + // shaped it (health=error, the real lastError text, the real Error + // instance); LastEvaluation tracks the poll so staleness (a + // different coverage check, §14) never becomes the actual cause. + r := base + r.LastEvaluation = clock.Now() + return Observation{Rules: []StateRule{r}, GrafanaNow: clock.Now(), Latency: 200 * time.Millisecond}, nil + }) + src.defs = []Definition{def} + + res, err := check(context.Background(), cfg, src) + if err == nil { + t.Fatalf("check() = nil, want an error: continuous health=error must be unobservable (§22.1, H6/H7)\nnotes:\n%s", notesOf(cfg)) + } + if len(res.Verdicts) != 1 || res.Verdicts[0].Outcome != OutcomeUnobservable { + t.Fatalf("Verdicts = %+v, want one unobservable verdict", res.Verdicts) + } + if cov := res.Coverage[def.UID]; cov.Reason != ReasonHealthError { + t.Fatalf("Coverage[%s].Reason = %q, want %q", def.UID, cov.Reason, ReasonHealthError) + } +} + // H5: a certain violation does not release the runner early, and it does not // stop the gate reporting exit-1 shape — violations with a nil error. func TestCheckSingleStepFiringInstanceReportsWithoutExitingEarly(t *testing.T) { @@ -329,6 +407,65 @@ func TestCheckSingleStepFiringInstanceReportsWithoutExitingEarly(t *testing.T) { } } +// §22.8: "newly_bad at from+30s gives exit 1, but ONLY after +// to+transition_grace." The test above pins H5 for a rule already bad +// before the window opened (persistently_bad); this pins the anti-fail-fast +// case the plan names explicitly — a fresh onset just inside the window +// must not release the runner the instant it is first observed. +func TestCheckSingleStepNewOnsetDoesNotExitEarly(t *testing.T) { + clock := newVirtualClock(testNow) + cfg := baseConfig(t, clock) + onset := testNow.Add(30 * time.Second) + src := newCheckSource(func(_ string, _ int) (Observation, error) { + now := clock.Now() + if now.Before(onset) { + return healthyObservation(now), nil + } + firing := Instance{ + Labels: map[string]string{"alertname": checkTitle, "instance": "a"}, + State: StateFiring, + ActiveAt: onset, + } + return healthyObservation(now, firing), nil + }) + + res, err := check(context.Background(), cfg, src) + if err != nil { + t.Fatalf("check() = %v, want nil (a violation is exit 1, not an error)", err) + } + if len(res.Violations) != 1 || res.Violations[0].Outcome != OutcomeNewlyBad { + t.Fatalf("Violations = %+v, want exactly one newly_bad", res.Violations) + } + if windowEnd := cfg.To.Add(checkGrace); clock.Now().Before(windowEnd) { + t.Errorf("exited early at %s; H5 requires collecting to %s even for a fresh onset at from+30s", clock.Now(), windowEnd) + } +} + +// §22.9: an ABSENT `from` in single-step mode (as opposed to recorder mode, +// which hard-errors — TestCheckValidateRejectsBadConfigurations's "log mode +// without from") falls back to the start of this check step, with the same +// declared-blind-interval warning as an explicit early `from`. +func TestCheckSingleStepAbsentFromFallsBackToStepStart(t *testing.T) { + clock := newVirtualClock(testNow) + cfg := baseConfig(t, clock) + cfg.From = time.Time{} + src := newCheckSource(func(_ string, _ int) (Observation, error) { + return healthyObservation(clock.Now()), nil + }) + + res, err := check(context.Background(), cfg, src) + if err != nil { + t.Fatalf("check() = %v, want nil: an absent `from` in single-step mode is a fallback, not an error\nnotes:\n%s", err, notesOf(cfg)) + } + notes := notesOf(cfg) + if !strings.Contains(notes, "no `from` given") { + t.Errorf("want the §4.2 fallback note; notes were:\n%s", notes) + } + if !res.From.Equal(testNow) { + t.Errorf("Result.From = %s, want the step-start fallback %s", res.From, testNow) + } +} + // §4.2/§22.4: in single-step mode an explicit `from` earlier than the first // observation is a DECLARED blind interval — a warning and a pass, naming the // exact interval it cannot see. Recorder mode keeps P7 check 2 strict. @@ -645,6 +782,51 @@ func TestCheckFailClosedOnCoverageGap(t *testing.T) { } } +// §22.4: "an episode fully between the deploy and the start of the check" — +// recorder mode must find this at the LEADING edge of the window too, right +// after `from` (the deploy's completion), not only in the middle +// (TestCheckFailClosedOnCoverageGap above). No poll exists for +// [from, from+3m): whatever happened there is invisible to every per-poll +// check, so only the coverage gap itself can catch it — the reason this +// two-phase recorder model exists at all (§4.2). +func TestCheckRecorderModeFindsAGapImmediatelyAfterTheDeploy(t *testing.T) { + dir := t.TempDir() + windowEnd := testNow.Add(5*time.Minute + checkGrace) + path := filepath.Join(dir, "log.jsonl") + clock := newFakeClock(windowEnd.Add(30 * time.Second)) + w, err := NewWriter(path, clock) + if err != nil { + t.Fatalf("NewWriter: %v", err) + } + if err := w.WriteHeader(Header{ + URL: "https://grafana.example.com", GrafanaVersion: "13.1.0", StartedAt: testNow.Add(-time.Minute), + Rules: []LoggedRule{{UID: checkUID, Title: checkTitle, IntervalSeconds: 60, PollEverySeconds: checkPollEvery.Seconds()}}, + }); err != nil { + t.Fatalf("WriteHeader: %v", err) + } + gapEnd := testNow.Add(3 * time.Minute) // nothing recorded from `from` (testNow) to here + for at := gapEnd; !at.After(windowEnd.Add(30 * time.Second)); at = at.Add(checkPollEvery) { + if err := w.WritePoll(Poll{ + RuleUID: checkUID, GrafanaNow: at, Found: true, State: "inactive", Health: "ok", LastEvaluation: at, + }); err != nil { + t.Fatalf("WritePoll: %v", err) + } + } + if err := w.Stop(); err != nil { + t.Fatalf("Stop: %v", err) + } + writePid(t, path+".pid", fmt.Sprintf("%d\n", deadPid(t))) + + cfg := recorderConfig(t, newVirtualClock(testNow), path) + res, err := check(context.Background(), cfg, newCheckSource(nil)) + if err == nil { + t.Fatalf("check() = nil, want exit 2: a hole right after the deploy hides whatever happened there just as much as one in the middle") + } + if len(res.Verdicts) != 1 || res.Verdicts[0].Outcome != OutcomeUnobservable { + t.Fatalf("Verdicts = %+v, want one unobservable verdict, never clean", res.Verdicts) + } +} + // §19.3 case 5: the drain limit passed. The recording itself is clean, so this // isolates the drain wait — the rule simply never evaluates through the end of // the window, and a rule that cannot answer that question is unobservable. @@ -1027,6 +1209,88 @@ func TestCheckDoesNotSignalABystanderHoldingAReusedPid(t *testing.T) { } } +// §22.5: a dead pidfile (the recorder process has already exited, holding no +// flock) with NO sentinel in the log — the shape a killed `watch` leaves +// behind — must not hang the stop wait: the flock is free immediately, so +// check reads the log at once, finds no sentinel, and fails closed. +func TestCheckDeadPidWithNoSentinelIsUnobservable(t *testing.T) { + dir := t.TempDir() + windowEnd := testNow.Add(5*time.Minute + checkGrace) + logPath := filepath.Join(dir, "log.jsonl") + w, err := NewWriter(logPath, newFakeClock(testNow)) + if err != nil { + t.Fatalf("NewWriter: %v", err) + } + if err := w.WriteHeader(Header{ + URL: "https://grafana.example.com", GrafanaVersion: "13.1.0", StartedAt: testNow.Add(-time.Minute), + Rules: []LoggedRule{{ + UID: checkUID, Title: checkTitle, Folder: "F", Group: "G", + IntervalSeconds: 60, NoDataState: "OK", ExecErrState: "OK", + PollEverySeconds: checkPollEvery.Seconds(), + }}, + }); err != nil { + t.Fatalf("WriteHeader: %v", err) + } + // Healthy heartbeats all the way past windowEnd — evaluatedThrough is + // satisfied, so the drain wait needs no live re-poll — but no sentinel is + // ever written: the recorder died before it could call Stop. + for at := testNow.Add(-time.Minute); !at.After(windowEnd.Add(30 * time.Second)); at = at.Add(checkPollEvery) { + if err := w.WritePoll(Poll{RuleUID: checkUID, GrafanaNow: at, Found: true, State: "inactive", Health: "ok", LastEvaluation: at}); err != nil { + t.Fatalf("WritePoll: %v", err) + } + } + if err := w.Close(); err != nil { // no sentinel — a clean exit would call Stop + t.Fatalf("Close: %v", err) + } + writePid(t, logPath+".pid", fmt.Sprintf("%d\n", deadPid(t))) + + cfg := recorderConfig(t, newVirtualClock(testNow), logPath) + res, err := check(context.Background(), cfg, newCheckSource(nil)) + if err == nil { + t.Fatalf("check() = nil, want an error: no sentinel means the recorder never proved it ran to the end") + } + if len(res.Verdicts) != 1 || res.Verdicts[0].Outcome != OutcomeUnobservable { + t.Fatalf("Verdicts = %+v, want one unobservable verdict", res.Verdicts) + } +} + +// §22.5: "an incomplete last line gives exit 2" is otherwise proven only +// indirectly — log_test.go's TestReadLogRejectsBadLogs pins ReadLog's own +// error, and TestExitCode pins that any non-nil error maps to exit 2 — but +// nothing feeds a genuinely truncated log through check() itself. This closes +// that seam: a raw file with a valid header and poll, then a torn JSON tail, +// exactly what a recorder killed mid-write leaves behind. +func TestCheckRecorderModeTruncatedLogFailsClosed(t *testing.T) { + dir := t.TempDir() + path := filepath.Join(dir, "log.jsonl") + + h := Header{ + SchemaVersion: LogSchemaVersion, + URL: "https://grafana.example.com", GrafanaVersion: "13.1.0", StartedAt: testNow.Add(-time.Minute), + Rules: []LoggedRule{{UID: checkUID, Title: checkTitle, IntervalSeconds: 60, PollEverySeconds: checkPollEvery.Seconds()}}, + } + hb, err := json.Marshal(headerRecord{Type: RecordHeader, Header: h}) + if err != nil { + t.Fatalf("marshal header: %v", err) + } + pb, err := json.Marshal(pollRecord{Type: RecordPoll, Poll: Poll{ + RuleUID: checkUID, GrafanaNow: testNow, Found: true, State: "inactive", Health: "ok", LastEvaluation: testNow, + }}) + if err != nil { + t.Fatalf("marshal poll: %v", err) + } + content := string(hb) + "\n" + string(pb) + "\n" + `{"type":"poll","rule_ui` // torn mid-write + if err := os.WriteFile(path, []byte(content), 0o644); err != nil { + t.Fatalf("write log: %v", err) + } + writePid(t, path+".pid", fmt.Sprintf("%d\n", deadPid(t))) + + cfg := recorderConfig(t, newVirtualClock(testNow), path) + if _, err := check(context.Background(), cfg, newCheckSource(nil)); err == nil || !strings.Contains(err.Error(), "unparseable") { + t.Fatalf("check() = %v, want a refusal naming the unparseable tail", err) + } +} + // P5's "two authorities", from check's side: maxGap comes from the cadence the // header records, never from a re-derivation off intervalSeconds. The // fail-open direction is the one asserted — a log recorded at 5s on a 60s rule diff --git a/grafana-alertcheck/internal/gate/classify_test.go b/grafana-alertcheck/internal/gate/classify_test.go index 25d766b6c..6522f6e38 100644 --- a/grafana-alertcheck/internal/gate/classify_test.go +++ b/grafana-alertcheck/internal/gate/classify_test.go @@ -134,6 +134,33 @@ func TestClassifyRule_PreexistingThatRecoversIsRecoveredAndNotAViolation(t *test } } +// §22.2's "late condition": bad for 58 of a 60-minute window, clear at +// minute 58, still passes with a large BadFor — never a fail against some +// derived deadline (e.g. "must clear before 90% of the window"). +func TestClassifyRule_LateRecoveryPassesRegardlessOfHowLateItIs(t *testing.T) { + from := time.Date(2026, 1, 1, 0, 0, 0, 0, time.UTC) + to := from.Add(60 * time.Minute) + clearAt := from.Add(58 * time.Minute) + def := Definition{UID: "r1", Title: "R1"} + key := instanceKey(lbl("a")) + + polls := []Poll{ + abnormalPoll("r1", from, StateFiring, lbl("a"), from.Add(-time.Hour)), + clearedPoll("r1", clearAt, key), + quietPoll("r1", to), + } + outcome, badFor, viols := classifyRule(def, polls, from, to, defaultBad, PreexistingFailUnlessRecovered) + if outcome != OutcomeRecovered { + t.Fatalf("outcome = %v, want recovered even 58 minutes into a 60-minute window", outcome) + } + if want := clearAt.Sub(from); badFor != want { + t.Fatalf("badFor = %v, want the full %v bad duration, not a value clamped against a deadline", badFor, want) + } + if len(viols) != 0 { + t.Fatalf("viols = %+v, want none: there is no deadline a preexisting recovery must beat", viols) + } +} + func TestClassifyRule_PreexistingStillBadAtWindowEndIsPersistentlyBad(t *testing.T) { from := time.Date(2026, 1, 1, 0, 0, 0, 0, time.UTC) to := from.Add(10 * time.Minute) @@ -178,6 +205,45 @@ func TestClassifyRule_ClearThenBadAgainIsFlapping(t *testing.T) { } } +// §22.2: "a clear and then a second bad state gives flapping, at each +// possible time of the second bad state." A table over where the second +// onset lands — immediately after the clear, mid-window, and right at the +// last instant before windowEnd — closes the boundary this single fixed +// timing above cannot. +func TestClassifyRule_FlappingAtEveryTimingOfTheSecondOnset(t *testing.T) { + from := time.Date(2026, 1, 1, 0, 0, 0, 0, time.UTC) + to := from.Add(10 * time.Minute) + clearAt := from.Add(2 * time.Minute) + def := Definition{UID: "r1", Title: "R1"} + key := instanceKey(lbl("a")) + + tests := []struct { + name string + secondOnset time.Time + }{ + {"immediately after the clear", clearAt.Add(time.Second)}, + {"mid-window", from.Add(5 * time.Minute)}, + {"the last instant before windowEnd", to.Add(-time.Second)}, + } + for _, tc := range tests { + t.Run(tc.name, func(t *testing.T) { + polls := []Poll{ + abnormalPoll("r1", from, StateFiring, lbl("a"), from), + clearedPoll("r1", clearAt, key), + abnormalPoll("r1", tc.secondOnset, StateFiring, lbl("a"), tc.secondOnset), + quietPoll("r1", to), + } + outcome, _, viols := classifyRule(def, polls, from, to, defaultBad, PreexistingFailUnlessRecovered) + if outcome != OutcomeFlapping { + t.Fatalf("outcome = %v, want flapping for a second onset at %s", outcome, tc.secondOnset) + } + if len(viols) != 1 || viols[0].Outcome != OutcomeFlapping { + t.Fatalf("viols = %+v, want one flapping violation", viols) + } + }) + } +} + // --- H2: vanished is a discontinuity, never a clear --- func TestClassifyRule_VanishedWhileBadStaysPersistentlyBad(t *testing.T) { @@ -409,6 +475,148 @@ func TestDecide_UnobservableWinsEvenAlongsideARealViolation(t *testing.T) { } } +// §22.10: "a clean verdict with a coverage gap ... must never give exit 0", +// and "a recovered verdict and a skipped verdict also need proved coverage +// of the full window." One genuinely unobservable rule ("broken", zero +// polls) alongside a rule with each of the three favorable outcomes — none +// of them may waive the run. +func TestDecide_UnobservableRuleWinsOverEveryFavorableOutcome(t *testing.T) { + from := time.Date(2026, 1, 1, 0, 0, 0, 0, time.UTC) + to := from.Add(10 * time.Minute) + + tests := []struct { + name string + goodPolls []Poll + pausedAtStart bool + wantOutcome Outcome + }{ + { + name: "clean", + goodPolls: func() []Poll { + var polls []Poll + for ts := from; !ts.After(to); ts = ts.Add(30 * time.Second) { + polls = append(polls, quietPoll("good", ts)) + } + return polls + }(), + wantOutcome: OutcomeClean, + }, + { + // Dense 30s-spaced polls throughout, so "good"'s own coverage + // proves clean on its own — a sparse abnormal/cleared/quiet + // triple (enough for classifyRule alone) would leave a + // heartbeat gap that muddies which rule made the run fail. + name: "recovered", + goodPolls: func() []Poll { + var polls []Poll + clearAt := from.Add(3 * time.Minute) + key := instanceKey(lbl("a")) + cleared := false + for ts := from; !ts.After(to); ts = ts.Add(30 * time.Second) { + switch { + case ts.Equal(clearAt): + polls = append(polls, clearedPoll("good", ts, key)) + cleared = true + case !cleared: + polls = append(polls, abnormalPoll("good", ts, StateFiring, lbl("a"), from.Add(-time.Hour))) + default: + polls = append(polls, quietPoll("good", ts)) + } + } + return polls + }(), + wantOutcome: OutcomeRecovered, + }, + { + name: "skipped", + pausedAtStart: true, + wantOutcome: OutcomeSkipped, + }, + } + + for _, tc := range tests { + t.Run(tc.name, func(t *testing.T) { + defs := []Definition{{UID: "good", Title: "Good"}, {UID: "broken", Title: "Broken"}} + rt := map[string]ruleTimings{ + "good": newRuleTimings(30*time.Second, 60), + "broken": newRuleTimings(30*time.Second, 60), + } + gt := globalTimings{} + pol := Policy{From: from, To: to} + + h := Header{StartedAt: from.Add(-time.Hour)} + if tc.pausedAtStart { + h.Rules = []LoggedRule{{UID: "good", IsPaused: true}} + } + + // "broken" gets no polls at all: no sentinel-worthy heartbeats, + // so it is unobservable regardless of "good". + sentinel := to + res, err := decide(h, tc.goodPolls, &sentinel, defs, rt, gt, pol) + if err == nil { + t.Fatalf("err = nil, want non-nil: 'broken' is unobservable regardless of 'good' being %s", tc.name) + } + var gotGood, gotBroken Outcome + for _, v := range res.Verdicts { + switch v.RuleUID { + case "good": + gotGood = v.Outcome + case "broken": + gotBroken = v.Outcome + } + } + if gotGood != tc.wantOutcome { + t.Errorf("good.Outcome = %v, want %v", gotGood, tc.wantOutcome) + } + if gotBroken != OutcomeUnobservable { + t.Errorf("broken.Outcome = %v, want unobservable", gotBroken) + } + }) + } +} + +// §22.10: the table above puts the coverage gap on a DIFFERENT rule from the +// one with the favorable outcome. This pins the tighter claim: a rule that +// itself recovers, but ALSO itself has a coverage gap, is still overridden to +// unobservable — the favorable classification of a rule is never a reason to +// skip that same rule's own coverage check. +func TestDecide_RecoveredOutcomeOverriddenByItsOwnCoverageGap(t *testing.T) { + from := time.Date(2026, 1, 1, 0, 0, 0, 0, time.UTC) + to := from.Add(10 * time.Minute) + def := Definition{UID: "r1", Title: "R1"} + defs := []Definition{def} + rt := map[string]ruleTimings{"r1": newRuleTimings(30*time.Second, 60)} // maxGap = 60s + gt := globalTimings{} + pol := Policy{From: from, To: to} + key := instanceKey(lbl("a")) + + polls := []Poll{ + abnormalPoll("r1", from, StateFiring, lbl("a"), from.Add(-time.Hour)), + clearedPoll("r1", from.Add(30*time.Second), key), + } + for ts := from.Add(time.Minute); !ts.After(to); ts = ts.Add(30 * time.Second) { + // A gap from from+1.5m to from+4m — well past the 60s maxGap — + // sitting entirely AFTER the clear, so classifyRule alone would + // still call this rule `recovered`. + if ts.After(from.Add(90*time.Second)) && ts.Before(from.Add(4*time.Minute)) { + continue + } + polls = append(polls, quietPoll("r1", ts)) + } + sentinel := to + + res, err := decide(Header{StartedAt: from.Add(-time.Hour)}, polls, &sentinel, defs, rt, gt, pol) + if err == nil { + t.Fatalf("err = nil, want non-nil: r1's own coverage gap must fail the run even though it recovered") + } + if len(res.Verdicts) != 1 || res.Verdicts[0].Outcome != OutcomeUnobservable { + t.Fatalf("Verdicts = %+v, want unobservable, never recovered", res.Verdicts) + } + if cov := res.Coverage["r1"]; cov.Proved { + t.Fatalf("Coverage = %+v, want not proved", cov) + } +} + func TestDecide_CleanWindowIsAPass(t *testing.T) { from := time.Date(2026, 1, 1, 0, 0, 0, 0, time.UTC) to := from.Add(10 * time.Minute) @@ -436,6 +644,55 @@ func TestDecide_CleanWindowIsAPass(t *testing.T) { } } +// §22.7's second of the plan's "if only three tests could exist" cases: a +// pause and then an unpause inside the window, with an episode that would +// fire and resolve entirely inside the blind interval. A drain wait alone — +// "did the rule eventually evaluate through windowEnd?" — would see +// lastEvaluation catch up after the unpause and answer yes, a pass. decide() +// never runs a drain wait (that is check.go's I/O concern, §14.6); this pins +// that proveCoverage's own per-poll checks already refuse the window without +// one, so a live drain wait is not what is saving this case. +func TestDecide_PauseThenUnpauseWithHiddenEpisodeGivesUnobservableNotClean(t *testing.T) { + from := time.Date(2026, 1, 1, 0, 0, 0, 0, time.UTC) + to := from.Add(20 * time.Minute) + def := Definition{UID: "r1", Title: "R1"} + defs := []Definition{def} + rt := map[string]ruleTimings{"r1": newRuleTimings(30*time.Second, 60)} + gt := globalTimings{} + pol := Policy{From: from, To: to} + + pauseStart := from.Add(5 * time.Minute) + pauseEnd := from.Add(10 * time.Minute) + + var polls []Poll + for ts := from; !ts.After(pauseStart.Add(-30 * time.Second)); ts = ts.Add(30 * time.Second) { + polls = append(polls, quietPoll("r1", ts)) + } + for ts := pauseStart; !ts.After(pauseEnd); ts = ts.Add(30 * time.Second) { + // No fire/resolve is ever observed here: the rule was not + // evaluating, so any real episode inside this stretch is invisible + // to every poll (§14.7). + polls = append(polls, Poll{RuleUID: "r1", GrafanaNow: ts, Found: true, Health: "ok", IsPaused: true, LastEvaluation: pauseStart}) + } + for ts := pauseEnd.Add(30 * time.Second); !ts.After(to); ts = ts.Add(30 * time.Second) { + // Evaluations resume and catch straight up — a drain wait's final + // "did it reach windowEnd" question would answer yes. + polls = append(polls, quietPoll("r1", ts)) + } + sentinel := to + + res, err := decide(Header{StartedAt: from.Add(-time.Hour)}, polls, &sentinel, defs, rt, gt, pol) + if err == nil { + t.Fatalf("err = nil, want the pause-then-unpause blind interval to fail closed") + } + if len(res.Verdicts) != 1 || res.Verdicts[0].Outcome != OutcomeUnobservable { + t.Fatalf("Verdicts = %+v, want unobservable, never clean", res.Verdicts) + } + if cov := res.Coverage["r1"]; cov.Proved { + t.Fatalf("Coverage = %+v, want not proved", cov) + } +} + // --- MinObserved shortfall (§12) --- func TestDecide_SkippedOnlyShortfallProducesAViolationWithoutAnError(t *testing.T) { @@ -808,6 +1065,34 @@ func TestClassifyRule_OnsetJustPastWindowEndIsNewlyBadNotClean(t *testing.T) { } if len(viols) != 1 { t.Fatalf("viols = %+v, want exactly one newly_bad violation", viols) + + } +} + +// §22.2: "a clear after `to` gives persistently_bad." classifyRule filters +// its input to [from, windowEnd] itself (inWindowPolls), so a Cleared event +// GENUINELY past windowEnd — well beyond any skew bound, unlike the clamp +// case above — never reaches the timeline at all: the instance is still bad +// at windowEnd as far as this window is concerned. +func TestClassifyRule_ClearAfterWindowEndIsPersistentlyBad(t *testing.T) { + from := time.Date(2026, 1, 1, 0, 0, 0, 0, time.UTC) + to := from.Add(10 * time.Minute) + def := Definition{UID: "r1", Title: "R1"} + key := instanceKey(lbl("a")) + + polls := []Poll{ + abnormalPoll("r1", from, StateFiring, lbl("a"), from.Add(-time.Hour)), + clearedPoll("r1", to.Add(time.Hour), key), // far past `to`, not a boundary case + } + outcome, badFor, viols := classifyRule(def, polls, from, to, defaultBad, PreexistingFailUnlessRecovered) + if outcome != OutcomePersistentlyBad { + t.Fatalf("outcome = %v, want persistently_bad: a clear outside the window must not read as a recovery", outcome) + } + if badFor != to.Sub(from) { + t.Fatalf("badFor = %v, want the full window %v", badFor, to.Sub(from)) + } + if len(viols) != 1 || viols[0].Outcome != OutcomePersistentlyBad { + t.Fatalf("viols = %+v, want one persistently_bad violation", viols) } } diff --git a/grafana-alertcheck/internal/gate/coverage.go b/grafana-alertcheck/internal/gate/coverage.go index 03b1e1a75..77c66bb28 100644 --- a/grafana-alertcheck/internal/gate/coverage.go +++ b/grafana-alertcheck/internal/gate/coverage.go @@ -242,10 +242,21 @@ func proveCoverage(h Header, polls []Poll, sentinel *time.Time, t ruleTimings, d fail(ReasonRuleAbsent, fmt.Sprintf("state endpoint returned no rule on %d poll(s), first at %s", absentCount, absentAt.Format(time.RFC3339))) } - // Check 9 — KeepLast (§10.2). A note, never fatal. It surfaces only as an - // instance Reason after P1.2a's parsing, and Reasons keys can be - // comma-joined composites, so membership (reasonsContain) is required — - // indexing "KeepLast" directly would miss "KeepLast, MissingSeries". + // Check 9 — KeepLast (§10.2). Two distinct notes, both non-fatal: + // + // DECLARED: the rule's own no_data_state/exec_err_state is configured as + // KeepLast — a standing blind spot (§10.2's "unclear condition") whether + // or not it is ever exercised during this particular window. This reads + // def, not polls, so it fires exactly once regardless of poll content. + if def.NoDataState == keepLastReason || def.ExecErrState == keepLastReason { + res.Notes = append(res.Notes, fmt.Sprintf( + "rule %q: configured with no_data_state/exec_err_state=KeepLast — a stale state can continue past a real fault (§10.2)", def.Title)) + } + // OBSERVED: an instance actually reported the KeepLast reason during the + // window. It surfaces only as an instance Reason after P1.2a's parsing, + // and Reasons keys can be comma-joined composites, so membership + // (reasonsContain) is required — indexing "KeepLast" directly would miss + // "KeepLast, MissingSeries". for _, p := range inWindow { if reasonsContain(p.Reasons, keepLastReason) { res.Notes = append(res.Notes, fmt.Sprintf( diff --git a/grafana-alertcheck/internal/gate/coverage_test.go b/grafana-alertcheck/internal/gate/coverage_test.go index 6419a8b2e..3e6829e5a 100644 --- a/grafana-alertcheck/internal/gate/coverage_test.go +++ b/grafana-alertcheck/internal/gate/coverage_test.go @@ -71,6 +71,22 @@ func TestProveCoverage_SentinelBeforeGraceIsUnobservable(t *testing.T) { if res.Reason != ReasonSentinelEarly { t.Fatalf("Reason = %q, want sentinel_early", res.Reason) } + if !res.Unobservable || res.Proved { + t.Fatalf("res = %+v, want Unobservable and not Proved — a reason string with no consequence is not a coverage failure", res) + } + + // The consequence: decide() must turn this into exit 2, never a pass. + defs := []Definition{def} + drt := map[string]ruleTimings{def.UID: rt} + gt := globalTimings{transitionGrace: grace} + pol := Policy{From: from, To: to} + dres, err := decide(Header{StartedAt: from.Add(-time.Hour)}, nil, &sentinel, defs, drt, gt, pol) + if err == nil { + t.Fatalf("decide() err = nil, want non-nil: a sentinel short of to+grace must fail the run") + } + if len(dres.Verdicts) != 1 || dres.Verdicts[0].Outcome != OutcomeUnobservable { + t.Fatalf("Verdicts = %+v, want one unobservable verdict", dres.Verdicts) + } } func TestProveCoverage_SentinelExactlyAtGraceIsFine(t *testing.T) { @@ -107,6 +123,22 @@ func TestProveCoverage_FromBeforeRecordIsUnobservable(t *testing.T) { if res.Reason != ReasonFromBeforeRecord { t.Fatalf("Reason = %q, want from_before_record", res.Reason) } + if !res.Unobservable || res.Proved { + t.Fatalf("res = %+v, want Unobservable and not Proved — a reason string with no consequence is not a coverage failure", res) + } + + // The consequence: decide() must turn this into exit 2, never a pass. + defs := []Definition{def} + drt := map[string]ruleTimings{def.UID: rt} + gt := globalTimings{} + pol := Policy{From: from, To: to} + dres, err := decide(Header{StartedAt: started}, nil, &sentinel, defs, drt, gt, pol) + if err == nil { + t.Fatalf("decide() err = nil, want non-nil: `from` before the recording started must fail the run") + } + if len(dres.Verdicts) != 1 || dres.Verdicts[0].Outcome != OutcomeUnobservable { + t.Fatalf("Verdicts = %+v, want one unobservable verdict", dres.Verdicts) + } } // --- Check 3: heartbeat continuity (§6) --- @@ -388,7 +420,7 @@ func denseHealthyPolls(uid string, from, to time.Time, every time.Duration) []Po // --- Check 9: KeepLast (§10.2) --- -func TestProveCoverage_KeepLastIsNoteOnly(t *testing.T) { +func TestProveCoverage_KeepLastObservedIsNoteOnly(t *testing.T) { from := time.Date(2026, 1, 1, 0, 0, 0, 0, time.UTC) to := from.Add(10 * time.Minute) rt := newRuleTimings(30*time.Second, 60) @@ -414,6 +446,40 @@ func TestProveCoverage_KeepLastIsNoteOnly(t *testing.T) { } } +// §22.2/§10.2: "KeepLast in the configuration gives a note" — a DIFFERENT +// claim from the observed-reason test above. A rule DECLARED with +// no_data_state or exec_err_state = KeepLast is a standing blind spot +// whether or not any poll ever actually reports the reason, so the note +// must fire off the definition alone, over an otherwise perfectly healthy +// window with zero KeepLast reasons anywhere in it. +func TestProveCoverage_KeepLastConfiguredIsNoteOnly(t *testing.T) { + from := time.Date(2026, 1, 1, 0, 0, 0, 0, time.UTC) + to := from.Add(10 * time.Minute) + rt := newRuleTimings(30*time.Second, 60) + + tests := []struct { + name string + def Definition + }{ + {"no_data_state", Definition{UID: "r1", Title: "R1", NoDataState: "KeepLast"}}, + {"exec_err_state", Definition{UID: "r1", Title: "R1", ExecErrState: "KeepLast"}}, + } + for _, tc := range tests { + t.Run(tc.name, func(t *testing.T) { + polls := denseHealthyPolls("r1", from, to, 30*time.Second) + sentinel := to + + res := proveCoverage(Header{StartedAt: from.Add(-time.Hour)}, polls, &sentinel, rt, tc.def, from, to, 0) + if !res.Proved { + t.Fatalf("Proved = false, want true: a declared KeepLast is a note, never fatal: %+v", res) + } + if !anyContains(res.Notes, "KeepLast") { + t.Fatalf("Notes = %v, want a KeepLast note from the definition alone, with zero KeepLast reasons observed", res.Notes) + } + }) + } +} + // --- Clock domains (§16) --- // TestProveCoverage_SkewTranslationAtWindowBoundary pins §16's "Clock diff --git a/grafana-alertcheck/internal/gate/parse_state_test.go b/grafana-alertcheck/internal/gate/parse_state_test.go index 8fca6e7ab..bc8a2a64c 100644 --- a/grafana-alertcheck/internal/gate/parse_state_test.go +++ b/grafana-alertcheck/internal/gate/parse_state_test.go @@ -293,6 +293,59 @@ func TestInstanceKey_NoCollision(t *testing.T) { } } +// minimalStateBody is the smallest H1-legal state response: one group, one +// rule, no optional keys at all, plus whatever extra is spliced in verbatim +// before the rule's closing brace — for isolating one optional key at a time +// rather than relying on a fixture that removes several together. +func minimalStateBody(extraRuleJSON string) []byte { + return fmt.Appendf(nil, + `{"status":"success","data":{"groups":[{"file":"F","name":"G","interval":60,`+ + `"rules":[{"uid":"r1","name":"R1","state":"inactive","health":"ok","isPaused":false,`+ + `"lastEvaluation":"2026-01-01T00:00:00Z"%s}]}]}}`, extraRuleJSON) +} + +// §22.2: keepFiringFor is named alongside alerts/totals/labels as an optional +// key (§3.1), but state_missing_optional.json removes it together with +// everything else — never in isolation, so a regression that made it +// required specifically would not be caught by that fixture alone. +func TestParseState_KeepFiringForIsOptional(t *testing.T) { + tests := []struct { + name string + extra string + }{ + {"present", `,"keepFiringFor":300`}, + {"absent", ""}, + } + for _, tc := range tests { + t.Run(tc.name, func(t *testing.T) { + rules, err := ParseState(minimalStateBody(tc.extra)) + if err != nil { + t.Fatalf("ParseState: %v", err) + } + if len(rules) != 1 { + t.Fatalf("rules = %+v, want one", rules) + } + }) + } +} + +// §22.2: labels is optional at the INSTANCE level (opt(m, "labels", ...) in +// parseInstance), distinct from the rule-level labels state_missing_optional.json +// already covers — an instance can exist with no labels of its own. +func TestParseState_InstanceWithoutLabelsParses(t *testing.T) { + body := minimalStateBody(`,"alerts":[{"state":"Normal","activeAt":"2026-01-01T00:00:00Z"}]`) + rules, err := ParseState(body) + if err != nil { + t.Fatalf("ParseState: %v", err) + } + if len(rules) != 1 || len(rules[0].Instances) != 1 { + t.Fatalf("rules = %+v, want one rule with one instance", rules) + } + if got := rules[0].Instances[0].Labels; len(got) != 0 { + t.Errorf("Instance.Labels = %v, want empty/nil", got) + } +} + // synthesizeHighCardinalityState builds a state response with a single rule // holding `alerting` Alerting instances and `normal` Normal instances, by // cloning the one real instance in state_one_instance.json. It is never diff --git a/grafana-alertcheck/internal/gate/resolve_test.go b/grafana-alertcheck/internal/gate/resolve_test.go index 77fd3bb5d..192ed3dc2 100644 --- a/grafana-alertcheck/internal/gate/resolve_test.go +++ b/grafana-alertcheck/internal/gate/resolve_test.go @@ -233,6 +233,26 @@ func TestResolve_CollapseByUIDGivesNoteNotError(t *testing.T) { } } +// §22.2: "the same rule with two identical names ... must collapse to one +// rule" — the literal exact-duplicate-string case, distinct from the +// different-spellings case above. +func TestResolve_IdenticalDuplicateNameCollapsesWithNote(t *testing.T) { + defs := rulerDefs(t) + resolved, notes, err := Resolve(defs, []string{ + "example_workflow_paused_rule", + "example_workflow_paused_rule", + }, "") + if err != nil { + t.Fatalf("Resolve: unexpected error: %v", err) + } + if len(resolved) != 1 || resolved[0].UID != "rule0000007" { + t.Fatalf("resolved = %+v, want exactly one rule0000007", resolved) + } + if len(notes) != 1 { + t.Fatalf("notes = %v, want exactly one collapse note", notes) + } +} + func TestResolve_MinObservedCountIsPostCollapse(t *testing.T) { defs := rulerDefs(t) names := []string{ diff --git a/grafana-alertcheck/internal/gate/schedule_test.go b/grafana-alertcheck/internal/gate/schedule_test.go index b1871bb3d..aee22f37f 100644 --- a/grafana-alertcheck/internal/gate/schedule_test.go +++ b/grafana-alertcheck/internal/gate/schedule_test.go @@ -69,6 +69,29 @@ func TestDeriveTimings_TransitionGraceExcludesSkippedRule(t *testing.T) { } } +// §22.2/§22.3: `for: 1d` and `for: 1w` parse correctly (parse_ruler_test.go), +// but that alone never proves they flow into transitionGrace — a Prometheus +// duration parser that silently truncated to time.Duration's other units, or +// a transitionGrace derivation that only ever saw hand-built values, could +// each pass every existing test and still be wrong together. This drives the +// real ruler_rules.json fixture (rule0000010, for:1w, DERIVED to exercise the +// w unit — testdata/README.md) through ParseDefinitions and DeriveTimings. +func TestDeriveTimings_RealForOneWeekRuleSetsTransitionGrace(t *testing.T) { + defs := rulerDefs(t) + _, global, notes := DeriveTimings(defs, 0) + if len(notes) != 0 { + t.Fatalf("notes = %v, want none: no --poll-interval override is given, so no override note should fire", notes) + } + + want := 7*24*time.Hour + 60*time.Second // rule0000010: for=1w, intervalSeconds=60 + if global.transitionGrace != want { + t.Fatalf("transitionGrace = %s, want %s (rule0000010's for:1w plus its interval)", global.transitionGrace, want) + } + if !strings.Contains(global.graceSource, "Example Failure Ratio Above 10 Percent Weekly") { + t.Errorf("graceSource = %q, want it to name rule0000010", global.graceSource) + } +} + func TestDeriveTimings_TransitionGraceZeroWhenAllSkipped(t *testing.T) { defs := []Definition{{UID: "r1", Title: "R1", IntervalSeconds: 60, For: time.Hour, IsPaused: true}} _, global, _ := DeriveTimings(defs, 0) @@ -468,6 +491,34 @@ func TestStartupSummary_WarningWhenGraceTooLarge(t *testing.T) { } } +// §22.3: "a rule with for: 15m in a 10-minute window gives the warning about +// a large grace period" — no such rule exists in the real capture +// (testdata/README.md), so the test above pins the mechanism with a +// hand-built globalTimings. This drives the same warning off the real +// ruler_rules.json fixture's for:1w rule instead, tying ParseDefinitions and +// DeriveTimings into the warning end to end, not just the warning formula in +// isolation. +func TestStartupSummary_RealForOneWeekRuleTriggersWarning(t *testing.T) { + defs := rulerDefs(t) + _, global, notes := DeriveTimings(defs, 0) + if len(notes) != 0 { + t.Fatalf("notes = %v, want none: no --poll-interval override is given, so no override note should fire", notes) + } + + from := time.Date(2026, 1, 1, 0, 0, 0, 0, time.UTC) + to := from.Add(10 * time.Minute) // transitionGrace (>1w) dwarfs 1/4 of this window + summary, warning := StartupSummary(from, to, global) + if !strings.Contains(summary, "planned run time") { + t.Errorf("summary = %q, want it to name the planned run time", summary) + } + if warning == "" { + t.Fatal("warning = \"\", want one: a real for:1w rule's transitionGrace vastly exceeds 1/4 of a 10m window") + } + if !strings.Contains(warning, "Example Failure Ratio Above 10 Percent Weekly") { + t.Errorf("warning = %q, want it to name rule0000010", warning) + } +} + func TestStartupSummary_NoWarningWhenGraceSmall(t *testing.T) { from := time.Date(2026, 1, 1, 0, 0, 0, 0, time.UTC) to := from.Add(time.Hour) diff --git a/grafana-alertcheck/internal/gate/source_test.go b/grafana-alertcheck/internal/gate/source_test.go index ba864a1a8..2998b4776 100644 --- a/grafana-alertcheck/internal/gate/source_test.go +++ b/grafana-alertcheck/internal/gate/source_test.go @@ -349,6 +349,59 @@ func TestHTTPSource_ObservationTiming(t *testing.T) { } } +// §22.7/§16: a genuinely discriminating regression for "the gate compares +// staleness against the Date header, never the runner's clock." lastEvaluation +// sits 100s behind Grafana's TRUE now (obs.GrafanaNow, from the Date header) +// — under the 120s evalStaleAfter limit — but 130s behind the RUNNER's clock. +// An implementation that leaked the runner's clock into the staleness +// comparison, instead of the Date header, would report a false violation +// here; coverage_test.go's TestProveCoverage_SkewTranslationAtWindowBoundary +// cannot catch that, because it sets LastEvaluation equal to GrafanaNow on +// every poll, making staleness zero regardless of which clock is used. +func TestHTTPSourceStalenessNeverFalsePositiveUnderSkew(t *testing.T) { + runnerNow := time.Date(2026, 1, 1, 0, 0, 0, 0, time.UTC) + clock := newFakeClock(runnerNow) + const skew = 30 * time.Second // the runner's clock reads 30s ahead of Grafana's + serverDate := runnerNow.Add(-skew) + lastEval := serverDate.Add(-100 * time.Second) + + def := Definition{UID: "r1", Title: "Rule One"} + srv := rawHTTPServer(t, func(r *http.Request) []byte { + body := fmt.Sprintf(`{"status":"success","data":{"groups":[{"file":"F","name":"G","interval":60,"rules":[`+ + `{"uid":%q,"name":%q,"state":"inactive","health":"ok","isPaused":false,"lastEvaluation":%q}`+ + `]}]}}`, def.UID, def.Title, lastEval.UTC().Format(time.RFC3339)) + return rawResponse(200, "OK", map[string]string{ + "Content-Type": "application/json", + "Date": serverDate.UTC().Format(http.TimeFormat), + }, body) + }) + + src := NewHTTPSource(srv.URL, "", clock) + obs, err := src.RuleState(context.Background(), def.Title) + if err != nil { + t.Fatalf("RuleState(): %v", err) + } + if !obs.GrafanaNow.Equal(serverDate) { + t.Fatalf("GrafanaNow = %s, want the Date header %s, never the runner's clock %s", obs.GrafanaNow, serverDate, runnerNow) + } + if len(obs.Rules) != 1 { + t.Fatalf("Rules = %+v, want exactly one", obs.Rules) + } + + rt := newRuleTimings(30*time.Second, 60) // evalStaleAfter = 120s + from := serverDate.Add(-10 * time.Minute) + to := serverDate + polls := denseHealthyPolls(def.UID, from, to, 30*time.Second) + polls[len(polls)-1].LastEvaluation = obs.Rules[0].LastEvaluation // the real, HTTP-sourced value + sentinel := to + + res := proveCoverage(Header{StartedAt: from.Add(-time.Hour)}, polls, &sentinel, rt, def, from, to, 0) + if res.Unobservable { + t.Fatalf("Coverage = %+v, want no violation: 100s behind Grafana's TRUE now is under the 120s limit — "+ + "only a runner-clock leak (skewed +30s here) would push this over", res) + } +} + func TestHTTPSource_Retry_TransientRecovers(t *testing.T) { var mu sync.Mutex calls := 0 diff --git a/grafana-alertcheck/internal/gate/watch_test.go b/grafana-alertcheck/internal/gate/watch_test.go index fa2af889a..5a0337a37 100644 --- a/grafana-alertcheck/internal/gate/watch_test.go +++ b/grafana-alertcheck/internal/gate/watch_test.go @@ -4,6 +4,7 @@ import ( "context" "errors" "fmt" + "maps" "os" "path/filepath" "slices" @@ -429,6 +430,12 @@ func TestPrepareWatchDoesNotWaitForPausedRules(t *testing.T) { if !polls[0].Found || !polls[0].GrafanaNow.Equal(testNow) { t.Errorf("first poll = %+v, want a found observation at %s", polls[0], testNow) } + // §22.3: "the poll record holds the state histogram. Assert that watch + // writes it" — through a real prepareWatch()/Reducer call, not just + // log_test.go's hand-built Writer/ReadLog round trip. + if want := map[string]int{"normal": 1}; !maps.Equal(polls[0].Histogram, want) { + t.Errorf("Histogram = %v, want %v: watch must record the state histogram on every poll it writes", polls[0].Histogram, want) + } if !strings.Contains(notes.String(), watchPausedTitle) || !strings.Contains(notes.String(), "paused") { t.Errorf("notes do not mention the paused rule:\n%s", notes.String()) } From d9286338a972b6739ecf8739d1850f9ce1855767 Mon Sep 17 00:00:00 2001 From: Bartek Tofel Date: Mon, 7 Sep 2026 16:48:53 +0200 Subject: [PATCH 30/43] chore: address code review comments --- grafana-alertcheck/internal/gate/classify.go | 4 +--- grafana-alertcheck/internal/gate/coverage.go | 17 ++++++++++++----- 2 files changed, 13 insertions(+), 8 deletions(-) diff --git a/grafana-alertcheck/internal/gate/classify.go b/grafana-alertcheck/internal/gate/classify.go index c4bda0a54..76f05edc0 100644 --- a/grafana-alertcheck/internal/gate/classify.go +++ b/grafana-alertcheck/internal/gate/classify.go @@ -254,9 +254,7 @@ func classifyRule(def Definition, polls []Poll, from, windowEnd time.Time, badSt } // onsetOf resolves a fresh episode's start: the instance's own ActiveAt, // translated to the runner domain by this poll's skew, clamped to - // [from, windowEnd] so closeEpisode never has to undo its own clamp on a - // start that already overran the window (§16: a poll admitted by the skew - // bound can carry an ActiveAt past windowEnd). + // [from, windowEnd]. onsetOf := func(p Poll, inst Instance) time.Time { start := runnerTime(p, inst.ActiveAt) if start.Before(from) { diff --git a/grafana-alertcheck/internal/gate/coverage.go b/grafana-alertcheck/internal/gate/coverage.go index 77c66bb28..ec13ac91c 100644 --- a/grafana-alertcheck/internal/gate/coverage.go +++ b/grafana-alertcheck/internal/gate/coverage.go @@ -244,11 +244,18 @@ func proveCoverage(h Header, polls []Poll, sentinel *time.Time, t ruleTimings, d // Check 9 — KeepLast (§10.2). Two distinct notes, both non-fatal: // - // DECLARED: the rule's own no_data_state/exec_err_state is configured as - // KeepLast — a standing blind spot (§10.2's "unclear condition") whether - // or not it is ever exercised during this particular window. This reads - // def, not polls, so it fires exactly once regardless of poll content. - if def.NoDataState == keepLastReason || def.ExecErrState == keepLastReason { + // DECLARED: the rule's no_data_state/exec_err_state is configured as + // KeepLast — a standing blind spot (§10.2). Prefer the header's + // LoggedRule snapshot: in log mode def is re-resolved after the window + // and can drift (see pausedAtStart, log.go). Reads config, fires once. + nds, ees := def.NoDataState, def.ExecErrState + for _, lr := range h.Rules { + if lr.UID == def.UID { + nds, ees = lr.NoDataState, lr.ExecErrState + break + } + } + if nds == keepLastReason || ees == keepLastReason { res.Notes = append(res.Notes, fmt.Sprintf( "rule %q: configured with no_data_state/exec_err_state=KeepLast — a stale state can continue past a real fault (§10.2)", def.Title)) } From 2baacbb6daf625405bead7bfa0639aff4c28f051 Mon Sep 17 00:00:00 2001 From: Bartek Tofel Date: Tue, 1 Sep 2026 08:45:57 +0200 Subject: [PATCH 31/43] chore: more concise comments --- .../cmd/grafana-alertcheck/check.go | 35 +- .../cmd/grafana-alertcheck/check_test.go | 33 +- .../cmd/grafana-alertcheck/common.go | 27 +- .../cmd/grafana-alertcheck/env.go | 3 +- .../cmd/grafana-alertcheck/list.go | 11 +- .../cmd/grafana-alertcheck/main.go | 10 +- .../cmd/grafana-alertcheck/table.go | 30 +- .../cmd/grafana-alertcheck/table_test.go | 25 +- .../cmd/grafana-alertcheck/watch.go | 21 +- .../cmd/grafana-alertcheck/watch_test.go | 10 +- grafana-alertcheck/internal/gate/check.go | 405 ++++++++---------- .../internal/gate/check_process.go | 2 +- .../internal/gate/check_test.go | 203 +++++---- grafana-alertcheck/internal/gate/classify.go | 233 +++++----- .../internal/gate/classify_test.go | 114 +++-- grafana-alertcheck/internal/gate/coverage.go | 202 +++++---- .../internal/gate/coverage_test.go | 112 +++-- grafana-alertcheck/internal/gate/duration.go | 2 +- grafana-alertcheck/internal/gate/flock.go | 4 +- grafana-alertcheck/internal/gate/jsonreq.go | 2 +- grafana-alertcheck/internal/gate/log.go | 191 ++++----- grafana-alertcheck/internal/gate/log_test.go | 47 +- .../internal/gate/parse_ruler.go | 24 +- .../internal/gate/parse_ruler_test.go | 5 +- .../internal/gate/parse_state.go | 39 +- .../internal/gate/parse_state_test.go | 30 +- grafana-alertcheck/internal/gate/resolve.go | 47 +- .../internal/gate/resolve_test.go | 14 +- grafana-alertcheck/internal/gate/schedule.go | 181 ++++---- .../internal/gate/schedule_test.go | 36 +- grafana-alertcheck/internal/gate/source.go | 93 ++-- .../internal/gate/source_fake_test.go | 27 +- .../internal/gate/source_test.go | 17 +- .../internal/gate/testdata/README.md | 45 +- grafana-alertcheck/internal/gate/watch.go | 150 +++---- .../internal/gate/watch_daemon_test.go | 22 +- .../internal/gate/watch_process.go | 10 +- .../internal/gate/watch_test.go | 69 ++- 38 files changed, 1214 insertions(+), 1317 deletions(-) diff --git a/grafana-alertcheck/cmd/grafana-alertcheck/check.go b/grafana-alertcheck/cmd/grafana-alertcheck/check.go index f4e9dab26..3f2d295d6 100644 --- a/grafana-alertcheck/cmd/grafana-alertcheck/check.go +++ b/grafana-alertcheck/cmd/grafana-alertcheck/check.go @@ -17,28 +17,27 @@ const checkUsage = "usage: grafana-alertcheck check [--in ] [--pidfile F] "[--alerts ...] [--folder F] [--states ...] [--preexisting ...] [--min-observed N] [--allow-paused] " + "[--nodata-is-unobservable] [--concurrency N] [--output json]" -// runCheck is the classify step's CLI surface: parse flags into a -// gate.Config, run gate.Check, and translate its (Result, error) into -// §20.2/§20.3's output and exit code. All of the correctness lives in -// gate.Check (P9) and decide (P8) — this file's only job is presentation and -// the H6/H7 exit-code mapping, which exitCode below keeps as one pure -// function so it can be tested without a network. +// runCheck is the classify step's CLI surface: parse flags into a gate.Config, +// run gate.Check, and translate its (Result, error) into output and an exit +// code. All of the correctness lives in the gate package — this file's only job +// is presentation and the exit-code mapping, which exitCode below keeps as one +// pure function so it can be tested without a network. func runCheck(args []string, stdin io.Reader, stdout, stderr io.Writer) int { fs := flag.NewFlagSet("check", flag.ContinueOnError) fs.SetOutput(stderr) fs.Usage = func() { fmt.Fprintln(stderr, checkUsage) } common := registerCommon(fs) - in := fs.String("in", "", "path of a log recorded by watch; empty selects single-step mode (§9)") + in := fs.String("in", "", "path of a log recorded by watch; empty selects single-step mode") pidfile := fs.String("pidfile", "", "pidfile of the recorder to stop before reading --in (default .pid)") - from := fs.String("from", "", "the moment the deploy finished, RFC3339 (required in recorder mode, §7)") + from := fs.String("from", "", "the moment the deploy finished, RFC3339 (required with --in)") to := fs.String("to", "", "the end of the window to classify, RFC3339 (required)") - states := fs.String("states", "", "comma-separated bad states to classify against (default: firing, §13)") - preexisting := fs.String("preexisting", "", "how to judge an instance already bad at `from` (default: fail-unless-recovered, §11.7)") - minObserved := fs.Int("min-observed", 0, "minimum rules that must be observed (default: every resolved rule, §12)") - allowPaused := fs.Bool("allow-paused", false, "do not count a rule paused before the window against --min-observed (§12.1)") - nodataIsUnobservable := fs.Bool("nodata-is-unobservable", false, "treat a sustained health=nodata as unobservable rather than a note (§10.2)") - output := fs.String("output", "", `"json" writes the machine-readable Result to stdout in addition to the table (§20.2); default is the table alone`) + states := fs.String("states", "", "comma-separated bad states to classify against (default: firing)") + preexisting := fs.String("preexisting", "", "how to judge an instance already bad at `from` (default: fail-unless-recovered)") + minObserved := fs.Int("min-observed", 0, "minimum rules that must be observed (default: every resolved rule)") + allowPaused := fs.Bool("allow-paused", false, "do not count a rule paused before the window against --min-observed") + nodataIsUnobservable := fs.Bool("nodata-is-unobservable", false, "treat a sustained health=nodata as unobservable rather than a note") + output := fs.String("output", "", `"json" writes the machine-readable Result to stdout in addition to the table; default is the table alone`) if err := fs.Parse(args); err != nil { return 2 @@ -86,7 +85,7 @@ func runCheck(args []string, stdin io.Reader, stdout, stderr io.Writer) int { Notes: stderr, } if *to == "" { - fmt.Fprintln(stderr, "check: --to is required (§7)") + fmt.Fprintln(stderr, "check: --to is required") return 2 } t, err := time.Parse(time.RFC3339, *to) @@ -131,11 +130,11 @@ func runCheck(args []string, stdin io.Reader, stdout, stderr io.Writer) int { return exitCode(result, checkErr) } -// exitCode is §20.3/H6/H7's whole mapping, kept as one pure function of +// exitCode is the whole exit-code mapping, kept as one pure function of // exactly what Check returns so it is testable without a network: err != nil // is exit 2 UNCONDITIONALLY — never 0 and never 1, even alongside real -// violations, because inability beats violation (H6) and an error is never a -// pass (H7). Violations without an error is exit 1. Neither is exit 0. +// violations, because an inability to check beats a violation and an error is +// never a pass. Violations without an error is exit 1. Neither is exit 0. func exitCode(res gate.Result, err error) int { switch { case err != nil: diff --git a/grafana-alertcheck/cmd/grafana-alertcheck/check_test.go b/grafana-alertcheck/cmd/grafana-alertcheck/check_test.go index dbf81d3a9..99e8ebc3a 100644 --- a/grafana-alertcheck/cmd/grafana-alertcheck/check_test.go +++ b/grafana-alertcheck/cmd/grafana-alertcheck/check_test.go @@ -10,9 +10,9 @@ import ( "github.com/smartcontractkit/chainlink-testing-framework/grafana-alertcheck/internal/gate" ) -// TestExitCode pins §20.3/H6/H7's mapping directly against exitCode, with no -// network involved: err != nil is exit 2 even alongside violations (H6 — -// inability beats violation), violations alone are exit 1, and neither is 0. +// The exit-code mapping, pinned directly against exitCode with no network +// involved: err != nil is exit 2 even alongside violations (an inability to +// check beats a violation), violations alone are exit 1, and neither is 0. func TestExitCode(t *testing.T) { tests := []struct { name string @@ -43,10 +43,10 @@ func writeTempAlerts(t *testing.T) string { return path } -// TestRunCheck_FlagValidation is the flag-validation matrix: every one of -// these must fail before any network call, because Config.validate() (P9) -// runs first — an unreachable GRAFANA_URL succeeding or timing out is a -// different test than these, which check pure input validation. +// The flag-validation matrix: every one of these must fail before any network +// call, because gate.Config.validate() runs first — an unreachable GRAFANA_URL +// succeeding or timing out is a different test than these, which check pure +// input validation. func TestRunCheck_FlagValidation(t *testing.T) { tests := []struct { name string @@ -73,9 +73,9 @@ func TestRunCheck_FlagValidation(t *testing.T) { return []string{"--to", "2026-01-01T00:00:00Z", "--states", "bogus", "--alerts", writeTempAlerts(t)} }, "--states"}, {"states normal is rejected", true, func(t *testing.T) []string { - // normal is the good state, never a state to classify AS bad - // (R1): accepting it would make --states normal fail every - // healthy instance, the fail-open shape H7 exists to prevent. + // normal is the good state, never a state to classify AS bad: + // accepting it would make --states normal fail every healthy + // instance. return []string{"--to", "2026-01-01T00:00:00Z", "--states", "normal", "--alerts", writeTempAlerts(t)} }, "--states"}, {"bad preexisting", true, func(t *testing.T) []string { @@ -110,9 +110,8 @@ func TestRunCheck_FlagValidation(t *testing.T) { } } -// TestRunCheck_ToInPastNoLog pins §4.2's refusal: a `to` already in the past -// with no recorded log cannot be classified from anything, because nothing -// ever observed the window. +// A `to` already in the past with no recorded log cannot be classified from +// anything, because nothing ever observed the window. func TestRunCheck_ToInPastNoLog(t *testing.T) { t.Setenv("GRAFANA_URL", "http://example.invalid") t.Setenv("GRAFANA_TOKEN", "test-token") @@ -126,13 +125,13 @@ func TestRunCheck_ToInPastNoLog(t *testing.T) { t.Fatalf("code = %d, want 2; stderr = %q", code, stderr.String()) } if !strings.Contains(stderr.String(), "already passed") { - t.Fatalf("stderr = %q, want the §4.2 refusal", stderr.String()) + t.Fatalf("stderr = %q, want the past-`to` refusal", stderr.String()) } } -// TestRunCheck_NoResultOnConfigError pins §20.2: --output json never writes -// to stdout when Check was never reached, because there is no Result to -// encode — only the table (on stderr) can report a configuration failure. +// --output json never writes to stdout when Check was never reached, because +// there is no Result to encode — only the table (on stderr) can report a +// configuration failure. func TestRunCheck_NoResultOnConfigError(t *testing.T) { t.Setenv("GRAFANA_URL", "") t.Setenv("GRAFANA_TOKEN", "") diff --git a/grafana-alertcheck/cmd/grafana-alertcheck/common.go b/grafana-alertcheck/cmd/grafana-alertcheck/common.go index e605c029b..3c56cd63f 100644 --- a/grafana-alertcheck/cmd/grafana-alertcheck/common.go +++ b/grafana-alertcheck/cmd/grafana-alertcheck/common.go @@ -12,11 +12,11 @@ import ( ) // commonFlags is registerCommon's result: the exactly three flags watch and -// check share (§20). Connection details are never flags (§20.2) and states / -// poll-interval are deliberately NOT here — states is check-only because -// recording is unfiltered (P6), and poll-interval is watch-only because check -// reads the cadence from the log header (P5). Putting either here would -// silently reinstate a knob this plan removed. +// check share. Connection details are never flags, and states / poll-interval +// are deliberately NOT here — states is check-only because recording is +// unfiltered, and poll-interval is watch-only because check reads the cadence +// from the log header. Putting either here would give both commands an opinion +// about a value only one of them may set. type commonFlags struct { folder *string concurrency *int @@ -25,13 +25,13 @@ type commonFlags struct { func registerCommon(fs *flag.FlagSet) *commonFlags { return &commonFlags{ - folder: fs.String("folder", "", "default folder to scope an unqualified alert name to (§17)"), + folder: fs.String("folder", "", "default folder to scope an unqualified alert name to"), concurrency: fs.Int("concurrency", 1, "maximum concurrent requests to Grafana"), alerts: fs.String("alerts", "", "path to a file of alert names, one per line, or - for stdin"), } } -// readAlerts reads §17's alert names, one per line, from a file or from +// readAlerts reads alert names, one per line, from a file or from // stdin when path is "-". An empty path is not an error here — watch and // check each decide for themselves whether an empty list is allowed // (log mode never wants one; single-step / record mode always does). @@ -62,16 +62,15 @@ func readAlerts(stdin io.Reader, path string) ([]string, error) { } // parseStates parses check's --states flag: a comma-separated list of the -// "bad" state vocabulary Config.States matches against (§13, classify.go's +// "bad" state vocabulary Config.States matches against (classify.go's // badStateSet). An empty string is not resolved here — it means "use the // library default of {firing}" — so this returns nil, nil for "" rather than // an error. // -// normal is deliberately NOT accepted: the v2 plan fixes this vocabulary to -// firing | pending | nodata | error (line 378) precisely because "normal" is -// the good state, never a bad one to classify against. Accepting it here -// would let --states normal turn every healthy instance into a violation and -// fail every healthy fleet — the exact fail-open shape H7 exists to prevent. +// normal is deliberately NOT accepted. The vocabulary is fixed to +// firing | pending | nodata | error precisely because "normal" is the good +// state, never a bad one to classify against: --states normal would turn every +// healthy instance into a violation and fail every healthy fleet. func parseStates(s string) ([]gate.State, error) { if strings.TrimSpace(s) == "" { return nil, nil @@ -95,7 +94,7 @@ func parseStates(s string) ([]gate.State, error) { return out, nil } -// parsePreexisting parses check's --preexisting flag (§11.7). +// parsePreexisting parses check's --preexisting flag. func parsePreexisting(s string) (gate.PreexistingPolicy, error) { switch gate.PreexistingPolicy(s) { case "": diff --git a/grafana-alertcheck/cmd/grafana-alertcheck/env.go b/grafana-alertcheck/cmd/grafana-alertcheck/env.go index e5702a6d1..02a125f1d 100644 --- a/grafana-alertcheck/cmd/grafana-alertcheck/env.go +++ b/grafana-alertcheck/cmd/grafana-alertcheck/env.go @@ -7,8 +7,7 @@ import ( // grafanaEnv reads the connection details from the environment only, never // from a flag — a flag value lands in the process argv and in CI logs, and -// the token must never be logged or otherwise surface in an error string -// (§20.2). +// the token must never be logged or otherwise surface in an error string. func grafanaEnv() (url, token string, err error) { url = os.Getenv("GRAFANA_URL") if url == "" { diff --git a/grafana-alertcheck/cmd/grafana-alertcheck/list.go b/grafana-alertcheck/cmd/grafana-alertcheck/list.go index 687e5dd9f..c1d0e8532 100644 --- a/grafana-alertcheck/cmd/grafana-alertcheck/list.go +++ b/grafana-alertcheck/cmd/grafana-alertcheck/list.go @@ -11,11 +11,10 @@ import ( ) // runList reads every rule definition from the ruler endpoint and prints one -// line per rule: its kind, its Folder/Group/Title, and its uid. This is what -// makes the gate runnable end to end before any coverage logic exists (§9 -// rule 4) — it validates auth, the ruler parse, and the shapes Resolve -// matches against, all against a real Grafana. It is also the "did you mean" -// surface §17.2's no-match error points operators at. +// line per rule: its kind, its Folder/Group/Title, and its uid. It validates +// auth, the ruler parse, and the shapes Resolve matches against, all against a +// real Grafana, and it is the surface Resolve's no-match error points operators +// at. func runList(args []string, stdout, stderr io.Writer) int { if len(args) != 0 { fmt.Fprintf(stderr, "list takes no arguments, got %v\n", args) @@ -34,7 +33,7 @@ func runList(args []string, stdout, stderr io.Writer) int { // always terminates. It can still take minutes end-to-end under repeated // transient failures (5 retries * up to 30s backoff each, per call) — an // acceptable wait for an interactive `list`, not for `watch`/`check`, - // which get their own deadlines from `--until`/`to` in P10. + // which get their own deadlines from `--until`/`--to`. src := gate.NewHTTPSource(url, token, gate.SystemClock{}) version, err := src.Version(context.Background()) if err != nil { diff --git a/grafana-alertcheck/cmd/grafana-alertcheck/main.go b/grafana-alertcheck/cmd/grafana-alertcheck/main.go index db5f93197..2672de1a3 100644 --- a/grafana-alertcheck/cmd/grafana-alertcheck/main.go +++ b/grafana-alertcheck/cmd/grafana-alertcheck/main.go @@ -1,5 +1,5 @@ -// Command grafana-alertcheck is the CLI entry point for the gate: `list` -// (P3), `watch` (record, P10) and `check` (classify, P10). +// Command grafana-alertcheck is the CLI entry point for the gate: `list`, +// `watch` (record) and `check` (classify). package main import ( @@ -16,9 +16,9 @@ const usage = "usage: grafana-alertcheck " // run is the whole of main's testable surface: parse the subcommand, dispatch, // return the process exit code. Exit codes below 2 (pass/violations) belong to -// `check` alone (§20.3, P10); every failure reachable from here — a missing -// subcommand, a bad flag, a transport or auth failure — is a could-not-check -// condition and maps to 2, never to 0 or 1 (H7). +// `check` alone; every failure reachable from here — a missing subcommand, a +// bad flag, a transport or auth failure — is a could-not-check condition and +// maps to 2, never to 0 or 1. // // Requested help (-h/--help) is not a failure — it is the one exception to // that rule. Convention (and every stdlib flag.FlagSet default) is exit 0 to diff --git a/grafana-alertcheck/cmd/grafana-alertcheck/table.go b/grafana-alertcheck/cmd/grafana-alertcheck/table.go index 0e4c93c67..11b72a074 100644 --- a/grafana-alertcheck/cmd/grafana-alertcheck/table.go +++ b/grafana-alertcheck/cmd/grafana-alertcheck/table.go @@ -10,27 +10,25 @@ import ( "github.com/smartcontractkit/chainlink-testing-framework/grafana-alertcheck/internal/gate" ) -// renderTable is §20.2's required human table. It always writes to the -// writer it is given, which the caller (runCheck) always points at -// stderr — the human table is not the machine output §20.2 reserves stdout -// for. +// renderTable is the human table. It always writes to the writer it is given, +// which the caller (runCheck) always points at stderr — stdout is reserved for +// the machine-readable --output json. // // Three sections, in order: // // 1. one line per rule: outcome, BadFor, pollEvery, proved-or-not with the // largest gap; -// 2. one line per Violation (R2): a rule's worst-of outcome does not carry -// the State/Health of the instance that actually caused it — Violation -// does — so this is also where those two columns appear, sorted after -// the rule table rather than folded into it, and it is the only place an -// operator running WITHOUT --output json sees the §12.1 --allow-paused -// hint that Violation.Note already carries (classify.go); -// 3. a footer with the per-rule thresholds and the run-wide numbers §20.2 -// says are the answer to "why" on exit 2: each non-skipped rule's -// maxGap/healthGrace/evalStaleAfter, the global transitionGrace and +// 2. one line per Violation: a rule's worst-of outcome does not carry the +// State/Health of the instance that actually caused it — Violation does — +// so this is also where those two columns appear, sorted after the rule +// table rather than folded into it, and it is the only place an operator +// running WITHOUT --output json sees the --allow-paused hint that +// Violation.Note already carries (classify.go); +// 3. a footer with the numbers that answer "why" on exit 2: each non-skipped +// rule's maxGap/healthGrace/evalStaleAfter, the global transitionGrace and // drainTimeout, and the largest measured clock skew alongside its own -// error bound (RTT/2) — SkewHardLimit is a separate, fixed input -// threshold and is reported next to it, never as if it were that bound. +// error bound (RTT/2) — SkewHardLimit is a separate, fixed input threshold +// and is reported next to it, never as if it were that bound. func renderTable(w io.Writer, res gate.Result) error { alertOf := make(map[string]string, len(res.Verdicts)) for _, v := range res.Verdicts { @@ -77,7 +75,7 @@ func renderTable(w io.Writer, res gate.Result) error { // provedLabel is the table's PROVED column: "yes" for a clean coverage // proof, "no" with the reason and largest gap for an unobservable rule, and // "-" for a rule decide never asked proveCoverage about at all (skipped — -// paused before the window opened, §12). +// paused before the window opened). func provedLabel(cov gate.CoverageResult) string { if cov.Reason == "" && !cov.Unobservable && !cov.Proved { return "-" diff --git a/grafana-alertcheck/cmd/grafana-alertcheck/table_test.go b/grafana-alertcheck/cmd/grafana-alertcheck/table_test.go index 106760fbb..e585e9b61 100644 --- a/grafana-alertcheck/cmd/grafana-alertcheck/table_test.go +++ b/grafana-alertcheck/cmd/grafana-alertcheck/table_test.go @@ -9,10 +9,9 @@ import ( "github.com/smartcontractkit/chainlink-testing-framework/grafana-alertcheck/internal/gate" ) -// TestRenderTable is the golden table test: a fixed Result renders a -// deterministic, ordered rule table, a violations section (R2) and a footer -// carrying the per-rule and global thresholds plus the skew and its bound -// (R3) — with no live Check involved. +// The golden table test: a fixed Result renders a deterministic, ordered rule +// table, a violations section and a footer carrying the per-rule and global +// thresholds plus the skew and its bound — with no live Check involved. func TestRenderTable(t *testing.T) { gapAt := time.Date(2026, 1, 1, 12, 0, 0, 0, time.UTC) res := gate.Result{ @@ -64,13 +63,13 @@ func TestRenderTable(t *testing.T) { t.Fatalf("out = %q, want Zebra's clean row", out) } - // Violations section (R2): must show up even without --output json, and - // must carry the §12.1 --allow-paused hint text verbatim. + // The violations section must show up even without --output json, and must + // carry the --allow-paused hint text verbatim. if !strings.Contains(out, "VIOLATIONS") { t.Fatalf("out = %q, want a VIOLATIONS section", out) } if !strings.Contains(out, "--allow-paused") { - t.Fatalf("out = %q, want the §12.1 --allow-paused hint in the human table", out) + t.Fatalf("out = %q, want the --allow-paused hint in the human table", out) } if !strings.Contains(out, "STATE") || !strings.Contains(out, "HEALTH") { t.Fatalf("out = %q, want the violations table to have STATE and HEALTH columns", out) @@ -79,8 +78,8 @@ func TestRenderTable(t *testing.T) { t.Fatalf("out = %q, want Ape's violation State/Health", out) } - // Footer (R3): per-rule thresholds, global thresholds, and skew with its - // own bound rather than the fixed hard limit. + // The footer: per-rule thresholds, global thresholds, and skew with its own + // bound rather than the fixed hard limit. if !strings.Contains(out, "Ape Alert: maxGap=1m0s healthGrace=2m0s evalStaleAfter=1m0s") { t.Fatalf("out = %q, want Ape's per-rule thresholds", out) } @@ -88,7 +87,7 @@ func TestRenderTable(t *testing.T) { t.Fatalf("out = %q, want Zebra's per-rule thresholds", out) } if strings.Contains(out, "Paused Alert: maxGap") { - t.Fatalf("out = %q, a skipped rule must not report thresholds it never had (§12)", out) + t.Fatalf("out = %q, a skipped rule must not report thresholds it never had", out) } if !strings.Contains(out, "global: transitionGrace=5m0s (source: Ape Alert (for=5m)) drainTimeout=2m0s") { t.Fatalf("out = %q, want the global thresholds line", out) @@ -104,9 +103,9 @@ func TestRenderTable(t *testing.T) { } } -// TestProvedLabel_Skipped pins the "-" case: a rule decide never asked -// proveCoverage about (paused before the window opened, §12) has an empty -// CoverageResult and must not be reported as either proved or unobservable. +// The "-" case: a rule decide never asked proveCoverage about (paused before +// the window opened) has an empty CoverageResult and must not be reported as +// either proved or unobservable. func TestProvedLabel_Skipped(t *testing.T) { if got := provedLabel(gate.CoverageResult{}); got != "-" { t.Fatalf("provedLabel(zero value) = %q, want \"-\"", got) diff --git a/grafana-alertcheck/cmd/grafana-alertcheck/watch.go b/grafana-alertcheck/cmd/grafana-alertcheck/watch.go index f3ad57d92..df2c50029 100644 --- a/grafana-alertcheck/cmd/grafana-alertcheck/watch.go +++ b/grafana-alertcheck/cmd/grafana-alertcheck/watch.go @@ -15,19 +15,17 @@ const watchUsage = "usage: grafana-alertcheck watch --out [--pidfile F] [ // runWatch is the record step's entire CLI surface, split in two by one flag // set — gate.DaemonChildFlag ("--daemon-child") and gate.ReadyFDFlag -// ("--ready-fd") select which side of P6's parent/child split this -// invocation is: +// ("--ready-fd") select which side of the parent/child split this invocation +// is: // // - without them: the record command an operator types. It parses --out, // --alerts and the rest, builds a gate.WatchConfig and calls gate.Watch, // which resolves, records the first observation of every rule, and -// detaches the recorder before returning (§4.3). +// detaches the recorder before returning. // - with them: the detached recorder itself. gate.Watch's own childArgs // (watch_unix.go) is the only thing that ever sets them — an operator -// never types "--daemon-child" and it does not appear in watchUsage — -// and this dispatches straight to gate.RunDaemonChild. P6's integration -// test already covers the spawn; this is the one new test P10 owns: that -// seeing the flag reaches RunDaemonChild. +// never types "--daemon-child" and it does not appear in watchUsage — and +// this dispatches straight to gate.RunDaemonChild. // // Both flags live in the SAME flag set as the operator-facing ones rather // than a second, hidden set: the child is started with childArgs' exact @@ -106,11 +104,10 @@ func runWatch(args []string, stdin io.Reader, stdout, stderr io.Writer) int { return 0 } -// runDaemonChild is the detached recorder's whole entry point (P6's -// obligation on this phase). Its stdout and stderr are already the daemon -// log file — spawnChild (watch_unix.go) redirects both before Start — so -// writing to stderr here lands exactly where waitForChildReady's failure -// path quotes from. +// runDaemonChild is the detached recorder's whole entry point. Its stdout and +// stderr are already the daemon log file — spawnChild (watch_unix.go) +// redirects both before Start — so writing to stderr here lands exactly where +// waitForChildReady's failure path quotes from. func runDaemonChild(out, until string, concurrency, readyFD int, stderr io.Writer) int { url, token, err := grafanaEnv() if err != nil { diff --git a/grafana-alertcheck/cmd/grafana-alertcheck/watch_test.go b/grafana-alertcheck/cmd/grafana-alertcheck/watch_test.go index 267a312a2..6d8c8c1a6 100644 --- a/grafana-alertcheck/cmd/grafana-alertcheck/watch_test.go +++ b/grafana-alertcheck/cmd/grafana-alertcheck/watch_test.go @@ -7,9 +7,8 @@ import ( "testing" ) -// TestRunWatch_FlagValidation is the record step's flag-validation matrix. -// Every case fails inside gate.WatchConfig.validate() (P6) or before it, so -// none needs a reachable Grafana. +// The record step's flag-validation matrix. Every case fails inside +// gate.WatchConfig.validate() or before it, so none needs a reachable Grafana. func TestRunWatch_FlagValidation(t *testing.T) { tests := []struct { name string @@ -58,9 +57,8 @@ func TestRunWatch_FlagValidation(t *testing.T) { } } -// TestRunWatch_DaemonChildDispatch pins P6's obligation on this phase: seeing -// gate.DaemonChildFlag must dispatch to gate.RunDaemonChild, and the flag -// must never appear in watchUsage (an operator never types it). +// Seeing gate.DaemonChildFlag must dispatch to gate.RunDaemonChild, and the +// flag must never appear in watchUsage (an operator never types it). func TestRunWatch_DaemonChildDispatch(t *testing.T) { t.Setenv("GRAFANA_URL", "http://example.invalid") t.Setenv("GRAFANA_TOKEN", "test-token") diff --git a/grafana-alertcheck/internal/gate/check.go b/grafana-alertcheck/internal/gate/check.go index 710bedef9..0f21b4d86 100644 --- a/grafana-alertcheck/internal/gate/check.go +++ b/grafana-alertcheck/internal/gate/check.go @@ -11,46 +11,29 @@ import ( "time" ) -// Obligations this phase leaves for P10, carried forward the way P6 and P7 -// carried theirs so a later review has something concrete to check against: +// Check returns (Result, error) and no exit code: the code is a presentation +// decision the CLI makes. err != nil is exit 2 unconditionally, even alongside +// real violations; violations with err == nil is exit 1; neither is exit 0. // -// - Exit codes are the CLI's (§20.3, §9.1, H6/H7). Check returns -// (Result, error) and nothing else: err != nil is exit 2 unconditionally, -// never 0 and never 1, even alongside real violations; len(Violations) > 0 -// with err == nil is exit 1; both empty is exit 0. Check deliberately does -// not return a code, because a code is a presentation decision and the -// library must not make it. -// - §12.1 wants the paused rule AND --allow-paused both named to the user. -// decide names both in the shortfall Violation's Note (classify.go), and -// P10's renderer prints every Violation, including Note, in the human -// table — not only in --output json. -// - Config.Notes carries the running commentary (§13.2's planned run time, -// the countdown, the blind-interval warning). §20.2 puts the human output -// on stderr and reserves stdout for --output json, so the CLI must pass -// stderr here. -// - Check never reads the environment. GRAFANA_URL and GRAFANA_TOKEN are -// read by the CLI and passed in as fields, and the token must never reach -// a *flag.FlagSet (§20.2). +// Check never reads the environment either. The URL and token are read by the +// CLI and passed in as fields, and the token must never reach a *flag.FlagSet. // countdownEvery is how often the collection loop reports what it is waiting -// for (§13.2: "then print a countdown at regular intervals"). A silent wait is -// indistinguishable from a hung process, and the wait after `to` is the -// longest silence in the whole run. +// for. A silent wait is indistinguishable from a hung process, and the wait +// after `to` is the longest silence in the whole run. const countdownEvery = 30 * time.Second -// recorderStopTimeout bounds §4.4 step 3, the wait for the recorder's exit. -// Not in the source plan's table of values — a judgment call, on the same -// reasoning as childReadyTimeout (P6): everything the recorder does after -// SIGTERM is local (finish the in-flight write, append the sentinel, fsync) -// and an in-flight poll aborts through the child's own context, so the real -// figure is milliseconds. Loose enough for an overloaded runner, and a -// timeout is a hard error rather than a longer wait — a log a writer may -// still hold cannot be read at all (§4.4 step 4). +// recorderStopTimeout bounds the wait for the recorder's exit. Everything the +// recorder does after SIGTERM is local (finish the in-flight write, append the +// sentinel, fsync) and an in-flight poll aborts through the child's own +// context, so the real figure is milliseconds; this is loose enough for an +// overloaded runner. The timeout is a hard error rather than a longer wait — a +// log a writer may still hold cannot be read at all. const recorderStopTimeout = 30 * time.Second // recorderStopPoll is how often that wait re-checks the pid. There is no // wait(2) available: the recorder is a detached session leader, not this -// process's child (P6), so its exit can only be observed by polling. +// process's child, so its exit can only be observed by polling. const recorderStopPoll = 100 * time.Millisecond // Config is check's whole input. It is the CLI's view of a run, and it is @@ -59,14 +42,13 @@ const recorderStopPoll = 100 * time.Millisecond // never cross that line. type Config struct { // URL and Token are the connection details, read from the environment by - // the CLI and never registered as flags (§20.2). Token never enters the - // pure layer, an error string, or a Result. + // the CLI and never registered as flags. Token never enters the pure layer, + // an error string, or a Result. URL, Token string // Alerts is REQUIRED in single-step mode and must be EMPTY in log mode: - // with a log, the header IS the alert set (§19.1 step 3), and there is - // nothing to compare a second list against. Both directions are encoded, - // resolving the source plan's §19.1 step 1 / step 3 contradiction. + // with a log, the header IS the alert set, and there is nothing to compare + // a second list against. Alerts []string Folder string @@ -76,17 +58,16 @@ type Config struct { AllowPaused bool NodataIsUnobservable bool - // From is the moment the deploy finished and To is the end of the work - // (§7). They are different moments and both come from the work. In - // recorder mode an absent From is a hard error; in single-step mode it - // falls back to the start of this step, with the blind-interval warning - // §4.2 requires. + // From is the moment the deploy finished and To is the end of the work. + // They are different moments and both come from the work. In recorder mode + // an absent From is a hard error; in single-step mode it falls back to the + // start of this step, with a blind-interval warning. From, To time.Time // Log is the path of a recording made by watch; "" selects single-step // mode. PidFile defaults to .pid, the convention watch's parent - // writes (P6) and the only way check can reach the recorder it must stop - // before it may read the log (§4.4 steps 1-4). + // writes and the only way check can reach the recorder it must stop before + // it may read the log. Log string PidFile string @@ -94,16 +75,15 @@ type Config struct { // --poll-interval flag. In log mode the cadence comes from the header — // the cadence the recording actually used — and a second authority would // let an operator silently widen maxGap over evidence that was recorded at - // a different rate (P5, "two authorities"); in single-step mode the same - // process records and classifies, so §5's default is the only cadence - // there is. + // a different rate; in single-step mode the same process records and + // classifies, so the default cadence is the only cadence there is. Concurrency int Clock Clock // Notes is where the shell prints what an operator has to see while the // run is in progress: the planned run time, the grace and its source, the // countdown, the blind-interval warning. nil discards them. The library - // renders no table — the CLI owns presentation (§20.2). + // renders no table — the CLI owns presentation. Notes io.Writer } @@ -123,7 +103,7 @@ func (cfg Config) withDefaults() Config { return cfg } -// namedAlerts returns the alert names that survive §17.3's trim-and-discard, +// namedAlerts returns the alert names that survive Resolve's trim-and-discard, // so validation counts what Resolve will actually see rather than what the // caller happened to pass (a file ending in a newline yields an empty line). func (cfg Config) namedAlerts() []string { @@ -139,12 +119,12 @@ func (cfg Config) namedAlerts() []string { // Check is the I/O shell: HTTP, signals, the pidfile, file reads, the // countdown print. Every correctness question it touches is answered // elsewhere — by proveCoverage and decide, which are pure — and that split is -// the most important seam in the project (§2). Check therefore needs two +// the most important seam in the project. Check therefore needs two // integration tests; decide carries the suite. // -// H7 governs the return: a pass is exactly len(Violations) == 0 && err == nil. -// Every error path below leaves err non-nil, and no path anywhere in this file -// converts an error into an empty Result with a nil error. +// A pass is exactly len(Violations) == 0 && err == nil. Every error path below +// leaves err non-nil, and no path anywhere in this file converts an error into +// an empty Result with a nil error. func Check(ctx context.Context, cfg Config) (Result, error) { cfg = cfg.withDefaults() if err := cfg.validate(); err != nil { @@ -152,32 +132,31 @@ func Check(ctx context.Context, cfg Config) (Result, error) { } // The Source is built here and injected into check() so every behaviour // below is testable against a scripted fake — the same seam prepareWatch - // uses (P6), and the reason this file needs no test-only setter. + // uses, and the reason this file needs no test-only setter. return check(ctx, cfg, NewHTTPSource(cfg.URL, cfg.Token, cfg.Clock)) } -// validate is §19.1 step 1. It runs before any network call, so a -// configuration mistake costs nothing and, more importantly, is never -// discovered after a ten-minute wait. +// validate runs before any network call, so a configuration mistake costs +// nothing and, more importantly, is never discovered after a ten-minute wait. func (cfg Config) validate() error { if cfg.URL == "" { return errors.New("check: no grafana url") } if cfg.To.IsZero() { - return errors.New("check: no `to`: the end of the window is required (§7)") + return errors.New("check: no `to`: the end of the window is required") } named := cfg.namedAlerts() if cfg.Log == "" { - // §19.1 step 1: an empty Alerts is an error — but only without a log. + // An empty Alerts is an error — but only without a log. if len(named) == 0 { return errors.New("check: no alert names given and no recorded log to take them from") } } else if len(named) > 0 { - // §19.1 step 3, the other direction: the alert set comes from the log. - // Accepting both would mean reconciling two sets, which is the subset - // arithmetic the source plan removes by making the log the one source. - return fmt.Errorf("check: --alerts is refused with a recorded log: %s already names the alert set it recorded (§19.1 step 3)", cfg.Log) + // The other direction: with a log, the alert set comes from the log. + // Accepting both would mean reconciling two sets, which the log being + // the one source removes entirely. + return fmt.Errorf("check: --alerts is refused with a recorded log: %s already names the alert set it recorded", cfg.Log) } now := cfg.Clock.Now() @@ -188,13 +167,13 @@ func (cfg Config) validate() error { from := cfg.From switch { case from.IsZero() && cfg.Log != "": - // §7, and never a warning-and-continue: falling back to the start of - // the check step reinstates exactly the blind interval the recorder - // exists to remove, which is the fail-open shape this design refuses. - return errors.New("check: no `from` in recorder mode: the deploy step must emit a completion timestamp (§7)") + // Never a warning-and-continue: falling back to the start of the check + // step reinstates exactly the blind interval the recorder exists to + // remove, which is the fail-open shape this design refuses. + return errors.New("check: no `from` in recorder mode: the deploy step must emit a completion timestamp") case from.IsZero(): // Single-step only. The caller sees the resulting blind interval named - // exactly, once the first observation has fixed its end (§4.2). + // exactly, once the first observation has fixed its end. from = now } @@ -202,40 +181,39 @@ func (cfg Config) validate() error { return fmt.Errorf("check: `to` %s is before `from` %s", cfg.To.Format(time.RFC3339), from.Format(time.RFC3339)) } if from.After(now.Add(fromFutureTolerance)) { - return fmt.Errorf("check: `from` %s is more than %s ahead of this runner's clock %s (§7)", + return fmt.Errorf("check: `from` %s is more than %s ahead of this runner's clock %s", from.Format(time.RFC3339), fromFutureTolerance, now.Format(time.RFC3339)) } - // A `to` already in the past is not a special mode WITH a log (§7, §24.3): - // the collection loop's condition is simply already true and the evidence - // is classified immediately. Without one it is a different thing entirely - // — a request to prove a window that nothing observed. Refusing it is not + // A `to` already in the past is not a special mode WITH a log: the + // collection loop's condition is simply already true and the evidence is + // classified immediately. Without one it is a different thing entirely — a + // request to prove a window that nothing observed. Refusing it is not // pedantry: the coverage window would end before the first observation, // every heartbeat gap inside it would measure negative, and the run would // report a proved window it never saw. if cfg.Log == "" && !cfg.To.After(now) { - return fmt.Errorf("check: `to` %s has already passed and there is no recorded log: a window that ended before check started can only be classified from a recording (§4.2)", + return fmt.Errorf("check: `to` %s has already passed and there is no recorded log: a window that ended before check started can only be classified from a recording", cfg.To.Format(time.RFC3339)) } return nil } -// check is Check with the Source injected. Its body is §19.1 steps 1-9, one -// commented block each and in that order, so a review can diff it against the -// source plan line by line. +// check is Check with the Source injected, and its body is one commented block +// per stage of a run, in the order a run performs them. func check(ctx context.Context, cfg Config, src Source) (Result, error) { - // ---- §19.1 step 1 — validate the configuration. ----------------------- + // ---- Validate the configuration. -------------------------------------- // Done by Check before this function is reached, except for the one part // that needs a clock reading kept for later: the single-step fallback for // an absent `from`. from := cfg.From if from.IsZero() { from = cfg.Clock.Now() - fmt.Fprintf(cfg.Notes, "note: no `from` given; the window starts at the start of this step, %s (§4.2)\n", + fmt.Fprintf(cfg.Notes, "note: no `from` given; the window starts at the start of this step, %s\n", from.Format(time.RFC3339)) } - // ---- §19.1 step 2 — resolve the definitions from the ruler API. ------- + // ---- Resolve the definitions from the ruler API. ---------------------- // Unconditional, in BOTH modes. A log's header supplies the alert set as // UIDs and the recording facts, never the rule facts: `for`, // intervalSeconds and Kind always come from a fresh ruler read, which is @@ -249,17 +227,16 @@ func check(ctx context.Context, cfg Config, src Source) (Result, error) { } allDefs, err := src.Definitions(ctx) if err != nil { - // §19.3 case 2: resolution of the definitions failed. return Result{}, fmt.Errorf("read rule definitions: %w", err) } - // ---- §19.1 step 3 — with a log, validate its identity. ---------------- + // ---- With a log, validate its identity. ------------------------------- // The header is read early — line 1 only, the one line a writer can never // change (ReadLogHeader) — so a wrong URL or a rule that no longer // resolves fails closed NOW rather than after the whole window has // elapsed. It is advisory: the authoritative header comes from the single - // full ReadLog in step 6, after the writer has exited, and the identity is - // validated again against that one. + // full ReadLog once collection is over and the writer has exited, and the + // identity is validated again against that one. var ( resolved []Definition notes []string @@ -272,7 +249,6 @@ func check(ctx context.Context, cfg Config, src Source) (Result, error) { if cfg.Log != "" { earlyHdr, err = ReadLogHeader(cfg.Log) if err != nil { - // §19.3 case 3. return Result{}, fmt.Errorf("log identity: %w", err) } logHasHdr = true @@ -293,12 +269,12 @@ func check(ctx context.Context, cfg Config, src Source) (Result, error) { fmt.Fprintf(cfg.Notes, "note: %s\n", n) } - // ---- §19.1 step 4 — derive the timings, print the plan, fit the budget. + // ---- Derive the timings, print them, fit the request budget. ---------- if logHasHdr { // The header is the authority for the cadence actually recorded at; // re-deriving it from defs would compare gaps recorded at an override // cadence against thresholds computed from the default — fail-open in - // the faster-override direction (P5). + // the faster-override direction. rt, gt, err = DeriveTimingsFromLog(earlyHdr, resolved) if err != nil { return Result{}, fmt.Errorf("log identity: %w", err) @@ -316,10 +292,10 @@ func check(ctx context.Context, cfg Config, src Source) (Result, error) { } // The measurement pass and the budget check belong to single-step mode - // alone (§5.2): in recorder mode watch already took one observation of - // every rule and checked the budget against those measured latencies - // before it detached, and repeating it here would spend a second poll of - // every rule to re-answer a question already answered. + // alone: in recorder mode watch already took one observation of every rule + // and checked the budget against those measured latencies before it + // detached, and repeating it here would spend a second poll of every rule + // to re-answer a question already answered. var ( header Header initial []Poll @@ -328,8 +304,8 @@ func check(ctx context.Context, cfg Config, src Source) (Result, error) { if !logHasHdr { // StartedAt is fixed before the pass rather than after it, so the // interval it claims to have observed can only be wider than the one - // it really saw — and the first heartbeat's own boundary gap (P7 check - // 3) is what proves that interval, not this timestamp. + // it really saw — and the first heartbeat's own boundary gap is what + // proves that interval, not this timestamp. startedAt := cfg.Clock.Now() active := activeRules(resolved) var measured map[string]time.Duration @@ -341,10 +317,10 @@ func check(ctx context.Context, cfg Config, src Source) (Result, error) { return Result{}, err } - // Single-step synthesis — how the pure layer stays unconditional (§2). - // The shell builds the Header and later stamps the sentinel itself, so - // P7 checks 1 and 2 run exactly as they do over a recording and no - // mode flag ever reaches proveCoverage or decide. + // Single-step synthesis — how the pure layer stays unconditional. The + // shell builds the Header and later stamps the sentinel itself, so the + // sentinel and from-bounds coverage checks run exactly as they do over + // a recording and no mode flag ever reaches proveCoverage or decide. header = Header{ SchemaVersion: LogSchemaVersion, URL: cfg.URL, @@ -353,31 +329,31 @@ func check(ctx context.Context, cfg Config, src Source) (Result, error) { Rules: loggedRules(resolved, rt), } if from.Before(startedAt) { - // §4.2/§22.4's declared blind interval: in single-step mode this - // is a warning and a pass, and ONLY here. Recorder mode keeps P7 - // check 2 strict (§22.9), because there the recorder was supposed - // to be watching and the gap means it was not. - fmt.Fprintf(cfg.Notes, "warning: cannot see [%s, %s) — %s before the first observation; the window is classified from %s (§4.2)\n", + // The declared blind interval: in single-step mode this is a + // warning and a pass, and ONLY here. Recorder mode keeps the + // from-bounds coverage check strict, because there the recorder + // was supposed to be watching and the gap means it was not. + fmt.Fprintf(cfg.Notes, "warning: cannot see [%s, %s) — %s before the first observation; the window is classified from %s\n", from.Format(time.RFC3339), startedAt.Format(time.RFC3339), startedAt.Sub(from).Round(time.Second), startedAt.Format(time.RFC3339)) from = startedAt } } - // ---- §19.1 step 5 — apply MinObserved. -------------------------------- - // Its default is the resolved rule count AFTER the collapse (§17.3), which - // is len(resolved) by construction. decide defaults it identically; it is - // resolved here as well so the value the run will judge against is printed - // before the wait rather than inferred from the verdict afterwards. + // ---- Apply MinObserved. ----------------------------------------------- + // Its default is the resolved rule count AFTER duplicate names collapse, + // which is len(resolved) by construction. decide defaults it identically; + // it is resolved here as well so the value the run will judge against is + // printed before the wait rather than inferred from the verdict afterwards. minObserved := cfg.MinObserved if minObserved == 0 { minObserved = len(resolved) } fmt.Fprintf(cfg.Notes, "min-observed: %d of %d resolved rule(s)\n", minObserved, len(resolved)) - // ---- §19.1 step 6 — collect the evidence. ----------------------------- + // ---- Collect the evidence. -------------------------------------------- // Collect ONLY. No classification happens here and there is no early exit, - // even once a violation is certain (H5, §19.2): the loop always runs to + // even once a violation is certain: the loop always runs to // to + transitionGrace, which is what makes "did the early exit lose the // coverage proof?" a question that cannot be asked. windowEnd := cfg.To.Add(gt.transitionGrace) @@ -388,11 +364,10 @@ func check(ctx context.Context, cfg Config, src Source) (Result, error) { } collected, err := collectUntil(ctx, cfg, windowEnd, poller) if err != nil { - // §19.3 case 1: the failure limit was exceeded (retryTransport already - // gave every transient failure its backoff), or the context ended. - // Nothing collected is classified — the count is there so an operator - // can tell a run that failed at once from one that failed at minute - // nine. + // The failure limit was exceeded (retryTransport already gave every + // transient failure its backoff), or the context ended. Nothing + // collected is classified — the count is there so an operator can tell + // a run that failed at once from one that failed at minute nine. return Result{}, fmt.Errorf("collect evidence after %d poll(s): %w", len(collected), err) } @@ -401,10 +376,10 @@ func check(ctx context.Context, cfg Config, src Source) (Result, error) { sentinel *time.Time ) if logHasHdr { - // §4.4 steps 2-4, in this order and no other: signal the writer, wait - // for its exit, and only THEN read the log once. A log read while a - // writer can still append can only yield a shorter window than the one - // that was actually recorded. + // In this order and no other: signal the writer, wait for its exit, + // and only THEN read the log once. A log read while a writer can still + // append can only yield a shorter window than the one that was + // actually recorded. heldLog, err := stopRecorder(ctx, cfg) if err != nil { return Result{}, err @@ -442,23 +417,23 @@ func check(ctx context.Context, cfg Config, src Source) (Result, error) { polls = append(polls, collected...) // The shell stamps the sentinel itself, when the collection loop // exits: by construction that is at or after to + transitionGrace, so - // P7 check 1 passes for the same reason a clean recorder stop does, - // and for no other. + // the sentinel check passes for the same reason a clean recorder stop + // does, and for no other. stoppedAt := cfg.Clock.Now() sentinel = &stoppedAt } - // ---- §19.1 step 7 — the drain wait. ----------------------------------- - // The last instance of the liveness check (§14.6): did this rule evaluate - // through the end of the window? It is I/O and it is deliberately NOT part - // of proveCoverage — adding it there would put HTTP inside the pure layer - // and destroy the seam §2 depends on. + // ---- The drain wait. -------------------------------------------------- + // The last instance of the liveness check: did this rule evaluate through + // the end of the window? It is I/O and it is deliberately NOT part of + // proveCoverage — adding it there would put HTTP inside the pure layer and + // destroy the seam this design depends on. drained, err := drainWait(ctx, cfg, src, resolved, header.pausedAtStart(), rt, polls, windowEnd, gt.drainTimeout) if err != nil { return Result{}, err } - // ---- §19.1 step 8 — classify. ----------------------------------------- + // ---- Classify. -------------------------------------------------------- pol := Policy{ States: cfg.States, Preexisting: cfg.Preexisting, @@ -471,33 +446,34 @@ func check(ctx context.Context, cfg Config, src Source) (Result, error) { result, decideErr := decide(header, polls, sentinel, resolved, rt, gt, pol) result, drainErr := mergeDrainTimeouts(result, drained) - // ---- §19.1 step 9 — return Result. ------------------------------------ + // ---- Return the Result. ----------------------------------------------- // Both errors are joined rather than one shadowing the other: each names // rules the other does not, and on exit 2 that list IS the answer to - // "why". H7 needs only that err be non-nil when either fired. + // "why". return result, errors.Join(decideErr, drainErr) } -// resolveFromLog turns a log header into the resolved definitions, and is -// §19.1 step 3's identity check in practice. Three things are verified: the -// URL matches, the schema version matches (ReadLog/ReadLogHeader own that), -// and every header UID still resolves against the fresh ruler read. The alert -// set is TAKEN from the log, never compared — with Alerts required empty in -// log mode there is nothing to compare it against, and §22.4's "different -// alert set" refusal is exactly this URL-and-UID failure. +// resolveFromLog turns a log header into the resolved definitions, and is the +// log's identity check in practice. Three things are verified: the URL +// matches, the schema version matches (ReadLog/ReadLogHeader own that), and +// every header UID still resolves against the fresh ruler read. The alert set +// is TAKEN from the log, never compared — with Alerts required empty in log +// mode there is nothing to compare it against, and refusing a log recorded +// against a different alert set is exactly this URL-and-UID failure. // // Resolving through Resolve, by uid:, rather than by a private lookup, keeps -// one implementation of §17: a header naming a recording or datasource-managed -// rule gets the same specific refusal an operator would, and a header naming -// the same UID twice collapses with a note (DeriveTimingsFromLog rejects that -// case outright, so the note is belt and braces). +// one implementation of the resolution rules: a header naming a recording or +// datasource-managed rule gets the same specific refusal an operator would, +// and a header naming the same UID twice collapses with a note +// (DeriveTimingsFromLog rejects that case outright, so the note is belt and +// braces). // // Only the header-to-defs direction needs checking. The opposite direction // cannot fail here: resolved is BUILT from the header, so no resolved // definition can be absent from it. func resolveFromLog(allDefs []Definition, h Header, cfg Config) ([]Definition, []string, error) { if h.URL != cfg.URL { - return nil, nil, fmt.Errorf("log identity: %s recorded url %q but this run is configured for %q (§19.1 step 3)", + return nil, nil, fmt.Errorf("log identity: %s recorded url %q but this run is configured for %q", cfg.Log, h.URL, cfg.URL) } names := make([]string, 0, len(h.Rules)) @@ -506,16 +482,15 @@ func resolveFromLog(allDefs []Definition, h Header, cfg Config) ([]Definition, [ } resolved, notes, err := Resolve(allDefs, names, "") if err != nil { - return nil, nil, fmt.Errorf("log identity: %s names a rule that no longer resolves: %w (§19.1 step 3)", cfg.Log, err) + return nil, nil, fmt.Errorf("log identity: %s names a rule that no longer resolves: %w", cfg.Log, err) } return resolved, notes, nil } -// activeRules drops the rules whose DEFINITION says paused. They are skipped -// (§12): never polled, never waited for, and reported from the definitions -// alone — a skipped rule has no poll records at all, so it has no heartbeats -// to prove and no IsPaused poll to detect (P6 deviation 4, and the obligation -// it left P7/P8). +// activeRules drops the rules whose DEFINITION says paused. They are skipped: +// never polled, never waited for, and reported from the definitions alone — a +// skipped rule has no poll records at all, so it has no heartbeats to prove +// and no IsPaused poll to detect. func activeRules(defs []Definition) []Definition { out := make([]Definition, 0, len(defs)) for _, d := range defs { @@ -527,8 +502,9 @@ func activeRules(defs []Definition) []Definition { } // activeTimingsOf narrows the timings map to the rules that will actually be -// polled, which is what §5.2's budget is spent on: a skipped rule consumes -// none of the capacity, so counting it would refuse schedules that fit. +// polled, which is what the request budget is spent on: a skipped rule +// consumes none of the capacity, so counting it would refuse schedules that +// fit. func activeTimingsOf(active []Definition, rt map[string]ruleTimings) map[string]ruleTimings { out := make(map[string]ruleTimings, len(active)) for _, d := range active { @@ -538,14 +514,14 @@ func activeTimingsOf(active []Definition, rt map[string]ruleTimings) map[string] } // livePoller is single-step mode's collection engine: the same per-rule -// scheduler and the same Reducer the recorder uses (P4/P6), writing into -// memory instead of a log. Log mode has none — the recorder is doing this -// work in another process — and collectUntil takes a nil poller for it. +// scheduler and the same Reducer the recorder uses, writing into memory +// instead of a log. Log mode has none — the recorder is doing this work in +// another process — and collectUntil takes a nil poller for it. type livePoller struct { src Source reducer *Reducer sched *Scheduler - titles map[string]string // uid -> title: poll by title, select by UID (§14.5) + titles map[string]string // uid -> title: poll by title, select by UID concurrency int } @@ -573,10 +549,10 @@ func newLivePoller(src Source, reducer *Reducer, active []Definition, rt map[str // The successes are NOT kept for the reason watchLoopConfig.pollBatch keeps // its own: those go into a durable log that a later check will read, so // dropping one would turn a single rule's transport failure into a coverage -// gap for the others. Here there is no later reader. A terminal failure -// during collection is exit 2 (§19.3 case 1) and check discards the whole -// collection, so these come back only to let the error say how far the run -// got before it stopped — which is the one part of it an operator can act on. +// gap for the others. Here there is no later reader. A terminal failure during +// collection is exit 2 and check discards the whole collection, so these come +// back only to let the error say how far the run got before it stopped — +// which is the one part of it an operator can act on. func (p *livePoller) poll(ctx context.Context, uids []string) ([]Poll, error) { observed, obsErr := observeAll(ctx, p.src, p.titles, uids, p.concurrency) out := make([]Poll, 0, len(uids)) @@ -590,13 +566,13 @@ func (p *livePoller) poll(ctx context.Context, uids []string) ([]Poll, error) { return out, obsErr } -// collectUntil is §19.1 step 6's loop, shared by both modes. With a poller it +// collectUntil is the collection loop, shared by both modes. With a poller it // polls each rule on its own cadence; with nil it only waits, because in // recorder mode the evidence is being written by another process. Both print // the same countdown, because both are the same silence to an operator -// watching a job (§13.2). +// watching a job. // -// It never classifies and never exits early (H5). +// It never classifies and never exits early. func collectUntil(ctx context.Context, cfg Config, deadline time.Time, p *livePoller) ([]Poll, error) { var ( polls []Poll @@ -643,9 +619,9 @@ func collectUntil(ctx context.Context, cfg Config, deadline time.Time, p *livePo } } -// stopRecorder is §4.4 steps 2 and 3. Nothing here is best-effort: the log may -// not be read until the writer has provably gone, so every failure to reach -// that state is a hard error. +// stopRecorder signals the recorder and waits for it to go. Nothing here is +// best-effort: the log may not be read until the writer has provably gone, so +// every failure to reach that state is a hard error. // // It returns the log held under an exclusive flock. The caller must keep that // file open across ReadLog and close it afterwards — the lock is the proof @@ -654,12 +630,11 @@ func collectUntil(ctx context.Context, cfg Config, deadline time.Time, p *livePo // // Two authorities, and only one of them is evidence: // -// - The PIDFILE says whether a recording was ever started. An absent or -// unparseable one is the load-bearing case, and P6's obligation on this -// phase: it must never read as "there was nothing to stop". The parent -// writes the pidfile only AFTER the child reports that it holds the log -// and is polling, and removes it on every failing path, so a missing one -// means watch failed and this run has no evidence at all. +// - The PIDFILE says whether a recording was ever started, and an absent or +// unparseable one must never read as "there was nothing to stop". The +// parent writes the pidfile only AFTER the child reports that it holds the +// log and is polling, and removes it on every failing path, so a missing +// one means watch failed and this run has no evidence at all. // - The FLOCK says whether a writer exists RIGHT NOW. Nothing removes the // pidfile when a recorder exits cleanly — the parent has long returned and // the child never learns the path — so after a --until run, a supported @@ -673,7 +648,7 @@ func collectUntil(ctx context.Context, cfg Config, deadline time.Time, p *livePo func stopRecorder(ctx context.Context, cfg Config) (*os.File, error) { pid, err := ReadPidFile(cfg.PidFile) if err != nil { - return nil, fmt.Errorf("cannot stop the recorder: %w; a pidfile is written only once a recorder reports that it is running, so an unreadable one means the recording never started (§4.4)", err) + return nil, fmt.Errorf("cannot stop the recorder: %w; a pidfile is written only once a recorder reports that it is running, so an unreadable one means the recording never started", err) } log, err := os.Open(cfg.Log) @@ -690,7 +665,7 @@ func stopRecorder(ctx context.Context, cfg Config) (*os.File, error) { // No writer. Send no signal, whatever the pidfile says — the pid may // belong to somebody else entirely by now. Which of --until, a clean // stop and a death ended the recording is the sentinel's question, - // answered by P7 check 1 over the log this unblocks. + // answered by the coverage proof over the log this unblocks. fmt.Fprintf(cfg.Notes, "note: no writer holds %s; the recorder has already finished\n", cfg.Log) return log, nil } @@ -732,7 +707,7 @@ func stopRecorder(ctx context.Context, cfg Config) (*os.File, error) { } if !cfg.Clock.Now().Before(deadline) { log.Close() - return nil, fmt.Errorf("recorder pid %d still holds %s %s after SIGTERM; refusing to read a log a writer can still append to (§4.4 step 4)", + return nil, fmt.Errorf("recorder pid %d still holds %s %s after SIGTERM; refusing to read a log a writer can still append to", pid, cfg.Log, recorderStopTimeout) } } @@ -743,34 +718,33 @@ func stopRecorder(ctx context.Context, cfg Config) (*os.File, error) { // are genuinely different faults: drain_timeout means the rule is still there // and still behind, rule_absent means it is gone. Collapsing both into // drain_timeout would name the wait instead of the fault, and Reason is a -// published vocabulary that reaches the action's JSON (§19.0). +// published vocabulary that reaches the JSON output. type drainVerdict struct { reason UnobservableReason note string } -// drainWait is §19.1 step 7 and §14.6: the final instance of the liveness -// check, asking each rule the last question — did you evaluate through the end -// of the window? A rule that cannot answer within drainTimeout is -// unobservable, never a pass. +// drainWait is the final instance of the liveness check, asking each rule the +// last question — did you evaluate through the end of the window? A rule that +// cannot answer within drainTimeout is unobservable, never a pass. // // It returns one verdict per rule it could not clear, keyed by UID, which the // caller folds into the Result. It returns an error only for a hard failure of -// the wait itself (§19.3 case 1); a rule that simply never catches up is -// reported, not raised. +// the wait itself; a rule that simply never catches up is reported, not +// raised. // -// Two rules are excluded from the wait before it starts, and both are -// exclusions of work that could not change a verdict: +// Two kinds of rule are excluded before the wait starts, both because draining +// them could not change a verdict: // -// - a rule the HEADER says was already paused when the recording opened -// (§12): it is skipped, it was not evaluating, and it never was — there is -// no evaluation to wait for. The header and not the definition, for -// decide's reason (Header.pausedAtStart): a rule the header says was -// active must be drained or faulted, because a pause somebody applied -// after the window is not evidence about the window; -// - a rule whose last poll says Found == false: P7 check 8 already makes it -// unobservable, so the only thing draining it could add is drainTimeout of -// waiting before the same answer. +// - a rule the HEADER says was already paused when the recording opened: it +// is skipped, it was not evaluating, and it never was — there is no +// evaluation to wait for. The header and not the definition, for decide's +// reason (Header.pausedAtStart): a rule the header says was active must be +// drained or faulted, because a pause somebody applied after the window is +// not evidence about the window; +// - a rule whose last poll says Found == false: the rule-absent coverage +// check already makes it unobservable, so the only thing draining it could +// add is drainTimeout of waiting before the same answer. func drainWait(ctx context.Context, cfg Config, src Source, defs []Definition, pausedAtStart map[string]bool, rt map[string]ruleTimings, polls []Poll, windowEnd time.Time, timeout time.Duration) (map[string]drainVerdict, error) { @@ -817,15 +791,16 @@ func drainWait(ctx context.Context, cfg Config, src Source, defs []Definition, p } rule := stateRuleByUID(obs.Rules, uid) if rule == nil { - // §14.5: a 2xx that parsed and carries no matching rule is an - // authoritative "the rule is gone" — P2 retried every transport - // failure long before this Observation existed. It is knowable - // on the FIRST poll, so waiting the rest of drainTimeout would - // spend two minutes to reach the same verdict under a name that - // describes the wait rather than the fault. + // A 2xx that parsed and carries no matching rule is an + // authoritative "the rule is gone" — the transport retried + // every transient failure long before this Observation + // existed. It is knowable on the FIRST poll, so waiting the + // rest of drainTimeout would spend two minutes to reach the + // same verdict under a name that describes the wait rather + // than the fault. verdicts[uid] = drainVerdict{ reason: ReasonRuleAbsent, - note: fmt.Sprintf("rule %q: absent from the state endpoint during the drain wait; there is no evaluation to wait for (§14.5)", + note: fmt.Sprintf("rule %q: absent from the state endpoint during the drain wait; there is no evaluation to wait for", pending[uid]), } delete(pending, uid) @@ -835,11 +810,11 @@ func drainWait(ctx context.Context, cfg Config, src Source, defs []Definition, p // A paused rule does not evaluate, so this one can never catch // up and the rest of drainTimeout would buy nothing. The reason // stays drain_timeout: UnobservableReason is a published - // vocabulary that reaches the action's JSON (§19.0), and the - // prose below is where the detail belongs. + // vocabulary that reaches the JSON output, and the prose below + // is where the detail belongs. verdicts[uid] = drainVerdict{ reason: ReasonDrainTimeout, - note: fmt.Sprintf("rule %q: paused before it evaluated through %s, so it never will (§14.8)", + note: fmt.Sprintf("rule %q: paused before it evaluated through %s, so it never will", pending[uid], windowEnd.Format(time.RFC3339)), } delete(pending, uid) @@ -858,7 +833,7 @@ func drainWait(ctx context.Context, cfg Config, src Source, defs []Definition, p for uid, title := range pending { verdicts[uid] = drainVerdict{ reason: ReasonDrainTimeout, - note: fmt.Sprintf("rule %q: did not evaluate through %s within the %s drain limit (§19.1 step 7)", + note: fmt.Sprintf("rule %q: did not evaluate through %s within the %s drain limit", title, windowEnd.Format(time.RFC3339), timeout), } } @@ -867,8 +842,8 @@ func drainWait(ctx context.Context, cfg Config, src Source, defs []Definition, p // Re-ask no faster than the tightest cadence among the rules still // pending: a rule evaluating every 60s cannot answer differently 200ms - // later, and hammering it would spend the request budget §5 accounts - // for on nothing. + // later, and hammering it would spend the run's request budget on + // nothing. wait := deadline.Sub(now) for uid := range pending { if every := rt[uid].pollEvery; every > 0 { @@ -897,16 +872,16 @@ func anyPollEvaluatedThrough(polls []Poll, windowEnd time.Time) bool { return false } -// evaluatedThrough is the drain wait's one comparison, and it is cross-domain -// (§16): lastEvaluation is a Grafana timestamp and windowEnd is runner-domain, -// so the Grafana value is translated by its own poll's skew. The skew BOUND is -// then subtracted rather than added — the pessimistic end of the uncertainty — -// so an evaluation that only might have reached the end of the window does not +// evaluatedThrough is the drain wait's one comparison, and it is cross-domain: +// lastEvaluation is a Grafana timestamp and windowEnd is runner-domain, so the +// Grafana value is translated by its own poll's skew. The skew BOUND is then +// subtracted rather than added — the pessimistic end of the uncertainty — so +// an evaluation that only might have reached the end of the window does not // count as one that did. Understating it costs a few more seconds of waiting; // overstating it would pass an unproven window. // // A zero lastEvaluation never satisfies the wait: only a paused rule may -// legitimately report it (§2.3), and a paused rule has nothing to drain. +// legitimately report it, and a paused rule has nothing to drain. func evaluatedThrough(lastEval time.Time, skew, bound time.Duration, windowEnd time.Time) bool { if lastEval.IsZero() { return false @@ -915,19 +890,15 @@ func evaluatedThrough(lastEval time.Time, skew, bound time.Duration, windowEnd t } // mergeDrainTimeouts folds the I/O drain wait's verdicts into the pure layer's -// Result. P7 places this merge "before decide runs"; it cannot be, because -// decide owns proveCoverage and therefore builds the Coverage map itself — so -// the merge happens immediately after, which is the same thing from every -// caller's point of view and keeps decide's signature a pure function of its -// arguments. +// Result. It runs immediately after decide rather than before it, because +// decide owns proveCoverage and therefore builds the Coverage map itself; that +// keeps decide a pure function of its arguments. // // It returns its own error rather than mutating decide's, so neither hides the // other: a run with one rule unobservable from the coverage proof and another -// from the drain wait must name both (H6 — inability beats violation, and it -// beats a second inability being dropped from the message too). The error says -// "at the drain wait" for that reason: the two are joined into one message, and -// two counts under one identical phrase read as a contradiction rather than as -// two findings. +// from the drain wait must name both. The error says "at the drain wait" for +// that reason — the two are joined into one message, and two counts under one +// identical phrase read as a contradiction rather than as two findings. func mergeDrainTimeouts(res Result, drained map[string]drainVerdict) (Result, error) { if len(drained) == 0 { return res, nil diff --git a/grafana-alertcheck/internal/gate/check_process.go b/grafana-alertcheck/internal/gate/check_process.go index 580a2a143..86dd73148 100644 --- a/grafana-alertcheck/internal/gate/check_process.go +++ b/grafana-alertcheck/internal/gate/check_process.go @@ -6,7 +6,7 @@ import ( "syscall" ) -// signalRecorder asks the recorder to stop (§4.4 step 2). +// signalRecorder asks the recorder to stop. // // The caller must have established that a writer is alive — by taking the // log's flock and being refused — before it calls this. Nothing removes the diff --git a/grafana-alertcheck/internal/gate/check_test.go b/grafana-alertcheck/internal/gate/check_test.go index 8cd94468f..d13914a8e 100644 --- a/grafana-alertcheck/internal/gate/check_test.go +++ b/grafana-alertcheck/internal/gate/check_test.go @@ -21,7 +21,7 @@ import ( // from those three numbers, and the tests assert against them by name rather // than by magic constant: // -// pollEvery 30s (§5: intervalSeconds/2) +// pollEvery 30s (intervalSeconds/2) // maxGap 60s (2 x pollEvery) // healthGrace 60s (max(maxGap, interval)) // evalStaleAfter 120s (2 x interval) @@ -99,9 +99,9 @@ var _ Source = (*checkSource)(nil) // checkStateRule builds one state-endpoint rule whose totals agree with the // instances it carries. That agreement is load-bearing: a totals map claiming -// normal instances that the instance list does not contain fails §3.2's -// verification (VerifyNormalInstancesVisible), which is a different failure -// from the one most of these tests are about. +// normal instances that the instance list does not contain fails +// VerifyNormalInstancesVisible, which is a different failure from the one most +// of these tests are about. func checkStateRule(lastEval time.Time, insts ...Instance) StateRule { totals := map[string]int{} for _, i := range insts { @@ -143,7 +143,7 @@ func baseConfig(t *testing.T, clock Clock) Config { func notesOf(cfg Config) string { return cfg.Notes.(*strings.Builder).String() } // --------------------------------------------------------------------------- -// §19.1 step 1 — configuration validation +// Configuration validation // --------------------------------------------------------------------------- func TestCheckValidateRejectsBadConfigurations(t *testing.T) { @@ -173,7 +173,7 @@ func TestCheckValidateRejectsBadConfigurations(t *testing.T) { wantErr: "no `to`", }, { - // §19.1 step 1: an empty Alerts is an error — but only without a log. + // An empty Alerts is an error — but only without a log. name: "single-step without alerts", mutate: func(c *Config) {}, wantErr: "no alert names given", @@ -185,13 +185,13 @@ func TestCheckValidateRejectsBadConfigurations(t *testing.T) { wantErr: "no alert names given", }, { - // §19.1 step 3, the other direction: the log names the alert set. + // The other direction: with a log, the log names the alert set. name: "log mode with alerts", mutate: func(c *Config) { c.Log = "log.jsonl"; c.Alerts = []string{"A"} }, wantErr: "--alerts is refused with a recorded log", }, { - // §7 — never a warning-and-continue. + // Never a warning-and-continue. name: "log mode without from", mutate: func(c *Config) { c.Log = "log.jsonl"; c.From = time.Time{} }, wantErr: "the deploy step must emit a completion timestamp", @@ -238,9 +238,9 @@ func TestCheckValidateRejectsBadConfigurations(t *testing.T) { } } -// A past `to` WITH a log is explicitly not a special mode (§7, §24.3): the -// collection loop's condition is already true and the evidence classifies -// immediately. No branch, and no refusal. +// A past `to` WITH a log is not a special mode: the collection loop's condition +// is already true and the evidence classifies immediately. No branch, and no +// refusal. func TestCheckValidateAcceptsAPastToWithALog(t *testing.T) { cfg := Config{ URL: "https://grafana.example.com", @@ -273,7 +273,7 @@ func TestCheckSingleStepCleanWindowPasses(t *testing.T) { if err != nil { t.Fatalf("check() = %v, want nil\nnotes:\n%s", err, notesOf(cfg)) } - // H7: a pass is exactly this shape. + // A pass is exactly this shape. if len(res.Violations) != 0 { t.Fatalf("Violations = %+v, want none", res.Violations) } @@ -284,7 +284,7 @@ func TestCheckSingleStepCleanWindowPasses(t *testing.T) { t.Fatalf("Coverage = %+v, want proved", cov) } - // The collection loop ran to to+transitionGrace and no further (H5). + // The collection loop ran to to+transitionGrace and no further. windowEnd := cfg.To.Add(checkGrace) if clock.Now().Before(windowEnd) { t.Errorf("stopped collecting at %s, before to+grace %s", clock.Now(), windowEnd) @@ -297,15 +297,13 @@ func TestCheckSingleStepCleanWindowPasses(t *testing.T) { t.Errorf("polled %d times, want at least the ~13 a full 6-minute window at 30s implies", got) } if notes := notesOf(cfg); !strings.Contains(notes, "planned run time") { - t.Errorf("§13.2 requires the planned run time at start; notes were:\n%s", notes) + t.Errorf("the planned run time must be printed at start; notes were:\n%s", notes) } } -// §22.2: the collapse-note-plus-satisfied-MinObserved path (resolve_test.go's -// TestResolve_CollapseByUIDGivesNoteNotError and -// TestResolve_MinObservedCountIsPostCollapse) is proven only at Resolve() -// directly; this drives the same shape through check() end to end — the two -// input names must collapse to one verdict, the run must pass, and the +// resolve_test.go proves the collapse-note-plus-satisfied-MinObserved path at +// Resolve() directly; this drives the same shape through check() end to end — +// the two input names must collapse to one verdict, the run must pass, and the // collapse note must reach the run's own notes, not just Resolve()'s return // value. func TestCheckSingleStepDuplicateAlertNamesCollapseWithNote(t *testing.T) { @@ -331,11 +329,11 @@ func TestCheckSingleStepDuplicateAlertNamesCollapseWithNote(t *testing.T) { } } -// §22.1's highest-priority regression: a rule with health=error for the -// whole window is unobservable, exit 2 — using the real "[JD] No Job -// Proposals" capture (testdata/README.md), not a synthetic Poll table, so a -// change in how the real payload shapes health/lastError cannot slip past a -// hand-built fixture that happens to still look right. +// A rule with health=error for the whole window is unobservable, exit 2 — +// driven from the real "[JD] No Job Proposals" capture (testdata/README.md), +// not a synthetic Poll table, so a change in how the real payload shapes +// health/lastError cannot slip past a hand-built fixture that happens to still +// look right. func TestCheckSingleStepContinuousHealthErrorIsUnobservable(t *testing.T) { body := readFixture(t, "state_health_error.json") rules, err := ParseState(body) @@ -358,8 +356,8 @@ func TestCheckSingleStepContinuousHealthErrorIsUnobservable(t *testing.T) { src := newCheckSource(func(_ string, _ int) (Observation, error) { // Every field but LastEvaluation stays exactly as the real capture // shaped it (health=error, the real lastError text, the real Error - // instance); LastEvaluation tracks the poll so staleness (a - // different coverage check, §14) never becomes the actual cause. + // instance); LastEvaluation tracks the poll so staleness — a + // different coverage check — never becomes the actual cause. r := base r.LastEvaluation = clock.Now() return Observation{Rules: []StateRule{r}, GrafanaNow: clock.Now(), Latency: 200 * time.Millisecond}, nil @@ -368,7 +366,7 @@ func TestCheckSingleStepContinuousHealthErrorIsUnobservable(t *testing.T) { res, err := check(context.Background(), cfg, src) if err == nil { - t.Fatalf("check() = nil, want an error: continuous health=error must be unobservable (§22.1, H6/H7)\nnotes:\n%s", notesOf(cfg)) + t.Fatalf("check() = nil, want an error: continuous health=error must be unobservable\nnotes:\n%s", notesOf(cfg)) } if len(res.Verdicts) != 1 || res.Verdicts[0].Outcome != OutcomeUnobservable { t.Fatalf("Verdicts = %+v, want one unobservable verdict", res.Verdicts) @@ -378,8 +376,8 @@ func TestCheckSingleStepContinuousHealthErrorIsUnobservable(t *testing.T) { } } -// H5: a certain violation does not release the runner early, and it does not -// stop the gate reporting exit-1 shape — violations with a nil error. +// A certain violation does not release the runner early, and it does not stop +// the gate reporting exit-1 shape — violations with a nil error. func TestCheckSingleStepFiringInstanceReportsWithoutExitingEarly(t *testing.T) { clock := newVirtualClock(testNow) cfg := baseConfig(t, clock) @@ -403,15 +401,14 @@ func TestCheckSingleStepFiringInstanceReportsWithoutExitingEarly(t *testing.T) { t.Errorf("Outcome = %q, want %q", got, OutcomePersistentlyBad) } if windowEnd := cfg.To.Add(checkGrace); clock.Now().Before(windowEnd) { - t.Errorf("exited early at %s; H5 requires collecting to %s", clock.Now(), windowEnd) + t.Errorf("exited early at %s; collection must run to %s", clock.Now(), windowEnd) } } -// §22.8: "newly_bad at from+30s gives exit 1, but ONLY after -// to+transition_grace." The test above pins H5 for a rule already bad -// before the window opened (persistently_bad); this pins the anti-fail-fast -// case the plan names explicitly — a fresh onset just inside the window -// must not release the runner the instant it is first observed. +// A newly_bad instance at from+30s gives exit 1, but ONLY after +// to+transitionGrace. The test above covers a rule already bad before the +// window opened (persistently_bad); this covers a fresh onset just inside the +// window, which must not release the runner the instant it is first observed. func TestCheckSingleStepNewOnsetDoesNotExitEarly(t *testing.T) { clock := newVirtualClock(testNow) cfg := baseConfig(t, clock) @@ -437,13 +434,13 @@ func TestCheckSingleStepNewOnsetDoesNotExitEarly(t *testing.T) { t.Fatalf("Violations = %+v, want exactly one newly_bad", res.Violations) } if windowEnd := cfg.To.Add(checkGrace); clock.Now().Before(windowEnd) { - t.Errorf("exited early at %s; H5 requires collecting to %s even for a fresh onset at from+30s", clock.Now(), windowEnd) + t.Errorf("exited early at %s; collection must run to %s even for a fresh onset at from+30s", clock.Now(), windowEnd) } } -// §22.9: an ABSENT `from` in single-step mode (as opposed to recorder mode, -// which hard-errors — TestCheckValidateRejectsBadConfigurations's "log mode -// without from") falls back to the start of this check step, with the same +// An ABSENT `from` in single-step mode (as opposed to recorder mode, which +// hard-errors — TestCheckValidateRejectsBadConfigurations's "log mode without +// from") falls back to the start of this check step, with the same // declared-blind-interval warning as an explicit early `from`. func TestCheckSingleStepAbsentFromFallsBackToStepStart(t *testing.T) { clock := newVirtualClock(testNow) @@ -459,16 +456,16 @@ func TestCheckSingleStepAbsentFromFallsBackToStepStart(t *testing.T) { } notes := notesOf(cfg) if !strings.Contains(notes, "no `from` given") { - t.Errorf("want the §4.2 fallback note; notes were:\n%s", notes) + t.Errorf("want the step-start fallback note; notes were:\n%s", notes) } if !res.From.Equal(testNow) { t.Errorf("Result.From = %s, want the step-start fallback %s", res.From, testNow) } } -// §4.2/§22.4: in single-step mode an explicit `from` earlier than the first -// observation is a DECLARED blind interval — a warning and a pass, naming the -// exact interval it cannot see. Recorder mode keeps P7 check 2 strict. +// In single-step mode an explicit `from` earlier than the first observation is +// a DECLARED blind interval — a warning and a pass, naming the exact interval +// it cannot see. Recorder mode keeps the from-bounds coverage check strict. func TestCheckSingleStepFromBeforeFirstObservationWarnsAndPasses(t *testing.T) { clock := newVirtualClock(testNow) cfg := baseConfig(t, clock) @@ -492,9 +489,9 @@ func TestCheckSingleStepFromBeforeFirstObservationWarnsAndPasses(t *testing.T) { } } -// §19.3 case 1: the failure limit was exceeded. The measurement pass succeeds -// and the collection loop then hits a terminal failure, so this exercises the -// path a live run really takes. +// The failure limit was exceeded. The measurement pass succeeds and the +// collection loop then hits a terminal failure, so this exercises the path a +// live run really takes. func TestCheckFailClosedOnExhaustedRetries(t *testing.T) { clock := newVirtualClock(testNow) cfg := baseConfig(t, clock) @@ -517,8 +514,8 @@ func TestCheckFailClosedOnExhaustedRetries(t *testing.T) { } } -// §19.3 case 2: the resolution of the definitions failed. Both shapes — the -// ruler read itself failing, and a name that resolves to nothing. +// The resolution of the definitions failed. Both shapes — the ruler read +// itself failing, and a name that resolves to nothing. func TestCheckFailClosedOnDefinitionResolution(t *testing.T) { t.Run("ruler read fails", func(t *testing.T) { clock := newVirtualClock(testNow) @@ -545,8 +542,8 @@ func TestCheckFailClosedOnDefinitionResolution(t *testing.T) { }) } -// The version gate (§2.7 control 2): an unsupported Grafana is exit 2 before -// anything else is attempted. +// The version gate: an unsupported Grafana is exit 2 before anything else is +// attempted. func TestCheckRefusesUnsupportedGrafanaVersion(t *testing.T) { clock := newVirtualClock(testNow) cfg := baseConfig(t, clock) @@ -559,9 +556,9 @@ func TestCheckRefusesUnsupportedGrafanaVersion(t *testing.T) { } } -// §5.2: the budget is checked against the latencies the measurement pass -// actually measured, and a schedule that cannot fit errors at START rather -// than producing a gap-riddled recording nobody can classify. +// The budget is checked against the latencies the measurement pass actually +// measured, and a schedule that cannot fit errors at START rather than +// producing a gap-riddled recording nobody can classify. func TestCheckSingleStepRefusesAScheduleThatDoesNotFit(t *testing.T) { clock := newVirtualClock(testNow) cfg := baseConfig(t, clock) @@ -577,7 +574,7 @@ func TestCheckSingleStepRefusesAScheduleThatDoesNotFit(t *testing.T) { } for _, want := range []string{"raising concurrency", "raising poll-interval", "watching fewer alerts"} { if !strings.Contains(err.Error(), want) { - t.Errorf("err = %q, want it to name the control %q (§5.1)", err, want) + t.Errorf("err = %q, want it to name the control %q", err, want) } } } @@ -691,16 +688,15 @@ func TestCheckRecorderModeCleanWindowPasses(t *testing.T) { if res.GrafanaVersion != "13.1.0" { t.Errorf("GrafanaVersion = %q, want the recorded one", res.GrafanaVersion) } - // The collection loop still waited out to+transitionGrace (H5) even though - // the recorder had already finished. + // The collection loop still waited out to+transitionGrace even though the + // recorder had already finished. if clock.Now().Before(windowEnd) { t.Errorf("returned at %s, before to+grace %s", clock.Now(), windowEnd) } } -// §19.3 case 3: the identity of the log is not correct. The check runs against -// the header read EARLY, so it fails before the window's wait rather than -// after it. +// The identity of the log is not correct. The check runs against the header +// read EARLY, so it fails before the window's wait rather than after it. func TestCheckFailClosedOnWrongLogIdentity(t *testing.T) { t.Run("different url", func(t *testing.T) { dir := t.TempDir() @@ -736,8 +732,8 @@ func TestCheckFailClosedOnWrongLogIdentity(t *testing.T) { }) } -// §19.3 case 4: the coverage proof failed. A hole in the middle of the -// recording is not saved by healthy data at both ends (§22.4). +// The coverage proof failed: a hole in the middle of the recording is not +// saved by healthy data at both ends. func TestCheckFailClosedOnCoverageGap(t *testing.T) { dir := t.TempDir() windowEnd := testNow.Add(5*time.Minute + checkGrace) @@ -782,13 +778,12 @@ func TestCheckFailClosedOnCoverageGap(t *testing.T) { } } -// §22.4: "an episode fully between the deploy and the start of the check" — -// recorder mode must find this at the LEADING edge of the window too, right -// after `from` (the deploy's completion), not only in the middle -// (TestCheckFailClosedOnCoverageGap above). No poll exists for -// [from, from+3m): whatever happened there is invisible to every per-poll -// check, so only the coverage gap itself can catch it — the reason this -// two-phase recorder model exists at all (§4.2). +// An episode fully between the deploy and the start of the check: recorder +// mode must find this at the LEADING edge of the window too, right after `from` +// (the deploy's completion), not only in the middle +// (TestCheckFailClosedOnCoverageGap above). No poll exists for [from, from+3m), +// so whatever happened there is invisible to every per-poll check and only the +// coverage gap itself can catch it — the reason the recorder exists at all. func TestCheckRecorderModeFindsAGapImmediatelyAfterTheDeploy(t *testing.T) { dir := t.TempDir() windowEnd := testNow.Add(5*time.Minute + checkGrace) @@ -827,14 +822,14 @@ func TestCheckRecorderModeFindsAGapImmediatelyAfterTheDeploy(t *testing.T) { } } -// §19.3 case 5: the drain limit passed. The recording itself is clean, so this -// isolates the drain wait — the rule simply never evaluates through the end of -// the window, and a rule that cannot answer that question is unobservable. +// The drain limit passed. The recording itself is clean, so this isolates the +// drain wait — the rule simply never evaluates through the end of the window, +// and a rule that cannot answer that question is unobservable. func TestCheckFailClosedOnDrainTimeout(t *testing.T) { dir := t.TempDir() windowEnd := testNow.Add(5*time.Minute + checkGrace) - // A 45s lag keeps every poll inside evalStaleAfter (120s), so P7 check 6 - // is silent and only the drain wait can fail. + // A 45s lag keeps every poll inside evalStaleAfter (120s), so the liveness + // coverage check is silent and only the drain wait can fail. logPath := recordedLog(t, dir, "https://grafana.example.com", testNow.Add(-time.Minute), testNow.Add(-time.Minute), windowEnd, windowEnd.Add(30*time.Second), 45*time.Second) writePid(t, logPath+".pid", fmt.Sprintf("%d\n", deadPid(t))) @@ -868,8 +863,8 @@ func TestCheckFailClosedOnDrainTimeout(t *testing.T) { } } -// §14.5: a rule the state endpoint no longer serves is knowable on the FIRST -// drain poll, and the answer is rule_absent — the fault — rather than +// A rule the state endpoint no longer serves is knowable on the FIRST drain +// poll, and the answer is rule_absent — the fault — rather than // drain_timeout, which would only name the wait. It must not spend the whole // drain limit to reach it. func TestCheckDrainWaitNamesADeletedRuleAtOnce(t *testing.T) { @@ -881,9 +876,9 @@ func TestCheckDrainWaitNamesADeletedRuleAtOnce(t *testing.T) { clock := newVirtualClock(testNow) cfg := recorderConfig(t, clock, logPath) - // An authoritative 2xx that parsed and carries no matching rule. P2 - // retried every transport failure long before an Observation exists, so - // this is a deletion, not a hiccup. + // An authoritative 2xx that parsed and carries no matching rule. The + // transport retried every transient failure long before an Observation + // exists, so this is a deletion, not a hiccup. src := newCheckSource(func(_ string, _ int) (Observation, error) { return Observation{GrafanaNow: clock.Now()}, nil }) @@ -1016,8 +1011,7 @@ func TestCheckHeaderPausedRuleStaysSkipped(t *testing.T) { t.Fatalf("NewWriter: %v", err) } // Named in the header, is_paused true, and no poll records at all — the - // shape watch writes for a rule paused before the window opened (P6 - // deviation 4). + // shape watch writes for a rule paused before the window opened. if err := w.WriteHeader(Header{ URL: "https://grafana.example.com", GrafanaVersion: "13.1.0", StartedAt: testNow.Add(-time.Minute), Rules: []LoggedRule{{ @@ -1054,7 +1048,7 @@ func TestCheckHeaderPausedRuleStaysSkipped(t *testing.T) { t.Errorf("Coverage[%s] present, want absent: a skipped rule has no coverage to prove", checkUID) } if len(res.Violations) != 1 { - t.Errorf("Violations = %+v, want the MinObserved shortfall (§12.1)", res.Violations) + t.Errorf("Violations = %+v, want the MinObserved shortfall", res.Violations) } res, err = run(true) @@ -1089,7 +1083,7 @@ func TestCheckDrainWaitConcludesAtOnceOnAPausedRule(t *testing.T) { t.Errorf("polled %d times, want exactly 1: a paused rule can never catch up", got) } if got := res.Coverage[checkUID].Reason; got != ReasonDrainTimeout { - t.Errorf("Reason = %q, want %q — the vocabulary is published (§19.0), so the detail goes in the note", got, ReasonDrainTimeout) + t.Errorf("Reason = %q, want %q — the vocabulary is published, so the detail goes in the note", got, ReasonDrainTimeout) } if !strings.Contains(res.Verdicts[0].Note, "paused before it evaluated through") { t.Errorf("Note = %q, want it to say the rule was paused", res.Verdicts[0].Note) @@ -1099,10 +1093,10 @@ func TestCheckDrainWaitConcludesAtOnceOnAPausedRule(t *testing.T) { } } -// P6's obligation on this phase: an absent or unparseable pidfile is never -// "there was nothing to stop". The parent writes the pidfile only once the -// child reports that it is recording, so a missing one means the recording -// never started — and the log must not be read at all. +// An absent or unparseable pidfile is never "there was nothing to stop". The +// parent writes the pidfile only once the child reports that it is recording, +// so a missing one means the recording never started — and the log must not be +// read at all. func TestCheckRefusesToReadALogItCannotStop(t *testing.T) { windowEnd := testNow.Add(5*time.Minute + checkGrace) @@ -1159,8 +1153,8 @@ func startLockHolder(t *testing.T, logPath string) int { return cmd.Process.Pid } -// §4.4 step 4: a recorder that will not let go of the log means the log may -// still be appended to, and a log a writer can change cannot be read at all. +// A recorder that will not let go of the log means the log may still be +// appended to, and a log a writer can change cannot be read at all. func TestCheckFailsWhenTheRecorderWillNotExit(t *testing.T) { dir := t.TempDir() windowEnd := testNow.Add(5*time.Minute + checkGrace) @@ -1209,9 +1203,9 @@ func TestCheckDoesNotSignalABystanderHoldingAReusedPid(t *testing.T) { } } -// §22.5: a dead pidfile (the recorder process has already exited, holding no -// flock) with NO sentinel in the log — the shape a killed `watch` leaves -// behind — must not hang the stop wait: the flock is free immediately, so +// A dead pidfile (the recorder process has already exited, holding no flock) +// with NO sentinel in the log — the shape a killed `watch` leaves behind — +// must not hang the stop wait: the flock is free immediately, so // check reads the log at once, finds no sentinel, and fails closed. func TestCheckDeadPidWithNoSentinelIsUnobservable(t *testing.T) { dir := t.TempDir() @@ -1254,12 +1248,11 @@ func TestCheckDeadPidWithNoSentinelIsUnobservable(t *testing.T) { } } -// §22.5: "an incomplete last line gives exit 2" is otherwise proven only -// indirectly — log_test.go's TestReadLogRejectsBadLogs pins ReadLog's own -// error, and TestExitCode pins that any non-nil error maps to exit 2 — but -// nothing feeds a genuinely truncated log through check() itself. This closes -// that seam: a raw file with a valid header and poll, then a torn JSON tail, -// exactly what a recorder killed mid-write leaves behind. +// An incomplete last line gives exit 2. log_test.go's TestReadLogRejectsBadLogs +// pins ReadLog's own error and TestExitCode pins that any non-nil error maps to +// exit 2, but only this feeds a genuinely truncated log through check() itself: +// a raw file with a valid header and poll, then a torn JSON tail, exactly what +// a recorder killed mid-write leaves behind. func TestCheckRecorderModeTruncatedLogFailsClosed(t *testing.T) { dir := t.TempDir() path := filepath.Join(dir, "log.jsonl") @@ -1291,10 +1284,10 @@ func TestCheckRecorderModeTruncatedLogFailsClosed(t *testing.T) { } } -// P5's "two authorities", from check's side: maxGap comes from the cadence the -// header records, never from a re-derivation off intervalSeconds. The -// fail-open direction is the one asserted — a log recorded at 5s on a 60s rule -// must still fail on a hole a re-derived 30s maxGap would have forgiven. +// One authority for the cadence, from check's side: maxGap comes from the +// cadence the header records, never from a re-derivation off intervalSeconds. +// The fail-open direction is the one asserted — a log recorded at 5s on a 60s +// rule must still fail on a hole a re-derived 30s maxGap would have forgiven. func TestCheckDerivesMaxGapFromTheRecordedCadence(t *testing.T) { dir := t.TempDir() windowEnd := testNow.Add(5*time.Minute + checkGrace) @@ -1341,8 +1334,8 @@ func TestCheckDerivesMaxGapFromTheRecordedCadence(t *testing.T) { // The pieces, in isolation // --------------------------------------------------------------------------- -// The drain wait's one comparison is cross-domain (§16), and its uncertainty -// is spent in the fail-closed direction: an evaluation that only MIGHT have +// The drain wait's one comparison is cross-domain, and its uncertainty is +// spent in the fail-closed direction: an evaluation that only MIGHT have // reached the end of the window does not count as one that did. func TestEvaluatedThroughSpendsItsUncertaintyFailingClosed(t *testing.T) { end := testNow @@ -1381,8 +1374,8 @@ func TestEvaluatedThroughSpendsItsUncertaintyFailingClosed(t *testing.T) { } } -// H6 through the merge: a drain timeout on one rule and a coverage failure on -// another must both reach the message. Neither error may shadow the other. +// A drain timeout on one rule and a coverage failure on another must both +// reach the message. Neither error may shadow the other. func TestMergeDrainTimeoutsNamesEveryUnobservableRule(t *testing.T) { res := Result{ Coverage: map[string]CoverageResult{ diff --git a/grafana-alertcheck/internal/gate/classify.go b/grafana-alertcheck/internal/gate/classify.go index 76f05edc0..5d28c16ce 100644 --- a/grafana-alertcheck/internal/gate/classify.go +++ b/grafana-alertcheck/internal/gate/classify.go @@ -7,19 +7,19 @@ import ( "time" ) -// ReasonNodata is decide's own unobservable reason (§10.1/§10.2): proveCoverage -// (P7) deliberately never sets it — health=nodata is a note there, never fatal, -// because escalating it needs Policy.NodataIsUnobservable, and the pure -// coverage layer has no Policy to consult (coverage.go, check 5). decide is -// the seam that DOES have a Policy, so the escalation lives here. +// ReasonNodata is decide's own unobservable reason: proveCoverage deliberately +// never sets it — health=nodata is a note there, never fatal, because +// escalating it needs Policy.NodataIsUnobservable, and the pure coverage layer +// has no Policy to consult (coverage.go, check 5). decide is the seam that DOES +// have a Policy, so the escalation lives here. const ReasonNodata UnobservableReason = "nodata" // Outcome is the verdict of one instance's timeline, and — after decide takes -// the worst across a rule's instances — of the rule itself (§9). It is a -// published JSON output (§19.0): the three fail values stay distinct even -// though v1 maps all three to exit 1, because a later reason string cannot -// recover the information a single "fail" value would have thrown away, and -// because splitting them later would break a published interface for no gain. +// the worst across a rule's instances — of the rule itself. It is a published +// JSON output: the three fail values stay distinct even though v1 maps all +// three to exit 1, because a later reason string cannot recover the +// information a single "fail" value would have thrown away, and because +// splitting them later would break a published interface for no gain. type Outcome string const ( @@ -34,16 +34,15 @@ const ( // PreexistingPolicy governs only the ONE ambiguous case in the outcome table: // an instance that was already bad when the window opened. A newly_bad or -// flapping instance is a fail under every policy (§11.3) — the plan lists -// them among the outcomes that "do not change" — so this type only ever +// flapping instance is a fail under every policy, so this type only ever // changes how `recovered` and `persistently_bad` are judged (isViolation // below). type PreexistingPolicy string const ( - // PreexistingFailUnlessRecovered is the default (§11.7): a preexisting - // instance that clears and stays clear is a pass (`recovered`); one that - // never clears is still a fail (`persistently_bad`). + // PreexistingFailUnlessRecovered is the default: a preexisting instance + // that clears and stays clear is a pass (`recovered`); one that never + // clears is still a fail (`persistently_bad`). PreexistingFailUnlessRecovered PreexistingPolicy = "fail-unless-recovered" // PreexistingFail makes ANY preexisting instance a fail, even one that // recovers — for a user who wants no benefit of the doubt for a @@ -61,9 +60,9 @@ type Violation struct { Alert, RuleUID string Outcome Outcome State State - Health string // raw, reporting-only, like Poll.Health (P1.2a) + Health string // raw, reporting-only, like Poll.Health LastError string - // FirstSeen is the episode's onset, in the runner domain (§16): activeAt + // FirstSeen is the episode's onset, in the runner domain: activeAt // translated by its poll's own skew when the episode opened strictly // inside the window, or `from` itself when the instance was already bad // at window-open (preexisting) — never a raw, untranslated Grafana @@ -74,14 +73,14 @@ type Violation struct { ClearedAt time.Time InstanceLabels map[string]string // Note carries an explanation for a Violation that has no instance - // behind it — the synthetic MinObserved shortfall entry decide emits - // when the deficit exceeds what any named paused rule explains (§12). - // LastError is reporting-only rule state from a real poll and must not - // double as a message field for a Violation that never touched one. + // behind it — the synthetic MinObserved shortfall entry decide emits when + // the deficit exceeds what any named paused rule explains. LastError is + // reporting-only rule state from a real poll and must not double as a + // message field for a Violation that never touched one. Note string } -// RuleVerdict is one rule's worst-of outcome (§9), always present for every +// RuleVerdict is one rule's worst-of outcome, always present for every // resolved rule — Verdicts includes the passes, not only the failures — so a // human reading the table sees every alert that was asked for, not only the // ones that misbehaved. @@ -93,9 +92,9 @@ type RuleVerdict struct { Note string } -// Policy is decide's narrowed, pure-layer view of Config/Cfg (§9's P9 -// comment): the classification knobs and the window, nothing else. No URL, -// no token, no I/O handles — those never reach the pure layer. +// Policy is decide's narrowed, pure-layer view of a Config: the classification +// knobs and the window, nothing else. No URL, no token, no I/O handles — those +// never reach the pure layer. type Policy struct { States []State Preexisting PreexistingPolicy @@ -104,37 +103,35 @@ type Policy struct { From, To time.Time } -// RuleThresholds is one non-skipped rule's resolved coverage thresholds -// (§5/§10.1/§14.1), carried on Result so the CLI's table (P10, §20.2) can -// print the numbers that answer "why" on exit 2 without decide exposing the -// unexported ruleTimings type itself. +// RuleThresholds is one non-skipped rule's resolved coverage thresholds, +// carried on Result so the CLI's table can print the numbers that answer "why" +// on exit 2 without decide exposing the unexported ruleTimings type itself. type RuleThresholds struct { MaxGap time.Duration HealthGrace time.Duration EvalStaleAfter time.Duration } -// GlobalThresholds is the run-wide half of the same information (§13.1, -// §19): transitionGrace and drainTimeout apply once, across every -// non-skipped watched rule, not per rule (globalTimings). +// GlobalThresholds is the run-wide half of the same information: +// transitionGrace and drainTimeout apply once, across every non-skipped +// watched rule, not per rule (globalTimings). type GlobalThresholds struct { TransitionGrace time.Duration - // GraceSource names, and already carries the `for` value of, the rule - // that set TransitionGrace (§13.2 requires printing both). "none" when no - // rule contributed (TransitionGrace is then 0). + // GraceSource names, and already carries the `for` value of, the rule that + // set TransitionGrace — an operator has to see both. "none" when no rule + // contributed (TransitionGrace is then 0). GraceSource string DrainTimeout time.Duration } -// Result is decide's whole answer: everything §20.2's table and the action's -// JSON outputs need. Coverage carries one CoverageResult per non-skipped -// rule — no separate Interval type anywhere in the project (§2's -// simplification table). +// Result is decide's whole answer: everything the human table and the JSON +// output need. Coverage carries one CoverageResult per non-skipped rule — +// there is deliberately no separate Interval type anywhere in the project. type Result struct { From, To time.Time GrafanaVersion string ClockSkew time.Duration // the largest |skew| across every poll decide was given, not only the ones a rule's window actually used - // ClockSkewBound is the skew BOUND (RTT/2, §16) of that SAME poll — not + // ClockSkewBound is the skew BOUND (RTT/2) of that SAME poll — not // the largest bound seen overall, which would pair a wide bound from an // unrelated slow request with the worst skew and misstate how tightly // that skew is actually known. SkewHardLimit is a separate, fixed input @@ -143,9 +140,9 @@ type Result struct { ClockSkewBound time.Duration Coverage map[string]CoverageResult // Thresholds carries one RuleThresholds per rule Coverage also covers — - // every non-skipped rule, keyed by UID. A skipped rule has neither: it - // was never scheduled, so it has no maxGap/healthGrace/evalStaleAfter to - // report (§12). + // every non-skipped rule, keyed by UID. A skipped rule has neither: it was + // never scheduled, so it has no maxGap/healthGrace/evalStaleAfter to + // report. Thresholds map[string]RuleThresholds Global GlobalThresholds Verdicts []RuleVerdict @@ -154,19 +151,19 @@ type Result struct { // episode is one contiguous, policy-bad span of one instance's timeline, // already resolved to the runner domain and clamped to [from, windowEnd]. It -// never crosses a genuine Cleared event (H2): a Vanished marker freezes the -// state instead of closing the episode, which is what keeps a vanish from -// ever reading as a recovery. +// never crosses a genuine Cleared event: a Vanished marker freezes the state +// instead of closing the episode, which is what keeps a vanish from ever +// reading as a recovery. type episode struct { start, end time.Time closedByRealClear bool } // instanceTimeline accumulates one instance's walk across a rule's in-window -// polls. preexisting is decided once, the first time this key is seen bad: -// by the translated ActiveAt against `from` (§16), never by which poll -// happened to report it first — a poll's own cadence is not evidence of when -// the condition actually began (F1/F2). +// polls. preexisting is decided once, the first time this key is seen bad: by +// the translated ActiveAt against `from`, never by which poll happened to +// report it first — a poll's own cadence is not evidence of when the condition +// actually began. type instanceTimeline struct { labels map[string]string preexisting bool @@ -179,27 +176,27 @@ type instanceTimeline struct { episodes []episode } -// runnerTime translates a Grafana-domain timestamp recorded on poll p into -// the runner domain, undoing that poll's own measured skew (§16). GrafanaNow -// and ActiveAt come from the same response, so the same poll's skew applies -// to both. This is the single implementation of that translation for the -// package (same drift argument as pollsForRule, F5): coverage.go's window -// membership test and heartbeat boundary segments call it too, rather than -// each keeping its own copy of `p.GrafanaNow.Add(-p.Skew())` that could -// silently diverge from this one. +// runnerTime translates a Grafana-domain timestamp recorded on poll p into the +// runner domain, undoing that poll's own measured skew. GrafanaNow and +// ActiveAt come from the same response, so the same poll's skew applies to +// both. This is the single implementation of that translation for the package +// (same drift argument as pollsForRule): coverage.go's window membership test +// and heartbeat boundary segments call it too, rather than each keeping its own +// copy of `p.GrafanaNow.Add(-p.Skew())` that could silently diverge from this +// one. func runnerTime(p Poll, grafanaDomain time.Time) time.Time { return grafanaDomain.Add(-p.Skew()) } // classifyRule builds every instance timeline for one rule across -// [from, windowEnd] and reduces them to the rule's worst outcome (§9), its -// merged BadFor, and the Violations the preexisting policy actually charges -// against the run. It is PURE: no I/O, no clock reads (§2) — decide supplies -// windowEnd (to + transitionGrace) rather than this function deriving it, so -// a test can pin the boundary directly. +// [from, windowEnd] and reduces them to the rule's worst outcome, its merged +// BadFor, and the Violations the preexisting policy actually charges against +// the run. It is PURE: no I/O, no clock reads — decide supplies windowEnd +// (to + transitionGrace) rather than this function deriving it, so a test can +// pin the boundary directly. // // polls need not be pre-filtered to this rule, matching proveCoverage's own -// contract (§14.5): selection is by def.UID. +// contract: selection is by def.UID. func classifyRule(def Definition, polls []Poll, from, windowEnd time.Time, badStates map[State]bool, pol PreexistingPolicy) (Outcome, time.Duration, []Violation) { rulePolls := pollsForRule(polls, def.UID) inWindow := inWindowPolls(rulePolls, from, windowEnd) @@ -207,8 +204,8 @@ func classifyRule(def Definition, polls []Poll, from, windowEnd time.Time, badSt timelines := make(map[string]*instanceTimeline) order := make([]string, 0) - // get backfills labels the first time a real Instance is seen (F4): a key - // can be created earlier by a bare Cleared/Vanished marker, which carries + // get backfills labels the first time a real Instance is seen: a key can + // be created earlier by a bare Cleared/Vanished marker, which carries // no labels of its own, and the instance later re-firing must not report // an empty InstanceLabels just because of which event happened to create // the timeline first. @@ -233,7 +230,7 @@ func classifyRule(def Definition, polls []Poll, from, windowEnd time.Time, badSt closeEpisode := func(tl *instanceTimeline, end time.Time, real bool) { // inWindowPolls admits a poll whose translated time is up to its own // skew bound PAST windowEnd (the membership test widens the boundary - // outward, §16). Without this clamp a genuine Cleared event on such a + // outward). Without this clamp a genuine Cleared event on such a // poll would produce an episode.end slightly beyond windowEnd, // contradicting the episode type's own "clamped to // [from, windowEnd]" contract. @@ -279,12 +276,12 @@ func classifyRule(def Definition, polls []Poll, from, windowEnd time.Time, badSt case !tl.seen: tl.seen = true if bad { - // Fail-closed (§16): only call an onset "preexisting" - // when even the worst-case skew error still puts it at - // or before `from`. An onset that might really have - // landed just inside the window must classify as a new - // episode, never earn the `recovered` benefit of the - // doubt it would get if it later clears (F1/F2). + // Fail-closed: only call an onset "preexisting" when + // even the worst-case skew error still puts it at or + // before `from`. An onset that might really have landed + // just inside the window must classify as a new episode, + // never earn the `recovered` benefit of the doubt it + // would get if it later clears. activeAtRunner := runnerTime(p, inst.ActiveAt) tl.preexisting = !activeAtRunner.Add(p.SkewBound()).After(from) if tl.preexisting { @@ -318,7 +315,7 @@ func classifyRule(def Definition, polls []Poll, from, windowEnd time.Time, badSt tl.lastHealth, tl.lastError = p.Health, p.LastError } - // Vanished is a deliberate no-op (H2): freeze whatever badOpen/preexisting + // Vanished is a deliberate no-op: freeze whatever badOpen/preexisting // already holds. An instance that vanishes while bad must stay bad, and // one that vanishes while never having been bad must stay uninteresting. for _, key := range p.Vanished { @@ -361,8 +358,8 @@ func classifyRule(def Definition, polls []Poll, from, windowEnd time.Time, badSt } default: // A genuinely new onset always fails, whether or not it later - // clears within the window (§11.4 point 3): only a PREEXISTING - // condition earns the benefit of `recovered`. + // clears within the window: only a PREEXISTING condition earns + // the benefit of `recovered`. instOutcome = OutcomeNewlyBad } @@ -395,9 +392,9 @@ func classifyRule(def Definition, polls []Poll, from, windowEnd time.Time, badSt } // isViolation decides whether one instance's outcome counts against the run, -// once the preexisting policy is applied. newly_bad and flapping always do -// (§11.3): both contain a genuinely new bad episode, so no policy forgives -// them. recovered and persistently_bad are, by classifyRule's construction, +// once the preexisting policy is applied. newly_bad and flapping always do: +// both contain a genuinely new bad episode, so no policy forgives them. +// recovered and persistently_bad are, by classifyRule's construction, // ALWAYS preexisting (a non-preexisting single episode is newly_bad instead, // regardless of whether it clears) — so these are the only two policy can // change, and isViolation needs no separate preexisting flag to know that. @@ -415,15 +412,16 @@ func isViolation(o Outcome, pol PreexistingPolicy) bool { } // outcomeRank orders outcomes for classifyRule's worst-of reduction across a -// rule's instances (§9). The three fail values, and recovered above clean, -// give it exactly the ordering the table requires — -// "unobservable > {flapping, persistently_bad, newly_bad} > recovered > -// skipped > clean" — with unobservable and skipped applied outside this -// function (decide owns both: unobservable from CoverageResult, skipped from -// Definition.IsPaused). The table does not distinguish among the three fail -// values, so their relative order here (flapping above persistently_bad -// above newly_bad) is an arbitrary but fixed and documented tie-break, not a -// claim that one is worse than another. +// rule's instances: +// +// unobservable > {flapping, persistently_bad, newly_bad} > recovered > +// skipped > clean +// +// with unobservable and skipped applied outside this function (decide owns +// both: unobservable from CoverageResult, skipped from the log header). The +// three fail values are not ranked against each other by anything that reads +// this, so their relative order here is an arbitrary but fixed tie-break, not +// a claim that one is worse than another. func outcomeRank(o Outcome) int { switch o { case OutcomeFlapping: @@ -466,12 +464,11 @@ func mergeDurations(eps []episode) time.Duration { } // pollsForRule filters polls to one rule and sorts them by GrafanaNow, the -// same selection proveCoverage uses (§14.5: selection is by UID, never by -// title) — stable, because two polls sharing a coarse Date header must not -// reorder nondeterministically in a pure function. This is the single -// filter+sort implementation for the package (F5): proveCoverage calls it -// too, rather than keeping its own copy that could silently drift from this -// one's membership test. +// same selection proveCoverage uses (by UID, never by title) — stable, because +// two polls sharing a coarse Date header must not reorder nondeterministically +// in a pure function. This is the single filter+sort implementation for the +// package: proveCoverage calls it too, rather than keeping its own copy that +// could silently drift from this one's membership test. func pollsForRule(polls []Poll, uid string) []Poll { var out []Poll for _, p := range polls { @@ -484,9 +481,9 @@ func pollsForRule(polls []Poll, uid string) []Poll { } // badStateSet turns Policy.States into a lookup set, defaulting to {firing} -// (§13) when the caller leaves States empty — decide applies the default -// itself so a test can pass a zero-value Policy and get v1's real default, -// rather than relying on a CLI layer that does not exist yet. +// when the caller leaves States empty — decide applies the default itself so a +// test can pass a zero-value Policy and get the real default, rather than +// depending on the CLI to have filled it in. func badStateSet(states []State) map[State]bool { if len(states) == 0 { states = []State{StateFiring} @@ -499,17 +496,16 @@ func badStateSet(states []State) map[State]bool { } // decide is the pure seam between the collected evidence and the CLI's exit -// code: nearly every §22 test targets this function, not Check (P9). It -// combines proveCoverage's nine checks with classifyRule's timelines under -// one Policy, and OWNS the H6 mapping: any unobservable rule makes decide -// return a non-nil error, which P10's CLI maps to exit 2 unconditionally -// (H7) — never to 0 or 1, and never suppressed by a real violation found -// alongside it. +// code, and carries nearly the whole test suite because of it. It combines +// proveCoverage's nine checks with classifyRule's timelines under one Policy, +// and owns the inability-beats-violation rule: any unobservable rule makes +// decide return a non-nil error, which the CLI maps to exit 2 unconditionally +// — never to 0 or 1, and never suppressed by a real violation found alongside +// it. // -// Result is fully populated even when the returned error is non-nil: H7's -// "err != nil, the violation list is irrelevant" means the CALLER must not -// use Violations to second-guess the error, not that Result stops being -// useful for the human table on exit 2. +// Result is fully populated even when the returned error is non-nil. A caller +// must not use Violations to second-guess the error, but Result stays useful +// for the human table on exit 2. func decide(h Header, polls []Poll, sentinel *time.Time, defs []Definition, rt map[string]ruleTimings, gt globalTimings, pol Policy) (Result, error) { @@ -560,7 +556,7 @@ func decide(h Header, polls []Poll, sentinel *time.Time, defs []Definition, unobservableNames []string ) - // `skipped` is decided from the header, never from defs (§12). defs are + // `skipped` is decided from the header, never from defs. defs are // resolved after the window has closed, so Definition.IsPaused describes // the present; Header.pausedAtStart describes the moment the recording // opened, which is the only moment "paused before the window opened" can @@ -617,6 +613,7 @@ func decide(h Header, polls []Poll, sentinel *time.Time, defs []Definition, }) } +<<<<<<< HEAD // MinObserved (§12): default len(defs) after the collapse (already done // by Resolve before decide ever sees defs). skipped rules count against // it unless AllowPaused says otherwise. A shortfall counts toward exit 1 @@ -627,6 +624,18 @@ func decide(h Header, polls []Poll, sentinel *time.Time, defs []Definition, // what could ever be resolved). counted := watchedCount var attributable []Definition +======= + // MinObserved defaults to len(defs) after duplicate names collapse + // (already done by Resolve before decide ever sees defs). Skipped rules + // count against it unless AllowPaused says otherwise. A shortfall counts + // toward exit 1, never exit 2 — decide never returns an error for this — + // and it has to surface through Violations like any other fail reason, so + // a shortfall always produces at least one, even when no rule is paused at + // all (an operator-supplied MinObserved that simply exceeds what could ever + // be resolved). + counted := observedCount + var chargeable []Definition +>>>>>>> 641701bb (chore: more concise comments) if pol.AllowPaused { counted += len(skippedRules) } else { @@ -638,10 +647,10 @@ func decide(h Header, polls []Poll, sentinel *time.Time, defs []Definition, if attributed >= shortfall { break } - // §12.1 requires the paused rule and --allow-paused both be - // named to the user; both live in this one Violation, in Note — - // P10's renderer prints Note verbatim rather than re-deriving - // the hint, so the exact wording here is what an operator reads. + // The paused rule and --allow-paused must both be named to the + // user; both live in this one Violation, in Note — the renderer + // prints Note verbatim rather than re-deriving the hint, so the + // exact wording here is what an operator reads. result.Violations = append(result.Violations, Violation{ Alert: def.Title, RuleUID: def.UID, Outcome: OutcomeSkipped, Note: "paused before the window opened; counts against --min-observed unless --allow-paused is set", diff --git a/grafana-alertcheck/internal/gate/classify_test.go b/grafana-alertcheck/internal/gate/classify_test.go index 6522f6e38..017d7ae6b 100644 --- a/grafana-alertcheck/internal/gate/classify_test.go +++ b/grafana-alertcheck/internal/gate/classify_test.go @@ -83,8 +83,7 @@ func TestClassifyRule_NewOnsetInsideWindowIsNewlyBad(t *testing.T) { } } -// TestClassifyRule_NewOnsetThatClearsStillFails pins §11.4 point 3: a -// genuinely new bad episode fails even if it clears again before the window +// A genuinely new bad episode fails even if it clears again before the window // ends — only a PREEXISTING condition earns the benefit of `recovered`. func TestClassifyRule_NewOnsetThatClearsStillFails(t *testing.T) { from := time.Date(2026, 1, 1, 0, 0, 0, 0, time.UTC) @@ -134,9 +133,9 @@ func TestClassifyRule_PreexistingThatRecoversIsRecoveredAndNotAViolation(t *test } } -// §22.2's "late condition": bad for 58 of a 60-minute window, clear at -// minute 58, still passes with a large BadFor — never a fail against some -// derived deadline (e.g. "must clear before 90% of the window"). +// The late condition: bad for 58 of a 60-minute window, clear at minute 58, +// still passes with a large BadFor — never a fail against some derived +// deadline (e.g. "must clear before 90% of the window"). func TestClassifyRule_LateRecoveryPassesRegardlessOfHowLateItIs(t *testing.T) { from := time.Date(2026, 1, 1, 0, 0, 0, 0, time.UTC) to := from.Add(60 * time.Minute) @@ -205,9 +204,9 @@ func TestClassifyRule_ClearThenBadAgainIsFlapping(t *testing.T) { } } -// §22.2: "a clear and then a second bad state gives flapping, at each -// possible time of the second bad state." A table over where the second -// onset lands — immediately after the clear, mid-window, and right at the +// A clear and then a second bad state gives flapping, wherever the second bad +// state lands. A table over where the second onset falls — immediately after +// the clear, mid-window, and right at the // last instant before windowEnd — closes the boundary this single fixed // timing above cannot. func TestClassifyRule_FlappingAtEveryTimingOfTheSecondOnset(t *testing.T) { @@ -244,7 +243,7 @@ func TestClassifyRule_FlappingAtEveryTimingOfTheSecondOnset(t *testing.T) { } } -// --- H2: vanished is a discontinuity, never a clear --- +// --- vanished is a discontinuity, never a clear --- func TestClassifyRule_VanishedWhileBadStaysPersistentlyBad(t *testing.T) { from := time.Date(2026, 1, 1, 0, 0, 0, 0, time.UTC) @@ -259,7 +258,7 @@ func TestClassifyRule_VanishedWhileBadStaysPersistentlyBad(t *testing.T) { } outcome, badFor, viols := classifyRule(def, polls, from, to, defaultBad, PreexistingFailUnlessRecovered) if outcome != OutcomePersistentlyBad { - t.Fatalf("outcome = %v, want persistently_bad: a vanish must never read as a recovery (H2)", outcome) + t.Fatalf("outcome = %v, want persistently_bad: a vanish must never read as a recovery", outcome) } if badFor != to.Sub(from) { t.Fatalf("badFor = %v, want the full window %v: the freeze must hold the episode open to windowEnd", badFor, to.Sub(from)) @@ -375,7 +374,7 @@ func TestClassifyRule_WorstOfMultipleInstancesWins(t *testing.T) { } } -// --- decide(): skipped rules, unobservable (H6), MinObserved, exit mapping (H7) --- +// --- decide(): skipped rules, unobservable, MinObserved, exit mapping --- func TestDecide_SkippedRuleNeverReachesProveCoverage(t *testing.T) { from := time.Date(2026, 1, 1, 0, 0, 0, 0, time.UTC) @@ -387,7 +386,7 @@ func TestDecide_SkippedRuleNeverReachesProveCoverage(t *testing.T) { pol := Policy{From: from, To: to, AllowPaused: true} // No polls, no sentinel at all: a heartbeat_gap/no_sentinel misclassification - // here would mean proveCoverage ran for a skipped rule (§4.3's obligation). + // here would mean proveCoverage ran for a skipped rule. // The HEADER is what says paused — decide reads skipped from there, not // from def.IsPaused, which is a post-window reading (Header.pausedAtStart). res, err := decide(pausedHeader(from.Add(-time.Hour), "r1"), nil, nil, defs, rt, gt, pol) @@ -415,16 +414,15 @@ func TestDecide_UnobservableRuleAlwaysReturnsAnError(t *testing.T) { // regardless of anything else. res, err := decide(Header{StartedAt: from.Add(-time.Hour)}, nil, nil, defs, rt, gt, pol) if err == nil { - t.Fatalf("err = nil, want non-nil: H6/H7 require an unobservable rule to always fail the run") + t.Fatalf("err = nil, want non-nil: an unobservable rule must always fail the run") } if len(res.Verdicts) != 1 || res.Verdicts[0].Outcome != OutcomeUnobservable { t.Fatalf("Verdicts = %+v, want exactly one unobservable verdict", res.Verdicts) } } -// TestDecide_UnobservableWinsEvenAlongsideARealViolation pins H6 exactly: -// "Any unobservable rule -> exit 2, no exception, even alongside a real -// newly_bad." +// Any unobservable rule means exit 2, with no exception — even alongside a +// real newly_bad. func TestDecide_UnobservableWinsEvenAlongsideARealViolation(t *testing.T) { from := time.Date(2026, 1, 1, 0, 0, 0, 0, time.UTC) to := from.Add(10 * time.Minute) @@ -468,18 +466,17 @@ func TestDecide_UnobservableWinsEvenAlongsideARealViolation(t *testing.T) { t.Fatalf("broken.Outcome = %v, want unobservable", gotBroken) } if gotBad != OutcomeNewlyBad { - t.Fatalf("bad.Outcome = %v, want newly_bad: classification still runs and is still visible in Verdicts (H5)", gotBad) + t.Fatalf("bad.Outcome = %v, want newly_bad: classification still runs and is still visible in Verdicts", gotBad) } if len(res.Violations) == 0 { t.Fatalf("Violations empty, want the newly_bad instance still reported even though the run fails on the unobservable rule") } } -// §22.10: "a clean verdict with a coverage gap ... must never give exit 0", -// and "a recovered verdict and a skipped verdict also need proved coverage -// of the full window." One genuinely unobservable rule ("broken", zero -// polls) alongside a rule with each of the three favorable outcomes — none -// of them may waive the run. +// A clean verdict with a coverage gap must never give exit 0, and recovered +// and skipped verdicts need proved coverage of the full window just as much. +// One genuinely unobservable rule ("broken", zero polls) alongside a rule with +// each of the three favorable outcomes — none of them may waive the run. func TestDecide_UnobservableRuleWinsOverEveryFavorableOutcome(t *testing.T) { from := time.Date(2026, 1, 1, 0, 0, 0, 0, time.UTC) to := from.Add(10 * time.Minute) @@ -575,8 +572,8 @@ func TestDecide_UnobservableRuleWinsOverEveryFavorableOutcome(t *testing.T) { } } -// §22.10: the table above puts the coverage gap on a DIFFERENT rule from the -// one with the favorable outcome. This pins the tighter claim: a rule that +// The table above puts the coverage gap on a DIFFERENT rule from the one with +// the favorable outcome. This pins the tighter claim: a rule that // itself recovers, but ALSO itself has a coverage gap, is still overridden to // unobservable — the favorable classification of a rule is never a reason to // skip that same rule's own coverage check. @@ -637,21 +634,20 @@ func TestDecide_CleanWindowIsAPass(t *testing.T) { t.Fatalf("err = %v, want nil", err) } if len(res.Violations) != 0 { - t.Fatalf("Violations = %+v, want none: H7 says a pass is exactly len(Violations)==0 && err==nil", res.Violations) + t.Fatalf("Violations = %+v, want none: a pass is exactly len(Violations)==0 && err==nil", res.Violations) } if res.Verdicts[0].Outcome != OutcomeClean { t.Fatalf("Outcome = %v, want clean", res.Verdicts[0].Outcome) } } -// §22.7's second of the plan's "if only three tests could exist" cases: a -// pause and then an unpause inside the window, with an episode that would +// A pause and then an unpause inside the window, with an episode that would // fire and resolve entirely inside the blind interval. A drain wait alone — // "did the rule eventually evaluate through windowEnd?" — would see // lastEvaluation catch up after the unpause and answer yes, a pass. decide() -// never runs a drain wait (that is check.go's I/O concern, §14.6); this pins -// that proveCoverage's own per-poll checks already refuse the window without -// one, so a live drain wait is not what is saving this case. +// never runs a drain wait (that is check.go's I/O concern); this pins that +// proveCoverage's own per-poll checks already refuse the window without one, +// so a live drain wait is not what is saving this case. func TestDecide_PauseThenUnpauseWithHiddenEpisodeGivesUnobservableNotClean(t *testing.T) { from := time.Date(2026, 1, 1, 0, 0, 0, 0, time.UTC) to := from.Add(20 * time.Minute) @@ -671,7 +667,7 @@ func TestDecide_PauseThenUnpauseWithHiddenEpisodeGivesUnobservableNotClean(t *te for ts := pauseStart; !ts.After(pauseEnd); ts = ts.Add(30 * time.Second) { // No fire/resolve is ever observed here: the rule was not // evaluating, so any real episode inside this stretch is invisible - // to every poll (§14.7). + // to every poll. polls = append(polls, Poll{RuleUID: "r1", GrafanaNow: ts, Found: true, Health: "ok", IsPaused: true, LastEvaluation: pauseStart}) } for ts := pauseEnd.Add(30 * time.Second); !ts.After(to); ts = ts.Add(30 * time.Second) { @@ -693,7 +689,7 @@ func TestDecide_PauseThenUnpauseWithHiddenEpisodeGivesUnobservableNotClean(t *te } } -// --- MinObserved shortfall (§12) --- +// --- MinObserved shortfall --- func TestDecide_SkippedOnlyShortfallProducesAViolationWithoutAnError(t *testing.T) { from := time.Date(2026, 1, 1, 0, 0, 0, 0, time.UTC) @@ -718,13 +714,13 @@ func TestDecide_SkippedOnlyShortfallProducesAViolationWithoutAnError(t *testing. res, err := decide(pausedHeader(from.Add(-time.Hour), "paused"), polls, &sentinel, defs, rt, gt, pol) if err != nil { - t.Fatalf("err = %v, want nil: a shortfall caused only by a skipped rule is exit 1, not exit 2 (§9.1)", err) + t.Fatalf("err = %v, want nil: a shortfall caused only by a skipped rule is exit 1, not exit 2", err) } if len(res.Violations) != 1 { - t.Fatalf("Violations = %+v, want exactly one: H7 needs the shortfall visible through Violations to keep its equivalence", res.Violations) + t.Fatalf("Violations = %+v, want exactly one: a shortfall must be visible through Violations like any other fail reason", res.Violations) } if v := res.Violations[0]; v.Outcome != OutcomeSkipped || v.RuleUID != "paused" || v.Alert != "Paused" { - t.Fatalf("Violations[0] = %+v, want Outcome=skipped naming the paused rule (§12.1: the message names the paused rule)", v) + t.Fatalf("Violations[0] = %+v, want Outcome=skipped naming the paused rule", v) } if res.Violations[0].Note == "" { t.Fatalf("Violations[0].Note is empty, want an explanation: the shortfall reason must not be smuggled into LastError, " + @@ -732,10 +728,9 @@ func TestDecide_SkippedOnlyShortfallProducesAViolationWithoutAnError(t *testing. } } -// TestDecide_ExplicitMinObservedShortfallWithNoPausedRuleStillProducesAViolation -// pins F3: an operator-supplied MinObserved that exceeds what could ever be -// resolved is still a shortfall, even with zero paused rules to blame it on -// — H7 must not let this silently read as a pass. +// An operator-supplied MinObserved that exceeds what could ever be resolved is +// still a shortfall, even with zero paused rules to blame it on — it must not +// silently read as a pass. func TestDecide_ExplicitMinObservedShortfallWithNoPausedRuleStillProducesAViolation(t *testing.T) { from := time.Date(2026, 1, 1, 0, 0, 0, 0, time.UTC) to := from.Add(10 * time.Minute) @@ -758,7 +753,7 @@ func TestDecide_ExplicitMinObservedShortfallWithNoPausedRuleStillProducesAViolat } if len(res.Violations) != 2 { t.Fatalf("Violations = %+v, want two: the shortfall (3-1=2) is not explained by any paused rule, "+ - "so H7 requires it to surface directly rather than pass silently", res.Violations) + "so it must surface directly rather than pass silently", res.Violations) } for _, v := range res.Violations { if v.Outcome != OutcomeSkipped { @@ -846,13 +841,12 @@ func TestDecide_NodataIsANoteByDefault(t *testing.T) { } } -// --- F1/F2 regressions: preexisting is decided by ActiveAt, not poll timing --- +// --- preexisting is decided by ActiveAt, not poll timing --- -// TestClassifyRule_OnsetBetweenFromAndFirstPollIsNewlyBadNotRecovered pins -// F1: an instance whose true onset (ActiveAt) falls strictly inside the -// window — even though the first poll that happens to observe it already -// shows it bad — must never be treated as preexisting. If it then clears, -// the plan requires newly_bad (exit 1), not recovered (exit 0). +// An instance whose true onset (ActiveAt) falls strictly inside the window — +// even though the first poll that happens to observe it already shows it bad — +// must never be treated as preexisting. If it then clears, that is newly_bad +// (exit 1), not recovered (exit 0). func TestClassifyRule_OnsetBetweenFromAndFirstPollIsNewlyBadNotRecovered(t *testing.T) { from := time.Date(2026, 1, 1, 0, 0, 0, 0, time.UTC) to := from.Add(10 * time.Minute) @@ -870,13 +864,13 @@ func TestClassifyRule_OnsetBetweenFromAndFirstPollIsNewlyBadNotRecovered(t *test outcome, badFor, viols := classifyRule(def, polls, from, to, defaultBad, PreexistingFailUnlessRecovered) if outcome != OutcomeNewlyBad { t.Fatalf("outcome = %v, want newly_bad: the onset is after `from`, so it is not preexisting even though "+ - "the FIRST in-window poll already observes it bad (F1)", outcome) + "the FIRST in-window poll already observes it bad", outcome) } if len(viols) != 1 || viols[0].Outcome != OutcomeNewlyBad { t.Fatalf("viols = %+v, want one newly_bad violation: a policy=fail-unless-recovered default must still fail this", viols) } if want := clearAt.Sub(onset); badFor != want { - t.Fatalf("badFor = %v, want %v: BadFor must count from the true onset, not from `from` (F1's overcount bug)", badFor, want) + t.Fatalf("badFor = %v, want %v: BadFor must count from the true onset, not from `from`", badFor, want) } } @@ -909,11 +903,9 @@ func TestClassifyRule_OnsetJustBeforeFromIsPreexisting(t *testing.T) { } } -// TestClassifyRule_SkewTranslatesActiveAtAcrossTheWindowBoundary pins F2: a -// poll carrying a nonzero skew must have its ActiveAt (and GrafanaNow) +// A poll carrying a nonzero skew must have its ActiveAt (and GrafanaNow) // translated to the runner domain before comparing against `from` — a raw, -// untranslated comparison would land on the wrong side of the F1 boundary -// check. +// untranslated comparison would land on the wrong side of that boundary. func TestClassifyRule_SkewTranslatesActiveAtAcrossTheWindowBoundary(t *testing.T) { from := time.Date(2026, 1, 1, 0, 0, 0, 0, time.UTC) to := from.Add(10 * time.Minute) @@ -937,14 +929,14 @@ func TestClassifyRule_SkewTranslatesActiveAtAcrossTheWindowBoundary(t *testing.T outcome, badFor, _ := classifyRule(def, []Poll{poll, stillBad}, from, to, defaultBad, PreexistingFailUnlessRecovered) if outcome != OutcomePersistentlyBad { - t.Fatalf("outcome = %v, want persistently_bad: a +90s skew must translate ActiveAt back to exactly `from` (F2)", outcome) + t.Fatalf("outcome = %v, want persistently_bad: a +90s skew must translate ActiveAt back to exactly `from`", outcome) } if badFor != to.Sub(from) { t.Fatalf("badFor = %v, want the full window %v", badFor, to.Sub(from)) } } -// --- F4: InstanceLabels must survive a timeline first created by a bare marker --- +// --- InstanceLabels must survive a timeline first created by a bare marker --- func TestClassifyRule_LabelsSurviveWhenTimelineStartsFromAClearedMarker(t *testing.T) { from := time.Date(2026, 1, 1, 0, 0, 0, 0, time.UTC) @@ -967,12 +959,11 @@ func TestClassifyRule_LabelsSurviveWhenTimelineStartsFromAClearedMarker(t *testi } if viols[0].InstanceLabels == nil || viols[0].InstanceLabels["instance"] != "a" { t.Fatalf("InstanceLabels = %+v, want {instance: a}: labels must backfill even though the "+ - "timeline was first created by a label-less Cleared marker (F4)", viols[0].InstanceLabels) + "timeline was first created by a label-less Cleared marker", viols[0].InstanceLabels) } } -// TestClassifyRule_ViolationFieldsArePrecise pins FirstSeen/ClearedAt exactly, -// not just that a violation exists (F7). +// FirstSeen/ClearedAt are pinned exactly, not just that a violation exists. func TestClassifyRule_ViolationFieldsArePrecise(t *testing.T) { from := time.Date(2026, 1, 1, 0, 0, 0, 0, time.UTC) to := from.Add(10 * time.Minute) @@ -1002,10 +993,9 @@ func TestClassifyRule_ViolationFieldsArePrecise(t *testing.T) { } } -// TestClassifyRule_ClearedEventPastWindowEndClampsToWindowEnd pins the -// episode.end clamp: inWindowPolls admits a poll up to its own skew bound -// past windowEnd (§16's widened membership test), so a genuine Cleared event -// on such a poll must not leave the episode extending beyond windowEnd. +// The episode.end clamp: inWindowPolls admits a poll up to its own skew bound +// past windowEnd, so a genuine Cleared event on such a poll must not leave the +// episode extending beyond windowEnd. func TestClassifyRule_ClearedEventPastWindowEndClampsToWindowEnd(t *testing.T) { from := time.Date(2026, 1, 1, 0, 0, 0, 0, time.UTC) to := from.Add(10 * time.Minute) @@ -1069,7 +1059,7 @@ func TestClassifyRule_OnsetJustPastWindowEndIsNewlyBadNotClean(t *testing.T) { } } -// §22.2: "a clear after `to` gives persistently_bad." classifyRule filters +// A clear after `to` gives persistently_bad. classifyRule filters // its input to [from, windowEnd] itself (inWindowPolls), so a Cleared event // GENUINELY past windowEnd — well beyond any skew bound, unlike the clamp // case above — never reaches the timeline at all: the instance is still bad diff --git a/grafana-alertcheck/internal/gate/coverage.go b/grafana-alertcheck/internal/gate/coverage.go index ec13ac91c..5c82458ee 100644 --- a/grafana-alertcheck/internal/gate/coverage.go +++ b/grafana-alertcheck/internal/gate/coverage.go @@ -5,32 +5,28 @@ import ( "time" ) -// keepLastReason is the instance Reason that check 9 watches for (§10.2). +// keepLastReason is the instance Reason that check 9 watches for. const keepLastReason = "KeepLast" -// Two things this file deliberately does not do, and where they are done -// instead — both were open obligations when P7 was written, and both are now -// discharged: +// Two things this file deliberately leaves to its callers: // -// - §7's second clause, "from more than fromFutureTolerance ahead is a hard -// error", is once-per-run input validation rather than a per-rule -// coverage check, and this function has no error return. Discharged by -// P9: the constant is fromFutureTolerance (schedule.go) and Config.validate -// (check.go) applies it. Check 2 below still owns the first clause, -// "from < StartedAt". -// - A rule paused before the window opened is never scheduled or polled -// (§4.3), so it reaches this function with zero polls and reads as one -// large heartbeat_gap, not as skipped (pinned by -// TestProveCoverage_SkippedRuleWithZeroPollsPinnedAsHeartbeatGap). -// Discharged by P8: decide returns before it ever calls proveCoverage for -// such a rule (classify.go). It reads skipped from the log header -// (Header.pausedAtStart), NOT from Definition.IsPaused — the definitions -// are re-resolved after the window closed, so they cannot answer what was -// paused when it opened. +// - "from more than fromFutureTolerance ahead of the runner's clock" is a +// hard error, but it is once-per-run input validation rather than a +// per-rule coverage check, and this function has no error return. +// Config.validate (check.go) applies it; check 2 below owns only the +// "from < StartedAt" half. +// - A rule paused before the window opened is never scheduled or polled, so +// it would reach this function with zero polls and read as one large +// heartbeat_gap rather than as skipped (pinned by +// TestProveCoverage_SkippedRuleWithZeroPollsPinnedAsHeartbeatGap). decide +// returns before it ever calls proveCoverage for such a rule, reading +// skipped from the log header (Header.pausedAtStart) and NOT from +// Definition.IsPaused — the definitions are re-resolved after the window +// closed, so they cannot answer what was paused when it opened. // UnobservableReason names why proveCoverage could not prove a rule's window. -// It is machine-readable — this reaches the action's JSON outputs, so it is a -// published vocabulary like Outcome (§19.0); prose belongs in Notes. +// It is machine-readable — this reaches the JSON output, so it is a published +// vocabulary like Outcome; prose belongs in Notes. type UnobservableReason string const ( @@ -43,16 +39,16 @@ const ( ReasonFutureEvaluation UnobservableReason = "future_evaluation" ReasonPausedInWindow UnobservableReason = "paused_in_window" ReasonRuleAbsent UnobservableReason = "rule_absent" - // ReasonDrainTimeout is set by check.go's drain wait (a later phase), - // never by proveCoverage: the wait is I/O and must not be added to this - // pure function — that would put HTTP inside the pure layer and destroy - // the seam §2's architecture depends on. + // ReasonDrainTimeout is set by check.go's drain wait, never by + // proveCoverage: the wait is I/O and must not be added to this pure + // function — that would put HTTP inside the pure layer and destroy the + // seam this design depends on. ReasonDrainTimeout UnobservableReason = "drain_timeout" ) // CoverageResult is proveCoverage's whole answer for one rule. No interval // list: proved-or-not plus the largest gap and where is everything a human -// reads on exit 2, and everything §20.2's table needs. +// reads on exit 2, and everything the rendered table needs. type CoverageResult struct { Proved bool LargestGap time.Duration @@ -65,21 +61,21 @@ type CoverageResult struct { BlindFor time.Duration } -// proveCoverage applies the nine coverage checks (§6, §10, §14) to one rule's -// polls and is PURE: no HTTP, no files, no clock reads — everything it needs -// arrives as an argument, which is what lets §22's tests build []Poll literals -// instead of a fixture server (§2). +// proveCoverage applies the nine coverage checks to one rule's polls and is +// PURE: no HTTP, no files, no clock reads — everything it needs arrives as an +// argument, which is what lets its tests build []Poll literals instead of a +// fixture server. // // polls need not be pre-filtered to this rule: proveCoverage selects by -// def.UID itself, exactly as Reduce selects by UID rather than by title -// (§14.5) — a caller handing it a whole log's polls must not have to -// pre-filter to get a correct answer. +// def.UID itself, exactly as Reduce selects by UID rather than by title — a +// caller handing it a whole log's polls must not have to pre-filter to get a +// correct answer. // // Every check always runs, even once an earlier one has already set // Unobservable: LargestGap and the notes are diagnostics an operator reads on -// exit 2 regardless of which check actually failed (§20.2). Reason names the -// FIRST check, in the order below, that failed; a later failure still adds -// its own Note. +// exit 2 regardless of which check actually failed. Reason names the FIRST +// check, in the order below, that failed; a later failure still adds its own +// Note. func proveCoverage(h Header, polls []Poll, sentinel *time.Time, t ruleTimings, def Definition, from, to time.Time, grace time.Duration) CoverageResult { @@ -101,9 +97,9 @@ func proveCoverage(h Header, polls []Poll, sentinel *time.Time, t ruleTimings, d res.Notes = append(res.Notes, fmt.Sprintf("rule %q: %s", def.Title, note)) } - // Check 1 — sentinel (§4.5). Present and At >= to+grace -> coverage - // provable; absent, or short of it, is never a pass. A recorder that died - // early must look exactly like a coverage gap, because it is one. + // Check 1 — sentinel. Present and At >= to+grace -> coverage provable; + // absent, or short of it, is never a pass. A recorder that died early must + // look exactly like a coverage gap, because it is one. switch { case sentinel == nil: fail(ReasonNoSentinel, "no stopped sentinel: the recorder never reported finishing") @@ -112,13 +108,12 @@ func proveCoverage(h Header, polls []Poll, sentinel *time.Time, t ruleTimings, d sentinel.Format(time.RFC3339), windowEnd.Format(time.RFC3339))) } - // Check 2 — from bounds (§7), first sentence only: from < StartedAt makes - // coverage unprovable, no matter how healthy the polls that DO exist look. - // Both are runner-domain clock reads (the recorder's own Clock.Now()), so - // no cross-domain translation applies here. The second sentence — from - // more than fromFutureTolerance ahead is a hard error — is Check's input - // validation, once per run rather than per rule, and belongs to a later - // phase: this function has no error return, only a per-rule verdict. + // Check 2 — from bounds: from < StartedAt makes coverage unprovable, no + // matter how healthy the polls that DO exist look. Both are runner-domain + // clock reads (the recorder's own Clock.Now()), so no cross-domain + // translation applies here. The other half of the bound — from too far + // ahead of the runner's clock — is Check's input validation, once per run + // rather than per rule. if from.Before(h.StartedAt) { fail(ReasonFromBeforeRecord, fmt.Sprintf( "requested from %s is before recording started at %s", from.Format(time.RFC3339), h.StartedAt.Format(time.RFC3339))) @@ -130,18 +125,18 @@ func proveCoverage(h Header, polls []Poll, sentinel *time.Time, t ruleTimings, d // drifting from the other's membership test. inWindow := inWindowPolls(rulePolls, from, windowEnd) - // Check 3 — heartbeat continuity (§6). Data at both ends with a hole in - // between is not enough (§22.4): this scans every gap inside the window, - // not just its edges. + // Check 3 — heartbeat continuity. Data at both ends with a hole in between + // is not enough: this scans every gap inside the window, not just its + // edges. res.LargestGap, res.LargestGapAt = ruleHeartbeatGap(inWindow, from, windowEnd) if res.LargestGap > t.maxGap { fail(ReasonHeartbeatGap, fmt.Sprintf( "gap of %s starting at %s exceeds maxGap %s", res.LargestGap, res.LargestGapAt.Format(time.RFC3339), t.maxGap)) } - // Check 4 — health=="error" (§10.1). A short blip is a note only (§22.1: - // one failed evaluation must not exit 2 over an otherwise clean window); - // only a run longer than healthGrace consumes coverage. + // Check 4 — health=="error". A short blip is a note only — one failed + // evaluation must not exit 2 over an otherwise clean window; only a run + // longer than healthGrace consumes coverage. if runLen, sawAny := longestHealthRun(inWindow, "error"); sawAny { res.Notes = append(res.Notes, fmt.Sprintf("rule %q: health=error observed (longest run %s)", def.Title, runLen)) if runLen > t.healthGrace { @@ -149,25 +144,25 @@ func proveCoverage(h Header, polls []Poll, sentinel *time.Time, t ruleTimings, d } } - // Check 5 — health=="nodata" (§10.1/§10.2). Never fatal here: 96% of the - // fleet runs no_data_state:OK, so treating this as fatal by default would - // block nearly every healthy deploy in an idle environment. Escalating it - // under Policy.NodataIsUnobservable is decide's job (a later phase), - // applied directly against the raw polls — this pure function has no - // Policy to consult and must not invent one. + // Check 5 — health=="nodata". Never fatal here: 96% of the fleet runs + // no_data_state:OK, so treating this as fatal by default would block + // nearly every healthy deploy in an idle environment. Escalating it under + // Policy.NodataIsUnobservable is decide's job, applied directly against + // the raw polls — this pure function has no Policy to consult and must not + // invent one. if _, sawAny := longestHealthRun(inWindow, "nodata"); sawAny { res.Notes = append(res.Notes, fmt.Sprintf("rule %q: health=nodata observed (not fatal; see --nodata-is-unobservable)", def.Title)) } - // Check 6 — liveness (H3). Absolute only, per poll: GrafanaNow and + // Check 6 — liveness. Absolute only, per poll: GrafanaNow and // LastEvaluation are both Grafana-domain reads off the SAME response, so // this is a same-domain comparison and uses raw values — never a delta // against a previous poll, which reports stale on ~half the polls of a // perfectly healthy rule (polling runs at intervalSeconds/2). // // Skipped only for a poll whose own flags SAY there is nothing to check: - // IsPaused (a zero LastEvaluation is legal only while paused, §2.3; check - // 7 is its detector) or !Found (no rule, no evaluation; check 8 is its + // IsPaused (a zero LastEvaluation is legal only while paused; check 7 is + // its detector) or !Found (no rule, no evaluation; check 8 is its // detector). Deliberately NOT skipped merely because LastEvaluation is // zero: ReadLog does no field validation, so a corrupted or hand-edited // log line can claim found:true, is_paused:false and still carry a zero @@ -204,11 +199,10 @@ func proveCoverage(h Header, polls []Poll, sentinel *time.Time, t ruleTimings, d staleCount, worstStale, t.evalStaleAfter, worstStaleAt.Format(time.RFC3339))) } - // Check 7 — isPaused in-window (§12.2, §14.8). The PRIMARY pause - // detector: liveness (check 6) is only the backup for what IsPaused - // cannot show (a deleted rule, a stopped scheduler, a blocked - // evaluation). This is what catches pause-then-unpause, which the drain - // wait alone passes (§14.7). + // Check 7 — isPaused in-window. The PRIMARY pause detector: liveness + // (check 6) is only the backup for what IsPaused cannot show (a deleted + // rule, a stopped scheduler, a blocked evaluation). This is what catches + // pause-then-unpause, which the drain wait alone passes. var pausedCount int var pausedAt time.Time for _, p := range inWindow { @@ -223,8 +217,8 @@ func proveCoverage(h Header, polls []Poll, sentinel *time.Time, t ruleTimings, d fail(ReasonPausedInWindow, fmt.Sprintf("observed paused on %d poll(s), first at %s", pausedCount, pausedAt.Format(time.RFC3339))) } - // Check 8 — rule absent (§14.5). Found==false is authoritative (P2 - // already retried every transport failure before a Poll record ever + // Check 8 — rule absent. Found==false is authoritative (the transport + // already retried every transient failure before a Poll record ever // exists): the rule resolved at resolve time but the state endpoint // stopped serving it. Never drop a watched rule from the verdict set // silently. @@ -242,12 +236,12 @@ func proveCoverage(h Header, polls []Poll, sentinel *time.Time, t ruleTimings, d fail(ReasonRuleAbsent, fmt.Sprintf("state endpoint returned no rule on %d poll(s), first at %s", absentCount, absentAt.Format(time.RFC3339))) } - // Check 9 — KeepLast (§10.2). Two distinct notes, both non-fatal: + // Check 9 — KeepLast. Two distinct notes, both non-fatal: // - // DECLARED: the rule's no_data_state/exec_err_state is configured as - // KeepLast — a standing blind spot (§10.2). Prefer the header's - // LoggedRule snapshot: in log mode def is re-resolved after the window - // and can drift (see pausedAtStart, log.go). Reads config, fires once. + // DECLARED: the rule's own no_data_state/exec_err_state is configured as + // KeepLast — a standing blind spot whether or not it is ever exercised + // during this particular window. This reads def, not polls, so it fires + // exactly once regardless of poll content. nds, ees := def.NoDataState, def.ExecErrState for _, lr := range h.Rules { if lr.UID == def.UID { @@ -257,13 +251,12 @@ func proveCoverage(h Header, polls []Poll, sentinel *time.Time, t ruleTimings, d } if nds == keepLastReason || ees == keepLastReason { res.Notes = append(res.Notes, fmt.Sprintf( - "rule %q: configured with no_data_state/exec_err_state=KeepLast — a stale state can continue past a real fault (§10.2)", def.Title)) + "rule %q: configured with no_data_state/exec_err_state=KeepLast — a stale state can continue past a real fault", def.Title)) } // OBSERVED: an instance actually reported the KeepLast reason during the - // window. It surfaces only as an instance Reason after P1.2a's parsing, - // and Reasons keys can be comma-joined composites, so membership - // (reasonsContain) is required — indexing "KeepLast" directly would miss - // "KeepLast, MissingSeries". + // window. It surfaces only as an instance Reason, and Reasons keys can be + // comma-joined composites, so membership (reasonsContain) is required — + // indexing "KeepLast" directly would miss "KeepLast, MissingSeries". for _, p := range inWindow { if reasonsContain(p.Reasons, keepLastReason) { res.Notes = append(res.Notes, fmt.Sprintf( @@ -277,16 +270,16 @@ func proveCoverage(h Header, polls []Poll, sentinel *time.Time, t ruleTimings, d } // inWindowPolls filters polls to those inside [from, windowEnd] using the -// CROSS-DOMAIN membership test (§16): each poll's Grafana-domain GrafanaNow -// is translated to the runner domain by its OWN skew, and its own skew bound -// is the membership tolerance, so a poll that is genuinely inside the window -// is never excluded by ordinary clock imprecision. +// CROSS-DOMAIN membership test: each poll's Grafana-domain GrafanaNow is +// translated to the runner domain by its OWN skew, and its own skew bound is +// the membership tolerance, so a poll that is genuinely inside the window is +// never excluded by ordinary clock imprecision. // -// Everything downstream of this filter (health runs, liveness, pause, -// absence) reads the poll's raw fields: GrafanaNow paired with -// LastEvaluation on the SAME response, or one poll's GrafanaNow against the -// next's, are same-domain comparisons and need no translation (§16, "Clock -// domains" — only window membership and check 3's two boundary segments do). +// Everything downstream of this filter (health runs, liveness, pause, absence) +// reads the poll's raw fields: GrafanaNow paired with LastEvaluation on the +// SAME response, or one poll's GrafanaNow against the next's, are same-domain +// comparisons and need no translation. Only window membership and check 3's +// two boundary segments cross domains. func inWindowPolls(polls []Poll, from, windowEnd time.Time) []Poll { var out []Poll for _, p := range polls { @@ -300,22 +293,22 @@ func inWindowPolls(polls []Poll, from, windowEnd time.Time) []Poll { return out } -// ruleHeartbeatGap finds the largest unobserved span inside [from, windowEnd] -// (§6), including the two boundary segments — which is why "data at both -// ends with a hole in the middle" still fails (§22.4): the segment between -// the polls just inside each edge is exactly what this measures. in must -// already be filtered to this window (inWindowPolls) and sorted by -// GrafanaNow — proveCoverage computes that filter once and threads it through -// every check, this one included, rather than each check re-filtering. +// ruleHeartbeatGap finds the largest unobserved span inside [from, windowEnd], +// including the two boundary segments — which is why "data at both ends with a +// hole in the middle" still fails: the segment between the polls just inside +// each edge is exactly what this measures. in must already be filtered to this +// window (inWindowPolls) and sorted by GrafanaNow — proveCoverage computes that +// filter once and threads it through every check, this one included, rather +// than each check re-filtering. // // The two boundary segments compare a Grafana-domain poll time against the // runner-domain from/windowEnd, so each is translated by its own poll's skew -// AND widened by that same poll's skew bound (§16: "with that poll's bound as -// the tolerance") — on the side that makes the segment larger, never smaller, -// so an uncertain boundary reads as at least as big a gap as it might really -// be. Understating it by up to the bound would be fail-open. The spacing -// BETWEEN consecutive polls compares two Grafana-domain reads to each other — -// same domain — and uses the raw GrafanaNow difference, no bound needed. +// AND widened by that same poll's skew bound — on the side that makes the +// segment larger, never smaller, so an uncertain boundary reads as at least as +// big a gap as it might really be. Understating it by up to the bound would be +// fail-open. The spacing BETWEEN consecutive polls compares two Grafana-domain +// reads to each other — same domain — and uses the raw GrafanaNow difference, +// no bound needed. func ruleHeartbeatGap(in []Poll, from, windowEnd time.Time) (largestGap time.Duration, largestGapAt time.Time) { if len(in) == 0 { return windowEnd.Sub(from), from @@ -339,16 +332,15 @@ func ruleHeartbeatGap(in []Poll, from, windowEnd time.Time) (largestGap time.Dur return largestGap, largestGapAt } -// longestHealthRun returns the longest contiguous wall-clock span (§10.1) -// during which polls — already sorted by GrafanaNow, same-domain spacing -// (§16) — read the given rule-level Health, and whether any poll matched it -// at all. +// longestHealthRun returns the longest contiguous wall-clock span during which +// polls — already sorted by GrafanaNow, same-domain spacing — read the given +// rule-level Health, and whether any poll matched it at all. // // It detects the span as it accumulates rather than waiting for the run to // end, so an open-ended run that is still failing at the last poll in the // window is measured correctly without needing data past the window: waiting // for the run to "end" would have to assume the best case about what happens -// next, which is exactly what this gate must not do (§1). +// next, which is exactly what this gate must not do. func longestHealthRun(polls []Poll, health string) (longest time.Duration, sawAny bool) { var runStart time.Time for _, p := range polls { diff --git a/grafana-alertcheck/internal/gate/coverage_test.go b/grafana-alertcheck/internal/gate/coverage_test.go index 3e6829e5a..720386d65 100644 --- a/grafana-alertcheck/internal/gate/coverage_test.go +++ b/grafana-alertcheck/internal/gate/coverage_test.go @@ -45,7 +45,7 @@ func TestProveCoverage_FiltersPollsByUID(t *testing.T) { } } -// --- Check 1: sentinel (§4.5) --- +// --- Check 1: sentinel --- func TestProveCoverage_NoSentinelIsUnobservable(t *testing.T) { from := time.Date(2026, 1, 1, 0, 0, 0, 0, time.UTC) @@ -109,7 +109,7 @@ func TestProveCoverage_SentinelExactlyAtGraceIsFine(t *testing.T) { } } -// --- Check 2: from bounds (§7) --- +// --- Check 2: from bounds --- func TestProveCoverage_FromBeforeRecordIsUnobservable(t *testing.T) { started := time.Date(2026, 1, 1, 1, 0, 0, 0, time.UTC) @@ -141,10 +141,10 @@ func TestProveCoverage_FromBeforeRecordIsUnobservable(t *testing.T) { } } -// --- Check 3: heartbeat continuity (§6) --- +// --- Check 3: heartbeat continuity --- -// TestProveCoverage_HeartbeatGapBetweenBoundariesIsUnobservable is §22.4's -// core regression: data at both ends with a hole between is not enough. +// The core heartbeat regression: data at both ends with a hole between is not +// enough. func TestProveCoverage_HeartbeatGapBetweenBoundariesIsUnobservable(t *testing.T) { from := time.Date(2026, 1, 1, 0, 0, 0, 0, time.UTC) to := from.Add(10 * time.Minute) @@ -159,7 +159,7 @@ func TestProveCoverage_HeartbeatGapBetweenBoundariesIsUnobservable(t *testing.T) res := proveCoverage(Header{StartedAt: from.Add(-time.Hour)}, polls, &sentinel, rt, def, from, to, 0) if res.Reason != ReasonHeartbeatGap { - t.Fatalf("Reason = %q, want heartbeat_gap: healthy edges with a hole in the middle must still fail (§22.4)", res.Reason) + t.Fatalf("Reason = %q, want heartbeat_gap: healthy edges with a hole in the middle must still fail", res.Reason) } // The gap is the SPACING between the two polls (598s), not either // boundary segment (1s each) — pin the actual values, not just the verdict. @@ -172,7 +172,7 @@ func TestProveCoverage_HeartbeatGapBetweenBoundariesIsUnobservable(t *testing.T) } } -// --- Check 4/5: health (§10.1/§10.2) --- +// --- Check 4/5: health --- func TestProveCoverage_HealthErrorShortBlipPassesWithNote(t *testing.T) { from := time.Date(2026, 1, 1, 0, 0, 0, 0, time.UTC) @@ -193,7 +193,7 @@ func TestProveCoverage_HealthErrorShortBlipPassesWithNote(t *testing.T) { res := proveCoverage(Header{StartedAt: from.Add(-time.Hour)}, polls, &sentinel, rt, def, from, to, 0) if !res.Proved { - t.Fatalf("Proved = false, want true: one failed evaluation must not fail an otherwise clean window (§22.1): %+v", res) + t.Fatalf("Proved = false, want true: one failed evaluation must not fail an otherwise clean window: %+v", res) } if !anyContains(res.Notes, "health=error") { t.Fatalf("Notes = %v, want a health=error note even though it did not fail the window", res.Notes) @@ -238,20 +238,19 @@ func TestProveCoverage_HealthNodataNeverFatalHere(t *testing.T) { res := proveCoverage(Header{StartedAt: from.Add(-time.Hour)}, polls, &sentinel, rt, def, from, to, 0) if !res.Proved { t.Fatalf("Proved = false, want true: health=nodata for the WHOLE window must still not be fatal by itself "+ - "(escalating it is Policy.NodataIsUnobservable's job, applied by decide in a later phase): %+v", res) + "(escalating it is Policy.NodataIsUnobservable's job, applied by decide): %+v", res) } if !anyContains(res.Notes, "health=nodata") { t.Fatalf("Notes = %v, want a health=nodata note", res.Notes) } } -// --- Check 6: liveness / H3 --- +// --- Check 6: liveness --- -// TestProveCoverage_LivenessAbsoluteNeverFalseStale is §22.7's disproportionate -// test: a healthy rule polled at intervalSeconds/2, across the full window, -// must show zero staleness violations. lastEvaluation only advances once per -// full evaluation interval here — the realistic shape a delta check -// misreads as stale on roughly half of all polls (H3). +// A healthy rule polled at intervalSeconds/2, across the full window, must +// show zero staleness violations. lastEvaluation only advances once per full +// evaluation interval here — the realistic shape a delta check misreads as +// stale on roughly half of all polls. func TestProveCoverage_LivenessAbsoluteNeverFalseStale(t *testing.T) { from := time.Date(2026, 1, 1, 0, 0, 0, 0, time.UTC) pollEvery := 30 * time.Second @@ -272,7 +271,7 @@ func TestProveCoverage_LivenessAbsoluteNeverFalseStale(t *testing.T) { res := proveCoverage(Header{StartedAt: from.Add(-time.Hour)}, polls, &sentinel, rt, def, from, windowEnd, 0) if res.Reason == ReasonStaleEvaluation || res.BlindFor != 0 { - t.Fatalf("proveCoverage flagged staleness on a healthy rule polled at intervalSeconds/2 — H3 must be absolute, "+ + t.Fatalf("proveCoverage flagged staleness on a healthy rule polled at intervalSeconds/2 — liveness must be absolute, "+ "never a delta against a previous poll: %+v", res) } if !res.Proved { @@ -312,9 +311,8 @@ func TestProveCoverage_ZeroLastEvaluationNeverFalseStale(t *testing.T) { rt := newRuleTimings(30*time.Second, 60) def := Definition{UID: "r1", Title: "R1"} - // A paused rule legitimately reports the zero time (§2.3); check 6 must - // not read that as an enormous staleness violation. Check 7 is its - // detector. + // A paused rule legitimately reports the zero time; check 6 must not read + // that as an enormous staleness violation. Check 7 is its detector. polls := []Poll{ {RuleUID: "r1", GrafanaNow: from.Add(time.Minute), Found: true, IsPaused: true}, } @@ -326,7 +324,7 @@ func TestProveCoverage_ZeroLastEvaluationNeverFalseStale(t *testing.T) { } } -// --- Check 7: isPaused in-window (§12.2, §14.8) --- +// --- Check 7: isPaused in-window --- func TestProveCoverage_PausedInWindowIsUnobservable(t *testing.T) { from := time.Date(2026, 1, 1, 0, 0, 0, 0, time.UTC) @@ -341,7 +339,7 @@ func TestProveCoverage_PausedInWindowIsUnobservable(t *testing.T) { for i := range polls { if polls[i].GrafanaNow.Equal(pausedAt) { polls[i].IsPaused = true - polls[i].LastEvaluation = time.Time{} // legal only while paused, §2.3 + polls[i].LastEvaluation = time.Time{} // legal only while paused } } sentinel := to @@ -379,7 +377,7 @@ func TestProveCoverage_PausedAfterWindowIsFine(t *testing.T) { } } -// --- Check 8: rule absent (§14.5) --- +// --- Check 8: rule absent --- func TestProveCoverage_RuleAbsentIsUnobservable(t *testing.T) { from := time.Date(2026, 1, 1, 0, 0, 0, 0, time.UTC) @@ -418,7 +416,7 @@ func denseHealthyPolls(uid string, from, to time.Time, every time.Duration) []Po return out } -// --- Check 9: KeepLast (§10.2) --- +// --- Check 9: KeepLast --- func TestProveCoverage_KeepLastObservedIsNoteOnly(t *testing.T) { from := time.Date(2026, 1, 1, 0, 0, 0, 0, time.UTC) @@ -431,7 +429,7 @@ func TestProveCoverage_KeepLastObservedIsNoteOnly(t *testing.T) { polls = append(polls, Poll{ RuleUID: "r1", GrafanaNow: ts, Found: true, Health: "ok", LastEvaluation: ts, // A comma-joined composite — reasonsContain must match by - // membership, never by an exact key, per P5's markers. + // membership, never by an exact key. Reasons: map[string]int{"KeepLast, MissingSeries": 1}, }) } @@ -446,8 +444,8 @@ func TestProveCoverage_KeepLastObservedIsNoteOnly(t *testing.T) { } } -// §22.2/§10.2: "KeepLast in the configuration gives a note" — a DIFFERENT -// claim from the observed-reason test above. A rule DECLARED with +// KeepLast in the CONFIGURATION gives a note — a different claim from the +// observed-reason test above. A rule DECLARED with // no_data_state or exec_err_state = KeepLast is a standing blind spot // whether or not any poll ever actually reports the reason, so the note // must fire off the definition alone, over an otherwise perfectly healthy @@ -480,12 +478,11 @@ func TestProveCoverage_KeepLastConfiguredIsNoteOnly(t *testing.T) { } } -// --- Clock domains (§16) --- +// --- Clock domains --- -// TestProveCoverage_SkewTranslationAtWindowBoundary pins §16's "Clock -// domains" rule: a constant clock skew on every poll must not itself read as -// a coverage gap or a from-before-record violation, because every -// cross-domain comparison translates by that poll's own skew first. +// A constant clock skew on every poll must not itself read as a coverage gap +// or a from-before-record violation, because every cross-domain comparison +// translates by that poll's own skew first. func TestProveCoverage_SkewTranslationAtWindowBoundary(t *testing.T) { from := time.Date(2026, 1, 1, 0, 0, 0, 0, time.UTC) to := from.Add(10 * time.Minute) @@ -507,15 +504,14 @@ func TestProveCoverage_SkewTranslationAtWindowBoundary(t *testing.T) { res := proveCoverage(Header{StartedAt: from.Add(-time.Hour)}, polls, &sentinel, rt, def, from, to, 0) if !res.Proved { - t.Fatalf("res = %+v, want proved: a constant clock skew must not itself read as a coverage gap (§16)", res) + t.Fatalf("res = %+v, want proved: a constant clock skew must not itself read as a coverage gap", res) } } -// --- Override round-trip (P5's "two authorities") --- +// --- Override round-trip: one authority for the cadence --- -// TestProveCoverage_OverrideRoundTrip is P7's other disproportionate done-gate -// test: it exercises DeriveTimingsFromLog and proveCoverage together, exactly -// as check will, to prove maxGap tracks the RECORDED cadence, never a +// This exercises DeriveTimingsFromLog and proveCoverage together, exactly as +// check does, to prove maxGap tracks the RECORDED cadence and never a // re-derivation from the rule's own evaluation interval. func TestProveCoverage_OverrideRoundTrip(t *testing.T) { from := time.Date(2026, 1, 1, 0, 0, 0, 0, time.UTC) @@ -577,7 +573,7 @@ func TestProveCoverage_OverrideRoundTrip(t *testing.T) { res := proveCoverage(h, polls, &sentinel, rt["r1"], defs[0], from, windowEnd, 0) if res.Reason != ReasonHeartbeatGap { t.Fatalf("Reason = %q, want heartbeat_gap: if maxGap had been re-derived from the 300s definition instead of "+ - "the recorded 5s cadence, this 250s gap would pass silently — the fail-open direction P5 warns about", res.Reason) + "the recorded 5s cadence, this 250s gap would pass silently — the fail-open direction", res.Reason) } }) } @@ -598,7 +594,7 @@ func anyContains(notes []string, substr string) bool { // found:true, is_paused:false and still carry a zero LastEvaluation (a // corrupted write, a hand-edited fixture, a future log format bug). That // combination must read as maximally stale, not be waved through the way a -// legitimately paused poll's zero time is (§2.3) — the skip must key off +// legitimately paused poll's zero time is — the skip must key off // IsPaused/Found, never off LastEvaluation being zero. func TestProveCoverage_ZeroLastEvaluationWithoutPauseIsStale(t *testing.T) { from := time.Date(2026, 1, 1, 0, 0, 0, 0, time.UTC) @@ -647,10 +643,9 @@ func TestProveCoverage_FutureLastEvaluationIsUnobservable(t *testing.T) { // --- Check 3, tightened: the boundary segments must widen by the skew bound --- -// TestProveCoverage_BoundaryGapWidensBySkewBound pins §16's "with that -// poll's bound as the tolerance" for the two boundary segments specifically: -// a boundary gap that lands EXACTLY at maxGap must still fail once the -// poll's own skew bound is added, because the translation is only a best +// The two boundary segments take their own poll's bound as the tolerance: a +// boundary gap that lands EXACTLY at maxGap must still fail once the poll's +// own skew bound is added, because the translation is only a best // estimate and understating the gap by up to the bound would be fail-open. func TestProveCoverage_BoundaryGapWidensBySkewBound(t *testing.T) { from := time.Date(2026, 1, 1, 0, 0, 0, 0, time.UTC) @@ -670,17 +665,16 @@ func TestProveCoverage_BoundaryGapWidensBySkewBound(t *testing.T) { res := proveCoverage(Header{StartedAt: from.Add(-time.Hour)}, polls, &sentinel, rt, def, from, to, 0) if res.Reason != ReasonHeartbeatGap { t.Fatalf("Reason = %q, want heartbeat_gap: the leading boundary segment sits at EXACTLY maxGap (60s) before "+ - "widening; the poll's own %s skew bound must push it past the threshold (§16), not just the skew translation", res.Reason, bound) + "widening; the poll's own %s skew bound must push it past the threshold, not just the skew translation", res.Reason, bound) } } // --- Multi-failure contract --- -// TestProveCoverage_MultipleFailuresReasonIsFirstButAllNoted exercises two -// checks failing in the same rule: check 7 (paused in-window) precedes check -// 8 (rule absent) in the §5 order, so Reason must name the pause even though -// the rule also goes absent later — and the later failure must still add its -// own Note rather than being swallowed once Reason is set. +// Two checks failing in the same rule: check 7 (paused in-window) runs before +// check 8 (rule absent), so Reason must name the pause even though the rule +// also goes absent later — and the later failure must still add its own Note +// rather than being swallowed once Reason is set. func TestProveCoverage_MultipleFailuresReasonIsFirstButAllNoted(t *testing.T) { from := time.Date(2026, 1, 1, 0, 0, 0, 0, time.UTC) to := from.Add(10 * time.Minute) @@ -705,7 +699,7 @@ func TestProveCoverage_MultipleFailuresReasonIsFirstButAllNoted(t *testing.T) { res := proveCoverage(Header{StartedAt: from.Add(-time.Hour)}, polls, &sentinel, rt, def, from, to, 0) if res.Reason != ReasonPausedInWindow { - t.Fatalf("Reason = %q, want paused_in_window (the FIRST check to fail, in §5's order)", res.Reason) + t.Fatalf("Reason = %q, want paused_in_window: the FIRST check to fail names the reason", res.Reason) } if !anyContains(res.Notes, "paused") { t.Fatalf("Notes = %v, want a note about the pause", res.Notes) @@ -716,18 +710,16 @@ func TestProveCoverage_MultipleFailuresReasonIsFirstButAllNoted(t *testing.T) { } } -// --- Skipped rules (P6/P8 obligation) --- +// --- Skipped rules --- -// TestProveCoverage_SkippedRuleWithZeroPollsPinnedAsHeartbeatGap pins a known -// gap in this function's contract, not a bug in it: a rule paused BEFORE the -// window opened is never scheduled or polled (watch.go, §4.3), so it reaches -// proveCoverage with zero polls at all. proveCoverage has no notion of +// A known limit of this function's contract, not a bug in it: a rule paused +// BEFORE the window opened is never scheduled or polled (watch.go), so it +// reaches proveCoverage with zero polls at all. proveCoverage has no notion of // "skipped" — that classification belongs to the definitions -// (LoggedRule.IsPaused / Definition.IsPaused), never to the polls — so today -// it reports the whole window as one big heartbeat_gap instead. decide (P8) -// MUST read skipped status from the definitions and either skip calling this -// function for that rule entirely, or override this result — this test pins -// today's behavior so that review has something concrete to check against. +// (LoggedRule.IsPaused / Definition.IsPaused), never to the polls — so it +// reports the whole window as one big heartbeat_gap instead. decide is what +// reads skipped status from the header and never calls this function for such +// a rule; this pins the behavior it relies on not reaching. func TestProveCoverage_SkippedRuleWithZeroPollsPinnedAsHeartbeatGap(t *testing.T) { from := time.Date(2026, 1, 1, 0, 0, 0, 0, time.UTC) to := from.Add(10 * time.Minute) @@ -738,7 +730,7 @@ func TestProveCoverage_SkippedRuleWithZeroPollsPinnedAsHeartbeatGap(t *testing.T res := proveCoverage(Header{StartedAt: from.Add(-time.Hour)}, nil, &sentinel, rt, def, from, to, 0) if res.Reason != ReasonHeartbeatGap { t.Fatalf("Reason = %q, want heartbeat_gap (pinned, not the desired end state): proveCoverage has no "+ - "'skipped' concept, so decide (P8) must handle a skipped rule's classification itself, before or "+ + "'skipped' concept, so decide must handle a skipped rule's classification itself, before or "+ "instead of calling this function", res.Reason) } } diff --git a/grafana-alertcheck/internal/gate/duration.go b/grafana-alertcheck/internal/gate/duration.go index 011c27f14..3690a2549 100644 --- a/grafana-alertcheck/internal/gate/duration.go +++ b/grafana-alertcheck/internal/gate/duration.go @@ -29,7 +29,7 @@ var promDurationUnits = []promDurationUnit{ } // ParsePromDuration parses a Grafana/Prometheus-style duration ("1h30m", "1d", "1w"). -// Unlike time.ParseDuration, it accepts "d" and "w" (§11.8). "" and "0" are 0. +// Unlike time.ParseDuration, it accepts "d" and "w". "" and "0" are 0. func ParsePromDuration(s string) (time.Duration, error) { if s == "" || s == "0" { return 0, nil diff --git a/grafana-alertcheck/internal/gate/flock.go b/grafana-alertcheck/internal/gate/flock.go index f2a3dbb24..ae21f943b 100644 --- a/grafana-alertcheck/internal/gate/flock.go +++ b/grafana-alertcheck/internal/gate/flock.go @@ -8,7 +8,7 @@ import ( ) // lockExclusive takes a non-blocking exclusive lock on f. Non-blocking is the -// point (§8): a second writer must fail immediately with an error the operator +// point: a second writer must fail immediately with an error the operator // sees, not queue behind the first and start appending to a log somebody else // already finished. func lockExclusive(f *os.File) error { @@ -30,7 +30,7 @@ func isLockContention(err error) bool { // // check needs that distinction where NewWriter does not. NewWriter is entitled // to treat any refusal as "another writer has it", because it wants the lock; -// check only wants to know whether a writer EXISTS (§4.4). The lock answers +// check only wants to know whether a writer EXISTS. The lock answers // that directly, where a pid can only infer it — the kernel releases a flock // when the holder exits, crash included, and pids get reused. func tryLockExclusive(f *os.File) (held bool, err error) { diff --git a/grafana-alertcheck/internal/gate/jsonreq.go b/grafana-alertcheck/internal/gate/jsonreq.go index bfddca382..5b0799a55 100644 --- a/grafana-alertcheck/internal/gate/jsonreq.go +++ b/grafana-alertcheck/internal/gate/jsonreq.go @@ -8,7 +8,7 @@ import ( // req decodes m[key] into *dst. It returns an error when key is absent from m // or explicitly JSON null, so a caller can never mistake absence for a zero -// value (H1) — json.Unmarshal treats "null" as a documented no-op for +// value — json.Unmarshal treats "null" as a documented no-op for // non-pointer targets (string, bool, int, ...), so without this check a // required field sent as null would silently pass through as its zero value. func req[T any](m map[string]json.RawMessage, key string, dst *T) error { diff --git a/grafana-alertcheck/internal/gate/log.go b/grafana-alertcheck/internal/gate/log.go index 243060535..2fdb71d5f 100644 --- a/grafana-alertcheck/internal/gate/log.go +++ b/grafana-alertcheck/internal/gate/log.go @@ -13,11 +13,11 @@ import ( // LogSchemaVersion is the version stamped into every log header. A log with // any other value is a read error, never a best-effort read: the log is the -// gate's only evidence, and misreading a stale shape is a fail-open (§5). +// gate's only evidence, and misreading a stale shape is a fail-open. const LogSchemaVersion = 1 // RecordType tags each JSONL line. There are exactly three, and a poll record -// IS the heartbeat — there is deliberately no separate heartbeat type (§4.6). +// IS the heartbeat — there is deliberately no separate heartbeat type. type RecordType string const ( @@ -28,13 +28,13 @@ const ( // missingSeriesReason is the reason Grafana parks a disappearing series at // ("Normal (MissingSeries)") for a couple of evaluations before deleting the -// instance. Reading that as a recovery is H2's named bug, so the markers below -// route it to Vanished (P1.2a). +// instance. Reading that as a recovery would turn a disappearing series into a +// fake recovery, so the markers below route it to Vanished. const missingSeriesReason = "MissingSeries" // LoggedRule is the per-rule identity written into the header. Together with -// the header URL it IS the log's identity, which check validates (§19.1 step -// 3), and it supplies the alert set in check mode. +// the header URL it IS the log's identity, which check validates, and it +// supplies the alert set in check mode. type LoggedRule struct { UID string `json:"uid"` Title string `json:"title"` @@ -42,17 +42,17 @@ type LoggedRule struct { Group string `json:"group"` // ForSeconds, IntervalSeconds, NoDataState and ExecErrState are purely // forensic: a resolve-time snapshot that makes the uploaded artifact - // self-describing to a human reading it after the runner is gone (§21.3). - // check never converts them back into a Definition — it always re-resolves - // definitions from the ruler API (§19.1 step 2). + // self-describing to a human reading it after the runner is gone. check + // never converts them back into a Definition — it always re-resolves + // definitions from the ruler API. ForSeconds float64 `json:"for_seconds"` IntervalSeconds int `json:"interval_seconds"` // IsPaused is NOT forensic, and is the second load-bearing field here // beside PollEverySeconds. It is the pause state at record start, which is - // the only moment `skipped` can honestly mean (§12), and decide reads it - // through Header.pausedAtStart rather than reading Definition.IsPaused off - // a ruler read taken after the window had already closed. See that method - // for what goes wrong the other way. + // the only moment `skipped` can honestly mean, and decide reads it through + // Header.pausedAtStart rather than reading Definition.IsPaused off a ruler + // read taken after the window had already closed. See that method for what + // goes wrong the other way. IsPaused bool `json:"is_paused"` NoDataState string `json:"no_data_state"` ExecErrState string `json:"exec_err_state"` @@ -60,34 +60,34 @@ type LoggedRule struct { // --poll-interval override. Load-bearing, not forensic: check derives // maxGap from it and never re-derives it from the definitions. Getting // that wrong is fail-open in the faster-override direction — a real - // recorder gap would pass silently (see "Two authorities", P5). + // recorder gap would pass silently. PollEverySeconds float64 `json:"poll_every_seconds"` } // Header is the log's first line: what was recorded, from where, and when the // recording started. It carries no States field — recording is deliberately // unfiltered, so the same log can be re-classified under different --states -// without re-recording (P6). +// without re-recording. type Header struct { SchemaVersion int `json:"schema_version"` - URL string `json:"url"` // the log's identity (§19.1 step 3) + URL string `json:"url"` // the log's identity GrafanaVersion string `json:"grafana_version"` - StartedAt time.Time `json:"started_at"` // the record start (§7 validation) - Rules []LoggedRule `json:"rules"` // THE alert set (§19.1 step 3) + StartedAt time.Time `json:"started_at"` // the record start + Rules []LoggedRule `json:"rules"` // THE alert set } // pausedAtStart reports, per rule UID, whether the rule was paused when the -// recording opened. That instant — and no other — is what `skipped` means -// (§12): a rule nobody was watching on purpose. +// recording opened. That instant — and no other — is what `skipped` means: a +// rule nobody was watching on purpose. // // It is the authority for `skipped` in BOTH modes, and the reason is that no // other source knows the right moment. `check` re-resolves the definitions -// AFTER the window closed (§19.1 step 2), so Definition.IsPaused there -// describes the present, not the window: a rule that fired and was then -// paused would read as skipped, its firing would never be classified, and -// under --allow-paused the run would pass. The header cannot drift that way, -// because watch stamps it before the deploy step runs and single-step check -// stamps it from definitions resolved at the start of its own step. +// AFTER the window closed, so Definition.IsPaused there describes the present, +// not the window: a rule that fired and was then paused would read as skipped, +// its firing would never be classified, and under --allow-paused the run would +// pass. The header cannot drift that way, because watch stamps it before the +// deploy step runs and single-step check stamps it from definitions resolved at +// the start of its own step. // // A UID the header does not name is reported NOT paused, which is the safe // direction: it then reaches proveCoverage with no polls and fails closed as @@ -104,7 +104,7 @@ func (h Header) pausedAtStart() map[string]bool { // only input the pure coverage and classification layers ever see. type Poll struct { RuleUID string `json:"rule_uid"` - GrafanaNow time.Time `json:"grafana_now"` // the Date header — H4 + GrafanaNow time.Time `json:"grafana_now"` // the response's Date header // SkewMS, SkewBoundMS and LatencyMS are milliseconds for JSONL // compactness ONLY. The pure layer never touches raw ms: it reads // Skew(), SkewBound() and Latency() below, which convert at the @@ -112,21 +112,21 @@ type Poll struct { SkewMS int64 `json:"skew_ms"` SkewBoundMS int64 `json:"skew_bound_ms"` LatencyMS int64 `json:"latency_ms"` - // Found false means an authoritative 2xx in which this rule was absent - // (§14.5) — never a transport failure, which P2 retried and never turns - // into a Poll. P7 check 8 turns it into unobservable. + // Found false means an authoritative 2xx in which this rule was absent — + // never a transport failure, which the transport retries and never turns + // into a Poll. The coverage proof turns it into unobservable. Found bool `json:"found"` // State, Health and LastError are the raw rule-level strings, reporting - // only and never classified (P1.2a). + // only and never classified. State string `json:"state,omitempty"` Health string `json:"health,omitempty"` LastError string `json:"last_error,omitempty"` - // omitzero, not omitempty: a not-found poll (and a paused rule, §2.3) has - // no evaluation time, and writing "0001-01-01T00:00:00Z" into an artifact - // humans and jq read (§21.3) invites reading it as a real timestamp. + // omitzero, not omitempty: a not-found poll (and a paused rule) has no + // evaluation time, and writing "0001-01-01T00:00:00Z" into an artifact + // humans and jq read invites reading it as a real timestamp. LastEvaluation time.Time `json:"last_evaluation,omitzero"` IsPaused bool `json:"is_paused"` - Histogram map[string]int `json:"histogram,omitempty"` // §4.9 — written, never analysed + Histogram map[string]int `json:"histogram,omitempty"` // written, never analysed // Reasons counts this poll's non-empty instance reasons, e.g. // {"NoData":1091,"Error":14}; nil when none. Reporting-only, and the ONLY // place composite states stay visible: they are canonical normal (so they @@ -134,37 +134,37 @@ type Poll struct { // // The KEYS are raw reason strings and can be comma-joined composites // ("KeepLast, MissingSeries") — newer Grafana versions join several - // reasons into one. So any consumer, P7 check 9's KeepLast note included, - // must test membership across the keys with reasonNames and must NEVER - // index a literal key: reasons["KeepLast"] misses every composite. + // reasons into one. So any consumer, the coverage proof's KeepLast note + // included, must test membership across the keys with reasonNames and must + // NEVER index a literal key: reasons["KeepLast"] misses every composite. Reasons map[string]int `json:"reasons,omitempty"` - // Abnormal holds the instances whose CANONICAL state is not normal - // (§4.6). "Normal (NoData)" and "Normal (Error)" are canonical normal and - // are deliberately not retained here (P1.2a). + // Abnormal holds the instances whose CANONICAL state is not normal. + // "Normal (NoData)" and "Normal (Error)" are canonical normal and are + // deliberately not retained here. Abnormal []Instance `json:"abnormal,omitempty"` - // Cleared and Vanished are instance keys (§4.7): keys that left the - // abnormal set, resolved against the SAME response — a clear and a - // discontinuity are not the same fact (H2). + // Cleared and Vanished are instance keys that left the abnormal set, + // resolved against the SAME response — a clear and a discontinuity are not + // the same fact. Cleared []string `json:"cleared,omitempty"` Vanished []string `json:"vanished,omitempty"` } -// Skew is the signed clock skew of this poll (§16). +// Skew is the signed clock skew of this poll. func (p Poll) Skew() time.Duration { return time.Duration(p.SkewMS) * time.Millisecond } // SkewBound is the uncertainty on Skew — the tolerance every cross-domain -// comparison in P7 applies alongside it. +// comparison applies alongside it. func (p Poll) SkewBound() time.Duration { return time.Duration(p.SkewBoundMS) * time.Millisecond } -// Latency is the wall time this poll's request took, feeding §5.2's budget check. +// Latency is the wall time this poll's request took, feeding the budget check. func (p Poll) Latency() time.Duration { return time.Duration(p.LatencyMS) * time.Millisecond } // Reducer turns each Observation into the single Poll record that goes into // the log. It holds the previous poll's abnormal instance keys per rule, which -// is all the state the transition markers need (§4.7). +// is all the state the transition markers need. // // A Reducer is safe for concurrent use: watch polls a fleet of rules -// concurrently (P6) and every one of those goroutines reduces through the same +// concurrently and every one of those goroutines reduces through the same // instance, because the per-rule marker state has to live in one place. The // lock is per-Reducer rather than per-rule — Reduce only touches maps and // slices, so it never blocks on I/O while holding it. @@ -179,12 +179,12 @@ func NewReducer() *Reducer { // Reduce selects the rule identified by uid out of obs and reduces it to a // Poll. Selection is BY UID, never by title: a filtered response can carry -// several rules sharing one title (the known 2-way collision, §14.5), and -// picking the first would silently watch the wrong rule. +// several rules sharing one title, and picking the first would silently watch +// the wrong rule. // -// The reduction (§4.6) keeps the rule-level fields, the raw totals histogram, -// the reason counts, and only the instances whose canonical state is not -// normal. That makes per-poll size independent of NORMAL cardinality — not of +// The reduction keeps the rule-level fields, the raw totals histogram, the +// reason counts, and only the instances whose canonical state is not normal. +// That makes per-poll size independent of NORMAL cardinality — not of // cardinality outright: a rule with 449 firing instances still stores all 449. func (r *Reducer) Reduce(uid string, obs Observation) Poll { r.mu.Lock() @@ -216,8 +216,8 @@ func (r *Reducer) Reduce(uid string, obs Observation) Poll { p.Histogram = rule.Totals // present indexes every instance in THIS response, normal ones included — - // the markers below must resolve a departed key against the same response - // (H2), which is impossible from the abnormal subset alone. + // the markers below must resolve a departed key against the same response, + // which is impossible from the abnormal subset alone. present := make(map[string]Instance, len(rule.Instances)) curAbnormal := make(map[string]struct{}) for _, inst := range rule.Instances { @@ -246,7 +246,7 @@ func (r *Reducer) Reduce(uid string, obs Observation) Poll { p.Vanished = append(p.Vanished, key) case reasonNames(inst.Reason, missingSeriesReason): // The vanish in disguise, caught one poll earlier than the fully - // absent case — H2's named bug. + // absent case. p.Vanished = append(p.Vanished, key) default: // Present as canonical normal without a MissingSeries reason. @@ -268,10 +268,10 @@ func (r *Reducer) Reduce(uid string, obs Observation) Poll { // // It exists for the one place a recording changes hands: watch's parent takes // the first observation of every rule and its detached child continues from -// there (P6). Without the seed, an instance that is abnormal in the parent's +// there. Without the seed, an instance that is abnormal in the parent's // observation and gone by the child's first poll produces no marker at all — -// it leaves the record as though it had never been bad, which is H2's -// fail-open reached through the handoff rather than through a reason string. +// it leaves the record as though it had never been bad, the same fail-open a +// misread MissingSeries causes, reached through the handoff instead. // // Not-found polls are skipped, mirroring Reduce: an absent rule leaves the // previous abnormal set untouched rather than emptying it. @@ -291,7 +291,7 @@ func (r *Reducer) seedFrom(polls []Poll) { } // stateRuleByUID picks one rule out of a state-endpoint response BY UID, and -// nil means the response is an authoritative "the rule is absent" (§14.5). +// nil means the response is an authoritative "the rule is absent". // // Never by title: the ?rule_name= filter is a title filter, and a filtered // response can carry several rules sharing one title (the known 2-way @@ -311,7 +311,7 @@ func stateRuleByUID(rules []StateRule, uid string) *StateRule { // reasonNames reports whether reason names want. Newer Grafana versions // comma-join several reasons into one string, so this tests membership rather -// than equality (P7 check 9 needs the same test for KeepLast). +// than equality. func reasonNames(reason, want string) bool { for part := range strings.SplitSeq(reason, ",") { if strings.TrimSpace(part) == want { @@ -321,12 +321,12 @@ func reasonNames(reason, want string) bool { return false } -// VerifyNormalInstancesVisible checks §3.2's assumption on a first -// observation: that the state endpoint really does return normal instances, -// not only the abnormal ones. If it ever stops doing so, the reduction's -// "keep the non-normal instances" becomes "keep everything the API happened to -// send" and the transition markers lose their ground truth — a silent -// fail-open. So this is verified at start, never assumed. +// VerifyNormalInstancesVisible checks, on a first observation, that the state +// endpoint really does return normal instances and not only the abnormal ones. +// If it ever stops doing so, the reduction's "keep the non-normal instances" +// becomes "keep everything the API happened to send" and the transition markers +// lose their ground truth — a silent fail-open. So this is verified at start, +// never assumed. // // The counts are summed over every totals key whose LOWERCASED name is // "normal" or "inactive". Never index one literal key: the captured @@ -351,7 +351,7 @@ func VerifyNormalInstancesVisible(rules []StateRule) error { } return fmt.Errorf( "rule %q (%s): totals claim %d normal instances but the response returned none — "+ - "the state endpoint no longer returns normal instances, which the §3.2 reduction depends on", + "the state endpoint no longer returns normal instances, which the reduction depends on", r.Title, r.UID, claimed) } return nil @@ -384,10 +384,10 @@ type stoppedRecord struct { At time.Time `json:"at"` } -// Writer appends records to the JSONL log. It is append-only by construction -// (§8): O_APPEND|O_CREATE|O_WRONLY, never O_TRUNC, so no writer can ever -// destroy evidence a previous one recorded. An exclusive non-blocking flock -// makes a second writer fail immediately rather than interleave. +// Writer appends records to the JSONL log. It is append-only by construction — +// O_APPEND|O_CREATE|O_WRONLY, never O_TRUNC — so no writer can ever destroy +// evidence a previous one recorded. An exclusive non-blocking flock makes a +// second writer fail immediately rather than interleave. type Writer struct { mu sync.Mutex f *os.File @@ -418,8 +418,8 @@ func NewWriter(path string, clock Clock) (*Writer, error) { // WriteHeader writes line 1 and stamps the current schema version, so no // caller can leave it at zero. It refuses a non-empty file: the log already // has a header, and a second one would make ReadLog's "header is line 1" -// contract a lie. In the P6 handoff the parent writes the header and the child -// only appends polls. +// contract a lie. In watch's handoff the parent writes the header and the +// detached child only appends polls. func (w *Writer) WriteHeader(h Header) error { w.mu.Lock() defer w.mu.Unlock() @@ -453,15 +453,14 @@ func (w *Writer) WritePoll(p Poll) error { return nil } -// Stop finishes recording in the fixed §4.4 order, which must not be -// reordered: let the in-flight write finish (the mutex), append the stopped -// sentinel, fsync, then release. Any other order can leave a log whose last -// durable byte is a sentinel that was never actually preceded by the polls it -// vouches for. +// Stop finishes recording in a fixed order that must not be rearranged: let +// the in-flight write finish (the mutex), append the stopped sentinel, fsync, +// then release. Any other order can leave a log whose last durable byte is a +// sentinel that was never actually preceded by the polls it vouches for. // // Stop writes the sentinel with the recorder's OWN stop time and makes no // comparison against `to` — watch never knows `to` or the transition grace. -// check does that comparison, after this writer has exited (§4.5). +// check does that comparison, after this writer has exited. // // Calling Stop twice is a no-op: watch reaches it from both a signal handler // and a defer, and a second sentinel would be indistinguishable from a second @@ -491,9 +490,8 @@ func (w *Writer) Stop() error { // Close releases the file and the lock WITHOUT writing a sentinel. It exists // for exactly one caller: watch's parent, which writes the header and then -// hands the log to the detached child that will finish it (P6). A sentinel -// here would tell check the recording ended before the child had even -// started. +// hands the log to the detached child that will finish it. A sentinel here +// would tell check the recording ended before the child had even started. func (w *Writer) Close() error { w.mu.Lock() defer w.mu.Unlock() @@ -510,15 +508,15 @@ func (w *Writer) Close() error { // ReadLogHeader reads ONLY line 1 and is the one read of a log that a writer // may still hold. That is safe for exactly one line and for no other: the // header is written once, by watch's parent, before any child appends a byte, -// the file is opened O_APPEND and never O_TRUNC (§8), so line 1 is complete -// and immutable for the whole life of the recording. +// and the file is opened O_APPEND and never O_TRUNC, so line 1 is complete and +// immutable for the whole life of the recording. // -// It exists so check can fail closed EARLY (§19.1 steps 3-4): the log's -// identity, the rule set and the cadences are all knowable at the start, and -// discovering a wrong URL or an unresolvable rule after a ten-minute wait -// helps nobody. It is advisory only — the authoritative read is still ReadLog, -// once, after the writer has exited (§4.4 step 4), and check re-validates the -// identity against that header rather than trusting this one. +// It exists so check can fail closed EARLY: the log's identity, the rule set +// and the cadences are all knowable at the start, and discovering a wrong URL +// or an unresolvable rule after a ten-minute wait helps nobody. It is advisory +// only — the authoritative read is still ReadLog, once, after the writer has +// exited, and check re-validates the identity against that header rather than +// trusting this one. func ReadLogHeader(path string) (Header, error) { f, err := os.Open(path) if err != nil { @@ -553,12 +551,11 @@ func ReadLogHeader(path string) (Header, error) { // recording never finished — check turns that into unobservable, never a // pass). // -// Call this only after the writer has exited (§4.4 step 4). Reading a log a -// writer can still append to can only produce a shorter window than the one -// that was recorded. +// Call this only after the writer has exited. Reading a log a writer can still +// append to can only produce a shorter window than the one that was recorded. // -// The parse rules are deliberately the crudest possible (§24.2): the header -// must be line 1 with a matching schema version, and ANY unparseable line — +// The parse rules are deliberately the crudest possible: the header must be +// line 1 with a matching schema version, and ANY unparseable line — // including the last one, and including a last line that follows a sentinel — // is an error, full stop. No heuristics, no discarding an untidy tail: a // truncated log is evidence that something killed the recorder, which is diff --git a/grafana-alertcheck/internal/gate/log_test.go b/grafana-alertcheck/internal/gate/log_test.go index 494c9535c..179adb5fe 100644 --- a/grafana-alertcheck/internal/gate/log_test.go +++ b/grafana-alertcheck/internal/gate/log_test.go @@ -48,7 +48,7 @@ func TestLogReduceKeepsOnlyAbnormalInstances(t *testing.T) { Instances: []Instance{ testInstance(StateNormal, "", "a"), testInstance(StateFiring, "", "b"), - // Both composites are canonical normal (P1.2a): they must NOT be + // Both composites are canonical normal: they must NOT be // retained as abnormal, and their reasons must still be counted. testInstance(StateNormal, "NoData", "c"), testInstance(StateNormal, "Error", "d"), @@ -67,11 +67,11 @@ func TestLogReduceKeepsOnlyAbnormalInstances(t *testing.T) { t.Errorf("Reasons = %v, want %v", p.Reasons, want) } // The histogram is a verbatim copy of the response totals — raw keys, no - // normalization (§4.9). + // normalization. if want := map[string]int{"alerting": 1, "normal": 2}; !reflect.DeepEqual(p.Histogram, want) { t.Errorf("Histogram = %v, want %v", p.Histogram, want) } - // Rule-level state and health stay raw and unnormalized (P1.2a). + // Rule-level state and health stay raw and unnormalized. if p.State != "firing" || p.Health != "ok" { t.Errorf("State/Health = %q/%q, want firing/ok", p.State, p.Health) } @@ -83,8 +83,8 @@ func TestLogReduceKeepsOnlyAbnormalInstances(t *testing.T) { } } -// A filtered response can hold several rules sharing one title (the known -// 2-way collision, §14.5), so the reducer must select by UID. +// A filtered response can hold several rules sharing one title, so the reducer +// must select by UID. func TestLogReduceSelectsRuleByUID(t *testing.T) { first := StateRule{UID: "ruleA", Title: "Same Title", Health: "ok", State: "inactive", LastEvaluation: testNow} second := StateRule{ @@ -120,7 +120,7 @@ func TestLogReduceRuleAbsentIsAuthoritative(t *testing.T) { } } -// H2: an instance that leaves the abnormal set is resolved against the SAME +// An instance that leaves the abnormal set is resolved against the SAME // response, and MissingSeries is a vanish, never a recovery. func TestTransitionMarkersClearedVersusVanished(t *testing.T) { badKey := instanceKey(testInstance(StateFiring, "", "b").Labels) @@ -249,8 +249,8 @@ func TestTransitionMarkersAreSortedAndPerRule(t *testing.T) { } } -// §3.2: the reduction depends on the state endpoint returning normal instances. -// If it ever stops, that must fail loudly at start, never be assumed. +// The reduction depends on the state endpoint returning normal instances. If it +// ever stops, that must fail loudly at start, never be assumed. func TestLogVerifyNormalInstancesVisible(t *testing.T) { cases := []struct { fixture string @@ -275,8 +275,8 @@ func TestLogVerifyNormalInstancesVisible(t *testing.T) { if err == nil { t.Fatalf("VerifyNormalInstancesVisible: want an error, got nil") } - if !strings.Contains(err.Error(), "§3.2") { - t.Errorf("error does not name §3.2: %v", err) + if !strings.Contains(err.Error(), "no longer returns normal instances") { + t.Errorf("error does not say the endpoint stopped returning normal instances: %v", err) } return } @@ -371,7 +371,7 @@ func TestLogModeCadenceComesFromTheHeader(t *testing.T) { } } -// watch polls a fleet concurrently through one Reducer (P6), so the marker +// watch polls a fleet concurrently through one Reducer, so the marker // state it holds per rule must be safe under -race — a latent data race here // surfaces as a wrong transition, which is the one thing markers exist to get // right. @@ -405,7 +405,7 @@ func TestLogReduceIsSafeForConcurrentUse(t *testing.T) { } // A not-found poll has no evaluation time, and the artifact is read by humans -// and jq (§21.3) — the zero time must not appear as though it were real. +// and jq — the zero time must not appear as though it were real. func TestLogPollOmitsTheZeroEvaluationTime(t *testing.T) { absent := NewReducer().Reduce("rule1", observation(testNow)) b, err := json.Marshal(pollRecord{Type: RecordPoll, Poll: absent}) @@ -504,13 +504,13 @@ func TestWriterReadLogRoundTrip(t *testing.T) { t.Fatalf("sentinel is nil after Stop") } // Stop stamps the recorder's own stop time and makes no comparison - // against `to` — watch never knows it (§4.5). + // against `to` — watch never knows it. if !sentinel.Equal(testNow.Add(2 * time.Minute)) { t.Errorf("sentinel = %s, want the writer's stop time %s", sentinel, testNow.Add(2*time.Minute)) } } -// §8: the log is append-only. A second run against the same path must never +// The log is append-only. A second run against the same path must never // destroy the evidence the first one recorded. func TestWriterAppendsAndNeverTruncates(t *testing.T) { path := filepath.Join(t.TempDir(), "log.jsonl") @@ -529,7 +529,7 @@ func TestWriterAppendsAndNeverTruncates(t *testing.T) { t.Fatalf("read: %v", err) } - // The P6 handoff: the parent wrote the header and closed; the child + // The handoff: the parent wrote the header and closed; the child // reopens the same path and appends without a second header. child, _ := newTestWriter(t, path) if err := child.WritePoll(Poll{RuleUID: "rule1", Found: true, GrafanaNow: testNow.Add(time.Minute)}); err != nil { @@ -633,8 +633,8 @@ func TestSentinelStopIsIdempotentAndLast(t *testing.T) { } } -// Close is the parent's handoff path in P6: a sentinel there would tell check -// the recording ended before the child had even started. +// Close is the parent's handoff path: a sentinel there would tell check the +// recording ended before the child had even started. func TestSentinelCloseWritesNone(t *testing.T) { path := filepath.Join(t.TempDir(), "log.jsonl") w, _ := newTestWriter(t, path) @@ -658,8 +658,9 @@ func TestSentinelCloseWritesNone(t *testing.T) { } // An unfinished recording reads cleanly with a nil sentinel — ReadLog reports -// the absence and P7 turns it into unobservable. It is never ReadLog's job to -// call that a failure, and never anyone's job to call it a pass. +// the absence and the coverage proof turns it into unobservable. It is never +// ReadLog's job to call that a failure, and never anyone's job to call it a +// pass. func TestReadLogWithoutASentinel(t *testing.T) { path := filepath.Join(t.TempDir(), "log.jsonl") w, _ := newTestWriter(t, path) @@ -682,7 +683,7 @@ func TestReadLogWithoutASentinel(t *testing.T) { } } -// The read rules are deliberately the crudest possible (§24.2): any unparseable +// The read rules are deliberately the crudest possible: any unparseable // line is an error, full stop — including the last one, and including a last // line that follows a sentinel. func TestReadLogRejectsBadLogs(t *testing.T) { @@ -778,7 +779,7 @@ func TestReadLogMissingFile(t *testing.T) { } } -// §22.3: per-poll log size must not grow across polls on a high-cardinality +// Per-poll log size must not grow across polls on a high-cardinality // rule, and the one firing instance among 2446 must still be attributed by its // labels. The reduction makes size independent of NORMAL cardinality — the // firing instances are still stored, which is why a clear shrinks the record. @@ -851,8 +852,8 @@ func TestLogSizeIsFlatAcrossPollsOnAHighCardinalityRule(t *testing.T) { } // The log must stay readable by anything that reads JSONL, one flat object per -// line with its type tag — an uploaded artifact (§21.3) is read by humans and -// by jq, not only by ReadLog. +// line with its type tag — an uploaded artifact is read by humans and by jq, +// not only by ReadLog. func TestLogRecordsAreFlatOneLineObjects(t *testing.T) { path := filepath.Join(t.TempDir(), "log.jsonl") w, _ := newTestWriter(t, path) diff --git a/grafana-alertcheck/internal/gate/parse_ruler.go b/grafana-alertcheck/internal/gate/parse_ruler.go index c27e5b9c4..ae878d4c9 100644 --- a/grafana-alertcheck/internal/gate/parse_ruler.go +++ b/grafana-alertcheck/internal/gate/parse_ruler.go @@ -7,9 +7,9 @@ import ( "time" ) -// RuleKind classifies a ruler-endpoint rule by shape, not by name (P1.3). -// P3 rejects KindDatasourceManaged and KindRecording, but only for rules a -// user actually named — ParseDefinitions itself never rejects. +// RuleKind classifies a ruler-endpoint rule by shape, not by name. Resolve +// rejects KindDatasourceManaged and KindRecording, but only for rules a user +// actually named — ParseDefinitions itself never rejects. type RuleKind int const ( @@ -22,8 +22,8 @@ const ( // (/api/ruler/grafana/api/v1/rules). IntervalSeconds, NoDataState and // ExecErrState live inside the grafana_alert block and are only populated for // KindGrafanaManaged — a datasource-managed rule has no such block by -// definition (§11.6 drops relativeTimeRange/keep_firing_for entirely; neither -// is parsed here). +// definition. relativeTimeRange and keep_firing_for are deliberately not +// parsed: nothing in the gate reads them. type Definition struct { UID, Title, Folder, FolderUID, Group string For time.Duration @@ -43,8 +43,8 @@ func ParseDefinitions(body []byte) ([]Definition, error) { } // Map iteration order is nondeterministic; sort namespace names so - // ParseDefinitions' output order is stable across calls (P3's candidate - // listings and any golden test depend on that). + // ParseDefinitions' output order is stable across calls — Resolve's + // candidate listings and the golden tests depend on that. names := make([]string, 0, len(namespaces)) for name := range namespaces { names = append(names, name) @@ -137,11 +137,11 @@ func parseDefinition(raw json.RawMessage, folder, group string) (Definition, err // Classify by the presence of "record" before requiring anything else. // no_data_state/exec_err_state/is_paused/intervalSeconds are alerting-only // concepts a recording rule may not carry at all — its real shape is - // unverified (none exist in the fleet capture) — and P3 refuses this - // Kind categorically before any of this would gate a release. Strict- - // parsing a recording rule into a hard error over fields it was never - // going to use would brick `list` and every resolve for rules nobody - // named (§11.6, "do not reject here"). + // unverified, none exist in the fleet capture — and Resolve refuses this + // Kind categorically before any of this would gate a release. + // Strict-parsing a recording rule into a hard error over fields it was + // never going to use would brick `list` and every resolve for rules nobody + // named. var record json.RawMessage if err := opt(ga, "record", &record); err != nil { return Definition{}, fmt.Errorf("rule %q: grafana_alert: %w", uid, err) diff --git a/grafana-alertcheck/internal/gate/parse_ruler_test.go b/grafana-alertcheck/internal/gate/parse_ruler_test.go index 3cf2e3261..b225ffc2c 100644 --- a/grafana-alertcheck/internal/gate/parse_ruler_test.go +++ b/grafana-alertcheck/internal/gate/parse_ruler_test.go @@ -20,7 +20,7 @@ func TestParseDefinitions_RulerRules(t *testing.T) { } // The real 2-way duplicate title: same folder, same group, same title, - // distinct UIDs (§17, §22.2). + // distinct UIDs — only uid: can tell them apart. a, ok := byUID["rule0000006a"] if !ok { t.Fatalf("missing rule0000006a") @@ -81,7 +81,8 @@ func TestParseDefinitions_DatasourceManaged(t *testing.T) { } // A datasource-managed rule has no uid in this shape; its only identity // is the Prometheus "alert" name — a synthetic UID would be invented - // shape, and an empty Title would make P3's refusal-by-name unreachable. + // shape, and an empty Title would make Resolve's refusal-by-name + // unreachable. if defs[0].Title != "ExampleTargetDown" { t.Errorf("Title = %q, want ExampleTargetDown", defs[0].Title) } diff --git a/grafana-alertcheck/internal/gate/parse_state.go b/grafana-alertcheck/internal/gate/parse_state.go index 7ffff1cfc..7c35c7e7d 100644 --- a/grafana-alertcheck/internal/gate/parse_state.go +++ b/grafana-alertcheck/internal/gate/parse_state.go @@ -7,7 +7,7 @@ import ( "time" ) -// State is the canonical instance state (P1.2a). It is distinct from the raw, +// State is the canonical instance state. It is distinct from the raw, // unnormalized vocabularies the API uses at the rule level and at the instance // level — see normalizeInstanceState. type State string @@ -22,12 +22,12 @@ const ( // Instance is one entry of a rule's alerts[]. State is always canonical; Reason // is the opaque suffix of a "State (Reason)" composite ("" when the API gave a -// bare state). Reason is reporting-only except for the H2 MissingSeries routing +// bare state). Reason is reporting-only except for the MissingSeries routing // done downstream in the log markers. // -// The json tags are for the JSONL log's abnormal-instance list (P5) only — -// parsing an API response never goes through them, because parseInstance -// decodes field by field through req/opt to keep H1's presence checks explicit. +// The json tags are for the JSONL log's abnormal-instance list only — parsing +// an API response never goes through them, because parseInstance decodes field +// by field through req/opt to keep the presence checks explicit. type Instance struct { Labels map[string]string `json:"labels"` State State `json:"state"` @@ -37,12 +37,12 @@ type Instance struct { } // StateRule is one rule from the state endpoint -// (/api/prometheus/grafana/api/v1/rules), fully and strictly parsed (H1). +// (/api/prometheus/grafana/api/v1/rules), fully and strictly parsed. type StateRule struct { UID, Title, Folder, Group string Interval time.Duration // State and Health are raw, lowercase, and reporting-only — never - // classified (P1.2a). State in particular is never normalized. + // classified. State in particular is never normalized. State, Health string LastError string LastEvaluation time.Time @@ -53,7 +53,7 @@ type StateRule struct { // ParseState strictly parses a state-endpoint response body into its rules. // A missing or unparseable required field (health, state, lastEvaluation on -// each rule; interval on each group) is an error, never a zero value (H1). +// each rule; interval on each group) is an error, never a zero value. func ParseState(body []byte) ([]StateRule, error) { var top map[string]json.RawMessage if err := json.Unmarshal(body, &top); err != nil { @@ -133,11 +133,10 @@ func parseStateRule(raw json.RawMessage, folder, group string, interval time.Dur if err := req(m, "health", &r.Health); err != nil { return StateRule{}, fmt.Errorf("rule %q: %w", uid, err) } - // isPaused is not one of H1's four named required fields, but this parser - // extends that contract to it: the zero-time rule below can't tell a - // paused rule from a broken one without it, and it's the primary - // in-window pause detector (H2/§12.2) — a silent false default would be - // exactly the fail-open bug H1 exists to kill. + // isPaused is required rather than optional: the zero-time rule below + // can't tell a paused rule from a broken one without it, and it's the + // primary in-window pause detector — a silent false default would be + // exactly the fail-open this parser's strictness exists to kill. if err := req(m, "isPaused", &r.IsPaused); err != nil { return StateRule{}, fmt.Errorf("rule %q: %w", uid, err) } @@ -150,7 +149,7 @@ func parseStateRule(raw json.RawMessage, folder, group string, interval time.Dur if err != nil { return StateRule{}, fmt.Errorf("rule %q: lastEvaluation: %w", uid, err) } - // The zero-time rule (§2.3): only a paused rule may report the zero time. + // Only a paused rule may report the zero time. if lastEval.IsZero() && !r.IsPaused { return StateRule{}, fmt.Errorf("rule %q: lastEvaluation is the zero time but isPaused is false", uid) } @@ -198,10 +197,10 @@ func parseInstance(raw json.RawMessage) (Instance, error) { return Instance{}, err } - // activeAt is also not in H1's named list, extended here for the same - // reason as StateRule.IsPaused: it's the onset time BadFor (P8) measures - // from, so a silently zeroed one would misclassify how long an instance - // has been bad rather than failing loudly. + // activeAt is required for the same reason as StateRule.IsPaused: it's the + // onset time BadFor measures from, so a silently zeroed one would + // misclassify how long an instance has been bad rather than failing + // loudly. var activeAtStr string if err := req(m, "activeAt", &activeAtStr); err != nil { return Instance{}, err @@ -224,8 +223,8 @@ func parseInstance(raw json.RawMessage) (Instance, error) { } // baseInstanceStates is the strict 5-value allowlist for the base of an -// instance state (P1.2a). Anything else — including an unrecognized base -// inside a "Base (Reason)" composite — is a parse error (H1, §2.7 control 3). +// instance state. Anything else — including an unrecognized base inside a +// "Base (Reason)" composite — is a parse error. var baseInstanceStates = map[string]State{ "Normal": StateNormal, "Alerting": StateFiring, diff --git a/grafana-alertcheck/internal/gate/parse_state_test.go b/grafana-alertcheck/internal/gate/parse_state_test.go index bc8a2a64c..4ccca1391 100644 --- a/grafana-alertcheck/internal/gate/parse_state_test.go +++ b/grafana-alertcheck/internal/gate/parse_state_test.go @@ -167,7 +167,7 @@ func TestParseState_HappyPaths(t *testing.T) { t.Fatalf("Instances = %+v, want one firing instance", r.Instances) } if r.Totals["normal"] == 0 { - t.Errorf(`Totals["normal"] = 0, want >0 (this is the §3.2 mismatch the fixture exists to capture)`) + t.Errorf(`Totals["normal"] = 0, want >0 (the totals/instances mismatch this fixture exists to capture)`) } }, }, @@ -192,11 +192,11 @@ func TestParseState_HappyPaths(t *testing.T) { } } -// TestParseState_MustError is the H1 regression suite: it doesn't just check -// err != nil (a stray comma in a fixture would keep that green forever while -// the actual check regressed) — it asserts the error names the specific -// offending field or value, so a real H1 check going missing fails loudly -// here instead of surviving unnoticed. +// The strict-parsing regression suite: it doesn't just check err != nil (a +// stray comma in a fixture would keep that green forever while the actual +// check regressed) — it asserts the error names the specific offending field +// or value, so a check going missing fails loudly here instead of surviving +// unnoticed. func TestParseState_MustError(t *testing.T) { cases := []struct { fixture string @@ -293,7 +293,7 @@ func TestInstanceKey_NoCollision(t *testing.T) { } } -// minimalStateBody is the smallest H1-legal state response: one group, one +// minimalStateBody is the smallest legal state response: one group, one // rule, no optional keys at all, plus whatever extra is spliced in verbatim // before the rule's closing brace — for isolating one optional key at a time // rather than relying on a fixture that removes several together. @@ -304,10 +304,10 @@ func minimalStateBody(extraRuleJSON string) []byte { `"lastEvaluation":"2026-01-01T00:00:00Z"%s}]}]}}`, extraRuleJSON) } -// §22.2: keepFiringFor is named alongside alerts/totals/labels as an optional -// key (§3.1), but state_missing_optional.json removes it together with -// everything else — never in isolation, so a regression that made it -// required specifically would not be caught by that fixture alone. +// keepFiringFor is optional alongside alerts/totals/labels, but +// state_missing_optional.json removes it together with everything else — never +// in isolation, so a regression that made it required specifically would not +// be caught by that fixture alone. func TestParseState_KeepFiringForIsOptional(t *testing.T) { tests := []struct { name string @@ -329,7 +329,7 @@ func TestParseState_KeepFiringForIsOptional(t *testing.T) { } } -// §22.2: labels is optional at the INSTANCE level (opt(m, "labels", ...) in +// labels is optional at the INSTANCE level (opt(m, "labels", ...) in // parseInstance), distinct from the rule-level labels state_missing_optional.json // already covers — an instance can exist with no labels of its own. func TestParseState_InstanceWithoutLabelsParses(t *testing.T) { @@ -349,9 +349,9 @@ func TestParseState_InstanceWithoutLabelsParses(t *testing.T) { // synthesizeHighCardinalityState builds a state response with a single rule // holding `alerting` Alerting instances and `normal` Normal instances, by // cloning the one real instance in state_one_instance.json. It is never -// committed (§3.2, §22.3, §22.6) — the 2446-instance rule this stands in for -// is ~600 KB and exists only to prove the parser and (in later phases) the -// reducer don't choke on real fleet cardinality. +// committed — the 2446-instance rule this stands in for is ~600 KB and exists +// only to prove the parser and the reducer don't choke on real fleet +// cardinality. func synthesizeHighCardinalityState(t *testing.T, alerting, normal int) []byte { t.Helper() base := readFixture(t, "state_one_instance.json") diff --git a/grafana-alertcheck/internal/gate/resolve.go b/grafana-alertcheck/internal/gate/resolve.go index 3c3710038..329644f95 100644 --- a/grafana-alertcheck/internal/gate/resolve.go +++ b/grafana-alertcheck/internal/gate/resolve.go @@ -7,8 +7,8 @@ import ( "strings" ) -// Resolve turns the operator-supplied alert names into resolved Definitions -// (§17). Order is load-bearing (§17.3): +// Resolve turns the operator-supplied alert names into resolved Definitions. +// Order is load-bearing: // // 1. Trim each name. // 2. Discard empty lines. @@ -18,9 +18,9 @@ import ( // user less than a failure). // // The caller-visible consequence: len(resolved) is the count *after* the -// collapse. A later phase's MinObserved must default from that length, never -// from len(names) — using the input line count would make one rule named -// twice turn an achievable default into an unsatisfiable one (§17.3). +// collapse, and MinObserved must default from that length, never from +// len(names) — using the input line count would make one rule named twice turn +// an achievable default into an unsatisfiable one. func Resolve(defs []Definition, names []string, folder string) (resolved []Definition, notes []string, err error) { seenUID := map[string]string{} // uid -> the first input name that resolved to it for _, raw := range names { @@ -45,15 +45,15 @@ func Resolve(defs []Definition, names []string, folder string) (resolved []Defin return resolved, notes, nil } -// resolveOne resolves a single trimmed, non-empty name against defs (§17.1): -// one match wins outright, zero is an error with suggestions, two or more is -// an error listing every candidate. folder scopes a bare title (no "/" in the -// name) to one folder; it is ignored for the "Folder/Title" and -// "Folder/Group/Title" forms, which already name their own folder. +// resolveOne resolves a single trimmed, non-empty name against defs: one match +// wins outright, zero is an error with suggestions, two or more is an error +// listing every candidate. folder scopes a bare title (no "/" in the name) to +// one folder; it is ignored for the "Folder/Title" and "Folder/Group/Title" +// forms, which already name their own folder. // -// Policy on unsupported kinds (datasource-managed, recording) — decided here -// because §17.1 only says to refuse them, not how they interact with the -// no-match/ambiguous surfaces: a name can still match an unsupported rule (so +// Unsupported kinds (datasource-managed, recording) are refused, and how that +// interacts with the no-match/ambiguous surfaces is decided here: a name can +// still match an unsupported rule (so // naming one by title still gets the specific, named refusal, not a bare "no // match"), but only *supported* candidates count for ambiguity — an // unsupported rule sharing a title with a supported one is resolved silently @@ -73,7 +73,7 @@ func resolveOne(defs []Definition, name, folder string) (Definition, error) { } // uid == "" falls through to the same message as "not found": several // Definition kinds legitimately carry UID == "" (datasource-managed - // rules have no uid at all, P1.3), so matching on an empty suffix + // rules have no uid at all), so matching on an empty suffix // would silently hit one of those and report a misleading // kind-specific refusal for what is really an empty/typo'd uid. This // deliberately does not go through noMatchError: that function's @@ -118,9 +118,9 @@ func resolveOne(defs []Definition, name, folder string) (Definition, error) { } } -// supportedDefs filters out the two kinds §17.1 refuses. Only these -// participate in name-based matching, the no-match rule count, and substring -// suggestions (see the policy note on resolveOne). +// supportedDefs filters out the two refused kinds. Only these participate in +// name-based matching, the no-match rule count, and substring suggestions (see +// the policy note on resolveOne). func supportedDefs(defs []Definition) []Definition { out := make([]Definition, 0, len(defs)) for _, d := range defs { @@ -132,7 +132,7 @@ func supportedDefs(defs []Definition) []Definition { } // classifyForm splits name into the Title | Folder/Title | Folder/Group/Title -// forms (§17). A bare title is scoped by folder when the caller supplied one; +// forms. A bare title is scoped by folder when the caller supplied one; // the two- and three-segment forms already carry their own folder and ignore // it. // @@ -158,8 +158,8 @@ func classifyForm(name, folder string) (wantFolder, wantGroup, wantTitle string, } } -// refuseUnsupportedKind rejects the two kinds §17.1 names explicitly with a -// clear, specific error — distinct from "no match" and from "ambiguous" — so +// refuseUnsupportedKind rejects the two unsupported kinds with a clear, +// specific error — distinct from "no match" and from "ambiguous" — so // an operator who names a recording or datasource-managed rule learns why, // not just that nothing matched. func refuseUnsupportedKind(name string, d Definition) (Definition, error) { @@ -174,8 +174,7 @@ func refuseUnsupportedKind(name string, d Definition) (Definition, error) { } // noMatchError reports a no-match with the count of rules the gate could see -// and, per Context decision 4, case-insensitive substring matches in place of -// the source plan's cut Levenshtein suggestions (§17.2). +// and case-insensitive substring matches as suggestions. func noMatchError(defs []Definition, name, wantTitle string) error { msg := fmt.Sprintf("no rule matched %q (%d rules available; run 'grafana-alertcheck list' to see titles)", name, len(defs)) @@ -194,8 +193,8 @@ func noMatchError(defs []Definition, name, wantTitle string) error { } // ambiguousError lists every candidate with its folder, its group, and the -// full copyable Folder/Group/Title (§17.1) — including the uid: form, which -// resolves unambiguously on the next attempt. +// full copyable Folder/Group/Title — including the uid: form, which resolves +// unambiguously on the next attempt. func ambiguousError(name string, candidates []Definition) error { sorted := append([]Definition(nil), candidates...) sort.Slice(sorted, func(i, j int) bool { return sorted[i].UID < sorted[j].UID }) diff --git a/grafana-alertcheck/internal/gate/resolve_test.go b/grafana-alertcheck/internal/gate/resolve_test.go index 192ed3dc2..d58ec73bd 100644 --- a/grafana-alertcheck/internal/gate/resolve_test.go +++ b/grafana-alertcheck/internal/gate/resolve_test.go @@ -143,8 +143,8 @@ func TestResolve_RejectsEmptySegments(t *testing.T) { } func TestResolve_UIDEmptySuffix(t *testing.T) { - // ruler_datasource_managed.json's only rule has UID == "" (P1.3: this - // shape has no uid at all). "uid:" with an empty suffix must not match it + // ruler_datasource_managed.json's only rule has UID == "" — that shape has + // no uid at all. "uid:" with an empty suffix must not match it // — that would report the misleading "datasource-managed rule, not // supported" for what is really a typo'd/empty uid. defs, err := ParseDefinitions(readFixture(t, "ruler_datasource_managed.json")) @@ -216,8 +216,7 @@ func TestResolve_UnsupportedHomonymResolvesSupportedSilently(t *testing.T) { func TestResolve_CollapseByUIDGivesNoteNotError(t *testing.T) { defs := rulerDefs(t) // The bare title and its Folder/Group/Title spelling both name the same - // rule (rule0000007) — a duplicate-name copy mistake, not an error - // (§17.3). + // rule (rule0000007) — a duplicate-name copy mistake, not an error. resolved, notes, err := Resolve(defs, []string{ "example_workflow_paused_rule", "ExampleObservability/Example Auth Production/example_workflow_paused_rule", @@ -233,9 +232,8 @@ func TestResolve_CollapseByUIDGivesNoteNotError(t *testing.T) { } } -// §22.2: "the same rule with two identical names ... must collapse to one -// rule" — the literal exact-duplicate-string case, distinct from the -// different-spellings case above. +// The same rule named twice with the identical string must collapse to one +// rule — distinct from the different-spellings case above. func TestResolve_IdenticalDuplicateNameCollapsesWithNote(t *testing.T) { defs := rulerDefs(t) resolved, notes, err := Resolve(defs, []string{ @@ -264,7 +262,7 @@ func TestResolve_MinObservedCountIsPostCollapse(t *testing.T) { if err != nil { t.Fatalf("Resolve: unexpected error: %v", err) } - // §17.3: the default MinObserved must come from len(resolved) (2 distinct + // The default MinObserved must come from len(resolved) (2 distinct // rules) — never len(names) (3 input lines), which would be unsatisfiable. if len(resolved) != 2 { t.Fatalf("resolved = %+v, want 2 distinct rules after collapse", resolved) diff --git a/grafana-alertcheck/internal/gate/schedule.go b/grafana-alertcheck/internal/gate/schedule.go index 0417af6a9..4541317e5 100644 --- a/grafana-alertcheck/internal/gate/schedule.go +++ b/grafana-alertcheck/internal/gate/schedule.go @@ -8,35 +8,30 @@ import ( "time" ) -// SkewHardLimit is one of §5's filled-in values (basis: §16; §22.11 asserts -// 120s errors, 30s does not). Defined here, in schedule.go's named-constants -// block, per §5's instruction — it moved out of source.go now that P4 exists; -// P2 needed it before this file did, so it started there. Exported (P10) so -// the CLI can report it verbatim next to a measured skew instead of keeping -// its own mirrored copy. +// SkewHardLimit is the largest clock skew between this runner and Grafana that +// a run tolerates before it errors out. Exported so the CLI can report it +// verbatim next to a measured skew instead of keeping its own mirrored copy. const SkewHardLimit = 60 * time.Second // fromFutureTolerance is how far ahead of the runner's own clock a supplied -// `from` may sit before check refuses it (§7: "from in the future, more than -// the skew tolerance — error"). §7 names no number, so this is the judgment -// call §5's table records: the same 60s as SkewHardLimit, because the only -// legitimate reason for a `from` in the future is clock disagreement between -// the deploy step and the check step, and that is bounded by the same figure. -// It is once-per-run input validation, not a per-rule coverage check, so -// Check applies it (P9) and proveCoverage does not. +// `from` may sit before check refuses it: the same 60s as SkewHardLimit, +// because the only legitimate reason for a `from` in the future is clock +// disagreement between the deploy step and the check step, and that is bounded +// by the same figure. It is once-per-run input validation, not a per-rule +// coverage check, so Check applies it and proveCoverage does not. const fromFutureTolerance = 60 * time.Second -// minDrainTimeout is §5's floor on drainTimeout: max(2 x max(intervalSeconds), -// 2m). Without the floor, a fleet of very tight rules would derive a -// drainTimeout too short to let a healthy in-flight poll land. +// minDrainTimeout is the floor on drainTimeout, which is otherwise +// 2 x max(intervalSeconds). Without the floor, a fleet of very tight rules +// would derive a drainTimeout too short to let a healthy in-flight poll land. const minDrainTimeout = 2 * time.Minute -// graceWarnFraction is §13.2's threshold for warning that transitionGrace eats -// too much of the requested window: "approximately one quarter of the window". +// graceWarnFraction is the share of the requested window above which +// transitionGrace is worth warning about. const graceWarnFraction = 0.25 -// ruleTimings groups the per-rule threshold values §5/§10.1/§14.1 derive from -// a rule's poll cadence and its own evaluation interval. +// ruleTimings groups the per-rule thresholds derived from a rule's poll +// cadence and its own evaluation interval. type ruleTimings struct { pollEvery time.Duration maxGap time.Duration @@ -45,26 +40,24 @@ type ruleTimings struct { } // globalTimings groups the values that apply to the whole run rather than to -// one rule: §13.1's transitionGrace and §19's drainTimeout are each derived -// once, across every non-skipped watched rule, not per rule. +// one rule: transitionGrace and drainTimeout are each derived once, across +// every non-skipped watched rule, not per rule. type globalTimings struct { transitionGrace time.Duration - // graceSource names, and already carries the `for` value of, the rule - // that set transitionGrace (§13.2 requires printing both) — one string - // field rather than a second (rule, duration) pair, matching this - // struct's fixed shape. "none" when no rule contributed (transitionGrace - // is then 0). + // graceSource names, and already carries the `for` value of, the rule that + // set transitionGrace — one string field rather than a second + // (rule, duration) pair, matching this struct's fixed shape. "none" when no + // rule contributed (transitionGrace is then 0). graceSource string drainTimeout time.Duration } // newRuleTimings derives one rule's thresholds from its fully-resolved poll -// cadence and its evaluation interval (§5, §10.1, §14.1). pollEvery arrives -// already resolved for the caller's mode — the §5 default, the operator's -// --poll-interval override, or (in log mode, a later phase) the cadence -// recorded in the log header. Deriving pollEvery inline here, instead of -// accepting it as an input, would let a caller in the wrong mode compute -// maxGap against the wrong authority — see the "Two authorities" note in P5. +// cadence and its evaluation interval. pollEvery arrives already resolved for +// the caller's mode — the default, the operator's --poll-interval override, or +// (in log mode) the cadence recorded in the log header. Deriving pollEvery +// inline here, instead of accepting it as an input, would let a caller in the +// wrong mode compute maxGap against the wrong authority. func newRuleTimings(pollEvery time.Duration, intervalSeconds int) ruleTimings { interval := time.Duration(intervalSeconds) * time.Second maxGap := 2 * pollEvery @@ -77,20 +70,20 @@ func newRuleTimings(pollEvery time.Duration, intervalSeconds int) ruleTimings { } } -// defaultPollEvery is §5's default per-rule cadence: half the rule's own +// defaultPollEvery is the default per-rule cadence: half the rule's own // evaluation interval. func defaultPollEvery(intervalSeconds int) time.Duration { return time.Duration(intervalSeconds) * time.Second / 2 } -// DeriveTimings computes every resolved rule's ruleTimings, keyed by UID, -// plus the shared globalTimings, from resolved definitions and watch's -// optional --poll-interval override (0 = no override: use each rule's §5 -// default of half its own interval). Per §5.1, a supplied override is used -// verbatim for every rule and is never clamped down to the default even when -// it exceeds intervalSeconds/2 — that case is reported back as a note, not -// silently corrected or refused, because clamping would defeat the one knob -// §5.1 gives an operator for making a tight schedule fit. +// DeriveTimings computes every resolved rule's ruleTimings, keyed by UID, plus +// the shared globalTimings, from resolved definitions and watch's optional +// --poll-interval override (0 = no override: use each rule's default of half +// its own interval). A supplied override is used verbatim for every rule and is +// never clamped down to the default even when it exceeds intervalSeconds/2 — +// that case is reported back as a note, not silently corrected or refused, +// because clamping would defeat the one knob an operator has for making a tight +// schedule fit. func DeriveTimings(defs []Definition, override time.Duration) (rules map[string]ruleTimings, global globalTimings, notes []string) { rules = make(map[string]ruleTimings, len(defs)) for _, d := range defs { @@ -124,14 +117,14 @@ func pausedSet(defs []Definition) map[string]bool { return paused } -// DeriveTimingsFromLog is DeriveTimings' log-mode counterpart, and the two -// authorities of P5 are the whole reason it exists as a separate function. -// pollEvery comes from the header — the cadence the recording ACTUALLY used, -// after any --poll-interval override — and maxGap and healthGrace follow from -// it. Re-deriving pollEvery from defs here would compare gaps recorded at the -// override cadence against thresholds computed from the default: exit 2 on a -// clean window when the override was slower, and, worse, a real recorder gap -// passing silently when it was faster. +// DeriveTimingsFromLog is DeriveTimings' log-mode counterpart, and having one +// authority for the cadence is the whole reason it exists as a separate +// function. pollEvery comes from the header — the cadence the recording +// ACTUALLY used, after any --poll-interval override — and maxGap and +// healthGrace follow from it. Re-deriving pollEvery from defs here would +// compare gaps recorded at the override cadence against thresholds computed +// from the default: exit 2 on a clean window when the override was slower, and, +// worse, a real recorder gap passing silently when it was faster. // // evalStaleAfter still comes from defs (2 x intervalSeconds): it is a property // of the rule's own evaluation cadence and is unaffected by how often the gate @@ -151,12 +144,12 @@ func pausedSet(defs []Definition) map[string]bool { // // It checks only the header-to-defs direction. The opposite direction — a // resolved definition absent from the header — is NOT this function's to -// judge: it is §19.1 step 3's log-identity validation, and it belongs to P9's -// Check, which is the only caller that knows both sets and can name the -// mismatch. Without that check a definition simply gets no timings entry, and -// a downstream lookup would read a zero maxGap: fail-closed (every gap -// exceeds it) but silent, so P9 must reject the set mismatch by name rather -// than let a rule fail for an unexplained reason. +// judge: it belongs to Check's log-identity validation, the only caller that +// knows both sets and can name the mismatch. Without that check a definition +// simply gets no timings entry, and a downstream lookup would read a zero +// maxGap: fail-closed (every gap exceeds it) but silent, so Check must reject +// the set mismatch by name rather than let a rule fail for an unexplained +// reason. func DeriveTimingsFromLog(h Header, defs []Definition) (rules map[string]ruleTimings, global globalTimings, err error) { byUID := make(map[string]Definition, len(defs)) for _, d := range defs { @@ -187,15 +180,13 @@ func DeriveTimingsFromLog(h Header, defs []Definition) (rules map[string]ruleTim return rules, deriveGlobalTimings(defs, h.pausedAtStart()), nil } -// deriveGlobalTimings computes transitionGrace and drainTimeout over defs -// (§5, §13.1, §19). +// deriveGlobalTimings computes transitionGrace and drainTimeout over defs. // -// A rule paused before the window opened — skipped, §12 — is excluded from the +// A rule paused before the window opened — skipped — is excluded from the // transitionGrace max: its `for` value can never fire during the window, so // counting it would only inflate the wait past what any watched rule actually -// needs (a judgment call the v2 plan makes explicitly for this formula; §19's -// drainTimeout carries no such exclusion, so it still runs over every resolved -// rule). +// needs. drainTimeout carries no such exclusion and still runs over every +// resolved rule. // // "Before the window opened" is the whole content of that exclusion, so the // authority is pausedAtStart and NEVER Definition.IsPaused: in log mode the @@ -226,9 +217,9 @@ func deriveGlobalTimings(defs []Definition, pausedAtStart map[string]bool) globa return g } -// Scheduler drives one per-rule schedule, never a global cycle (§5): a rule -// at intervalSeconds=10 alongside twenty at 300 keeps its own 5s cadence -// without forcing the same cadence onto the other twenty. +// Scheduler drives one per-rule schedule, never a global cycle: a rule at +// intervalSeconds=10 alongside twenty at 300 keeps its own 5s cadence without +// forcing the same cadence onto the other twenty. type Scheduler struct { next map[string]time.Time every map[string]time.Duration @@ -236,9 +227,9 @@ type Scheduler struct { // NewScheduler builds a Scheduler over per-rule cadences (keyed by UID), // staggering each rule's initial next-due time across [0, pollEvery) so the -// fleet does not start phase-aligned (§5's burst-bound proof depends on this: -// an already-staggered fleet only re-aligns by chance, briefly, not by -// construction). +// fleet does not start phase-aligned. The burst bound CheckBudget enforces +// depends on that: an already-staggered fleet only re-aligns by chance, +// briefly, not by construction. // // It takes cadences rather than whole ruleTimings on purpose: a scheduler // decides when to poll and nothing else, so it must not be handed maxGap, @@ -262,12 +253,12 @@ func NewScheduler(every map[string]time.Duration, now time.Time) *Scheduler { } // Due returns the UIDs whose next-due time has arrived, earliest-due-first. -// Ties (equal next-due time) break by tightest cadence first: the burst-bound -// proof in §5 assumes a newly-due tight rule waits at most for one in-flight -// request, which only holds if a simultaneous batch serves the tightest rule -// ahead of slacker ones. A tie-break that instead followed map iteration -// order would silently void that proof — nothing else would fail until a -// phase-aligned fleet opened a mid-run gap in production. +// Ties (equal next-due time) break by tightest cadence first: the burst bound +// assumes a newly-due tight rule waits at most for one in-flight request, which +// only holds if a simultaneous batch serves the tightest rule ahead of slacker +// ones. A tie-break that instead followed map iteration order would silently +// void that assumption — nothing else would fail until a phase-aligned fleet +// opened a mid-run gap in production. func (s *Scheduler) Due(now time.Time) []string { var due []string for uid, t := range s.next { @@ -303,11 +294,10 @@ func (s *Scheduler) Mark(uid string, now time.Time) error { } // earliestDue returns the earliest scheduled next-due time, and false when the -// scheduler holds no rules at all. The recorder's loop (P6) waits exactly that -// long instead of waking on a fixed tick: a fixed tick either polls a slack -// rule early — spending request budget the §5 formulas already accounted for — -// or wakes too late for the tightest rule and opens a gap inside its own -// maxGap. +// scheduler holds no rules at all. The recorder's loop waits exactly that long +// instead of waking on a fixed tick: a fixed tick either polls a slack rule +// early — spending request budget the schedule already accounted for — or wakes +// too late for the tightest rule and opens a gap inside its own maxGap. func (s *Scheduler) earliestDue() (time.Time, bool) { var earliest time.Time ok := false @@ -320,23 +310,21 @@ func (s *Scheduler) earliestDue() (time.Time, bool) { return earliest, ok } -// CheckBudget applies §5's error-at-start check to a fully resolved schedule. -// t and measured are both keyed by rule UID; measured must carry every UID in -// t; a rule this run never measured can't have its budget proved, and a -// silent zero-duration default would be exactly the kind of pass-on-an- -// unproven-window bug §5 exists to catch. CheckBudget fails when any of three -// conditions holds (sanity-checked against §22.3's mixed-interval regression -// in this phase's tests): +// CheckBudget proves at start that a fully resolved schedule can actually be +// served. t and measured are both keyed by rule UID, and measured must carry +// every UID in t: a rule this run never measured cannot have its budget +// proved, and a silent zero-duration default would be exactly the kind of +// pass-on-an-unproven-window bug this check exists to catch. It fails when any +// of three conditions holds: // // - utilization: the long-run request rate exceeds what concurrency serves; // - a single rule's own request cannot fit inside its own cadence; -// - the burst bound: the slowest measured request is slower than the -// fleet's tightest cadence, which — even under earliest-due-first -// ordering — can open a mid-run gap bigger than that rule's maxGap. +// - the burst bound: the slowest measured request is slower than the fleet's +// tightest cadence, which — even under earliest-due-first ordering — can +// open a mid-run gap bigger than that rule's maxGap. // -// The message never suggests a single interval (§5.1) — only the three -// controls an operator actually has: concurrency, poll-interval, and the -// alert list. +// The message names only the three controls an operator actually has: +// concurrency, poll-interval, and the alert list. func CheckBudget(t map[string]ruleTimings, measured map[string]time.Duration, concurrency int) error { if len(t) == 0 { return nil @@ -406,10 +394,11 @@ func CheckBudget(t map[string]ruleTimings, measured map[string]time.Duration, co return fmt.Errorf("%s", b.String()) } -// StartupSummary formats §13.2's required pre-run print: the total planned -// run time and the rule (with its `for` value) that set transitionGrace, plus -// a warning when the grace eats more than graceWarnFraction of the requested -// window. from/to are the requested classification window. +// StartupSummary formats the pre-run print an operator sees before the wait: +// the total planned run time and the rule (with its `for` value) that set +// transitionGrace, plus a warning when the grace eats more than +// graceWarnFraction of the requested window. from/to are the requested +// classification window. func StartupSummary(from, to time.Time, global globalTimings) (summary, warning string) { window := to.Sub(from) total := window + global.transitionGrace + global.drainTimeout diff --git a/grafana-alertcheck/internal/gate/schedule_test.go b/grafana-alertcheck/internal/gate/schedule_test.go index aee22f37f..70cc8817c 100644 --- a/grafana-alertcheck/internal/gate/schedule_test.go +++ b/grafana-alertcheck/internal/gate/schedule_test.go @@ -69,7 +69,7 @@ func TestDeriveTimings_TransitionGraceExcludesSkippedRule(t *testing.T) { } } -// §22.2/§22.3: `for: 1d` and `for: 1w` parse correctly (parse_ruler_test.go), +// `for: 1d` and `for: 1w` parse correctly (parse_ruler_test.go), // but that alone never proves they flow into transitionGrace — a Prometheus // duration parser that silently truncated to time.Duration's other units, or // a transitionGrace derivation that only ever saw hand-built values, could @@ -148,7 +148,7 @@ func TestDeriveTimingsFromLog_TransitionGraceFollowsTheHeaderNotTheDefinition(t }) t.Run("drainTimeout counts every rule either way", func(t *testing.T) { - // §19 puts no pause exclusion on drainTimeout, so both headers give the + // drainTimeout carries no pause exclusion, so both headers give the // same floor-bound value. for _, pausedAtStart := range []bool{false, true} { h := Header{Rules: []LoggedRule{loggedRule("r1", pausedAtStart)}} @@ -192,9 +192,8 @@ func TestDeriveTimings_DrainTimeoutAboveFloor(t *testing.T) { } } -// TestScheduler_DueOrderingTiesBreakByTightestCadence pins the ordering -// invariant the burst bound depends on (§5): when several rules become due at -// the exact same instant, Due must serve the tightest cadence first, not +// The ordering invariant the burst bound depends on: when several rules become +// due at the exact same instant, Due must serve the tightest cadence first, not // whatever order the underlying map happens to iterate in. A refactor that // loses this ordering must fail here, not in a production phase-aligned gap. func TestScheduler_DueOrderingTiesBreakByTightestCadence(t *testing.T) { @@ -261,8 +260,8 @@ func TestScheduler_MarkUnknownUIDFails(t *testing.T) { } // TestScheduler_PerRuleCadenceOverTime simulates a run and counts how often -// each rule comes due, pinning §5's core claim: schedules are per rule, never -// a global cycle. A tight rule must be polled at its own cadence regardless +// each rule comes due: schedules are per rule, never a global cycle. A tight +// rule must be polled at its own cadence regardless // of what slower rules in the same fleet need, and a slack rule must never be // forced onto the tight rule's cadence. func TestScheduler_PerRuleCadenceOverTime(t *testing.T) { @@ -350,10 +349,8 @@ func TestScheduler_EarliestDuePicksMinimum(t *testing.T) { } } -// TestCheckBudget_MixedIntervalRegression is §22.3's sanity check from the -// plan: one rule at 10s beside twenty at 300s, all measured ~1.8s, must not -// error at any reasonable concurrency — the exact case a naive worst-case-slot -// simulation would wrongly fail. +// One rule at 10s beside twenty at 300s, all measured ~1.8s, must not error at +// any reasonable concurrency — the exact case a naive worst-case-slot func TestCheckBudget_MixedIntervalRegression(t *testing.T) { timings := map[string]ruleTimings{"tight": {pollEvery: 5 * time.Second}} measured := map[string]time.Duration{"tight": 1800 * time.Millisecond} @@ -459,9 +456,9 @@ func TestCheckBudget_NonPositivePollIntervalIsAnError(t *testing.T) { } } -// assertBudgetMessage checks §5.1's required message contents: a measured -// duration is present, and all three controls are named — never a single -// suggested interval. +// assertBudgetMessage checks the message contents: a measured duration is +// present, and all three controls are named — never a single suggested +// interval. func assertBudgetMessage(t *testing.T, msg string) { t.Helper() for _, want := range []string{"measured", "concurrency", "poll-interval", "fewer"} { @@ -491,13 +488,10 @@ func TestStartupSummary_WarningWhenGraceTooLarge(t *testing.T) { } } -// §22.3: "a rule with for: 15m in a 10-minute window gives the warning about -// a large grace period" — no such rule exists in the real capture -// (testdata/README.md), so the test above pins the mechanism with a -// hand-built globalTimings. This drives the same warning off the real -// ruler_rules.json fixture's for:1w rule instead, tying ParseDefinitions and -// DeriveTimings into the warning end to end, not just the warning formula in -// isolation. +// The test above pins the warning formula with a hand-built globalTimings. +// This drives the same warning off the real ruler_rules.json fixture's for:1w +// rule instead, tying ParseDefinitions and DeriveTimings into the warning end +// to end. func TestStartupSummary_RealForOneWeekRuleTriggersWarning(t *testing.T) { defs := rulerDefs(t) _, global, notes := DeriveTimings(defs, 0) diff --git a/grafana-alertcheck/internal/gate/source.go b/grafana-alertcheck/internal/gate/source.go index d440d4938..82dcdc392 100644 --- a/grafana-alertcheck/internal/gate/source.go +++ b/grafana-alertcheck/internal/gate/source.go @@ -22,8 +22,8 @@ import ( // across retries. const maxResponseBytes = 25 << 20 // 25 MiB -// Clock is the seam that lets tests advance time without sleeping (§22) — the -// only two operations the gate ever needs from a clock. +// Clock is the seam that lets tests advance time without sleeping — the only +// two operations the gate ever needs from a clock. type Clock interface { Now() time.Time After(d time.Duration) <-chan time.Time @@ -37,9 +37,9 @@ func (SystemClock) After(d time.Duration) <-chan time.Time { return time.After(d // Observation is one successful poll of the state endpoint for a single rule. type Observation struct { - Rules []StateRule // may be empty — an authoritative 2xx saying the rule is absent (§14.5) - GrafanaNow time.Time // the Date header — H4 - Skew time.Duration // serverDate - (t_send+t_headers)/2, signed (§16) + Rules []StateRule // may be empty — an authoritative 2xx saying the rule is absent + GrafanaNow time.Time // the response's Date header + Skew time.Duration // serverDate - (t_send+t_headers)/2, signed SkewBound time.Duration // (t_headers-t_send)/2 — RTT/2 to the response headers Latency time.Duration // t_send through the full body read — see requestResult.Latency } @@ -48,7 +48,7 @@ type Observation struct { // network failure, or a body that failed to parse. It is never a deleted rule // (an authoritative 2xx with no matching rule is not this) and never a clock // problem (a missing/unparseable Date header or an out-of-bounds skew is a -// hard error instead — see doRequest). Never conflate them (§14.5). +// hard error instead — see doRequest). Never conflate them. type TransportError struct { Err error Status int // 0 when the failure never got a status (network/transport failure) @@ -67,10 +67,10 @@ func (e *TransportError) Unwrap() error { return e.Err } // too many sequential *TransportError failures. It deliberately does not // implement Unwrap into the underlying *TransportError: once retries are // exhausted the result is a hard, terminal failure, and -// errors.AsType[*TransportError] must never re-classify it as retryable — -// that is the exact conflation §19.3 case 1 forbids. Cause is still exposed -// as a plain field (and folded into Error()'s text) so a caller can log or -// inspect it; it just cannot flow back into the retry classification. +// errors.AsType[*TransportError] must never re-classify it as retryable. +// Cause is still exposed as a plain field (and folded into Error()'s text) so +// a caller can log or inspect it; it just cannot flow back into the retry +// classification. type RetryExhaustedError struct { Failures int Cause error @@ -81,7 +81,7 @@ func (e *RetryExhaustedError) Error() string { } // Source is everything the gate reads from Grafana. httpSource is the one -// production implementation; every later phase's tests use a scripted fake +// production implementation; the tests use a scripted fake // (source_fake_test.go) instead of real HTTP. type Source interface { Version(ctx context.Context) (string, error) @@ -137,7 +137,7 @@ func parseGrafanaVersion(s string) (grafanaVersion, error) { } // supportedGrafanaMin and supportedGrafanaMax bound the platform this gate is -// verified against (§2.7 control 2, §21.5): >= 13.0.0, < 14.0.0. +// verified against: >= 13.0.0, < 14.0.0. var ( supportedGrafanaMin = grafanaVersion{13, 0, 0} supportedGrafanaMax = grafanaVersion{14, 0, 0} // exclusive @@ -145,8 +145,9 @@ var ( // CheckGrafanaVersion enforces the supported range. An unparseable or // out-of-range version is a hard error naming both what was found and what is -// supported — trusting an unverified schema is exactly the deprecation risk -// §2.7 control 2 exists to catch. +// supported: the response schemas this gate parses are only verified against +// that range, and trusting an unverified one is how a deprecation turns into a +// silent misread. func CheckGrafanaVersion(version string) error { v, err := parseGrafanaVersion(version) if err != nil { @@ -160,12 +161,12 @@ func CheckGrafanaVersion(version string) error { return nil } -// httpSource is the production Source: stdlib net/http only, bearer auth -// from a token supplied at construction (the caller reads it from the -// environment — §20.2 — this type never touches env itself), and manual -// strict decoding via ParseState/ParseDefinitions (H1). The retry limit and -// backoff parameters are struct fields with production defaults set here, -// not package constants, so a test can shrink them without a hook. +// httpSource is the production Source: stdlib net/http only, bearer auth from +// a token supplied at construction (the caller reads it from the environment; +// this type never touches env itself), and manual strict decoding via +// ParseState/ParseDefinitions. The retry limit and backoff parameters are +// struct fields with production defaults set here, not package constants, so a +// test can shrink them without a hook. type httpSource struct { baseURL string token string @@ -177,9 +178,8 @@ type httpSource struct { backoffCap time.Duration } -// NewHTTPSource builds the production Source. token is never logged and -// never enters an error string (§20.2) — it is used only to set the -// Authorization header. +// NewHTTPSource builds the production Source. token is never logged and never +// enters an error string — it is used only to set the Authorization header. func NewHTTPSource(baseURL, token string, clock Clock) Source { return &httpSource{ baseURL: strings.TrimSuffix(baseURL, "/"), @@ -236,8 +236,8 @@ func (s *httpSource) RuleState(ctx context.Context, title string) (Observation, if parseErr != nil { // Treated as transient, not a schema break: an unparseable 2xx // is far more likely a mid-stream hiccup than a permanent shape - // change, and H1's strict parser already turns a real shape - // change into a loud per-field error the moment it's visible. + // change, and the strict parser already turns a real shape change + // into a loud per-field error the moment it's visible. return Observation{}, &TransportError{Err: fmt.Errorf("parse rule state: %w", parseErr)} } return Observation{ @@ -252,33 +252,32 @@ func (s *httpSource) RuleState(ctx context.Context, title string) (Observation, // requestResult is the outcome of one successful HTTP attempt in doRequest: // the raw body plus everything derived from timing the round trip against -// the response's own clock (§16). +// the response's own clock. type requestResult struct { Body []byte - ServerDate time.Time // the Date header — H4 + ServerDate time.Time // the response's Date header Skew time.Duration // serverDate - (t_send+t_headers)/2, signed SkewBound time.Duration // (t_headers-t_send)/2 — RTT/2 to the response headers - // Latency spans t_send through the full body read (§5.2's budget check - // needs the whole poll's wall time, or a schedule feasibility check that - // only sees header latency goes optimistic — fail-open). It does not - // include the caller's subsequent JSON parse (ParseState/ParseDefinitions - // run outside doRequest); if P4's budget accounting needs parse time - // folded in too, extend here rather than approximating it at the call - // site. + // Latency spans t_send through the full body read: the budget check needs + // the whole poll's wall time, or a schedule feasibility check that only + // sees header latency goes optimistic — fail-open. It does not include the + // caller's subsequent JSON parse (ParseState/ParseDefinitions run outside + // doRequest); if the budget accounting ever needs parse time folded in too, + // extend here rather than approximating it at the call site. Latency time.Duration } -// doRequest performs one HTTP GET and classifies the outcome (§14.5, §16): -// a network failure, a non-2xx status, or a body-read failure is retryable +// doRequest performs one HTTP GET and classifies the outcome: a network +// failure, a non-2xx status, or a body-read failure is retryable // (*TransportError); a missing or unparseable Date header, or a skew beyond // SkewHardLimit, is a hard error — retrying can never fix either, so neither -// may enter the backoff loop (H4). +// may enter the backoff loop. // // The Date-header/skew check runs for every endpoint this hits, including -// /api/health — broader than §16's own scope, which only discusses the state -// endpoint. Deliberate: a skewed clock discovered only once RuleState starts -// polling is a skew that has already masked whatever /api/health and the -// ruler read reported; failing closed at the first response catches it +// /api/health, and not only the state endpoint whose timestamps the gate +// actually compares. Deliberate: a skewed clock discovered only once RuleState +// starts polling is a skew that has already masked whatever /api/health and +// the ruler read reported; failing closed at the first response catches it // before any of that is trusted, and every response comes with a Date header // for free. func (s *httpSource) doRequest(ctx context.Context, path string) (requestResult, error) { @@ -317,7 +316,7 @@ func (s *httpSource) doRequest(ctx context.Context, path string) (requestResult, dateHeader := resp.Header.Get("Date") if dateHeader == "" { - return requestResult{}, fmt.Errorf("%s: response has no Date header (H4)", path) + return requestResult{}, fmt.Errorf("%s: response has no Date header", path) } serverDate, parseErr := http.ParseTime(dateHeader) if parseErr != nil { @@ -332,7 +331,7 @@ func (s *httpSource) doRequest(ctx context.Context, path string) (requestResult, absSkew = -absSkew } if absSkew > SkewHardLimit { - return requestResult{}, fmt.Errorf("%s: clock skew %s exceeds hard limit %s (§16)", path, absSkew, SkewHardLimit) + return requestResult{}, fmt.Errorf("%s: clock skew %s exceeds hard limit %s", path, absSkew, SkewHardLimit) } return requestResult{Body: b, ServerDate: serverDate, Skew: signedSkew, SkewBound: bound, Latency: latency}, nil @@ -341,9 +340,9 @@ func (s *httpSource) doRequest(ctx context.Context, path string) (requestResult, // retryTransport runs fn, retrying with backoff only while it fails with a // *TransportError — any other error is a hard error and returns immediately, // never retried. failures counts consecutive *TransportError results; -// exceeding maxFailures gives up with a wrapped hard error (§19.3 case 1). -// The wait between attempts goes through clock.After so a test with a fake -// Clock never sleeps on real time (§22). +// exceeding maxFailures gives up with a wrapped hard error. The wait between +// attempts goes through clock.After so a test with a fake Clock never sleeps on +// real time. func retryTransport[T any](ctx context.Context, clock Clock, maxFailures int, backoffBase, backoffCap time.Duration, fn func() (T, error)) (T, error) { var zero T failures := 0 @@ -368,7 +367,7 @@ func retryTransport[T any](ctx context.Context, clock Clock, maxFailures int, ba } // backoffDelay is 1s base, doubling per failure, capped at maxDelay, with -// ±20% jitter (§5's filled-in value for maxSequentialFailures). +// ±20% jitter. func backoffDelay(base, maxDelay time.Duration, failureCount int) time.Duration { d := base for i := 1; i < failureCount && d < maxDelay; i++ { diff --git a/grafana-alertcheck/internal/gate/source_fake_test.go b/grafana-alertcheck/internal/gate/source_fake_test.go index 40a2a8a05..97de92f02 100644 --- a/grafana-alertcheck/internal/gate/source_fake_test.go +++ b/grafana-alertcheck/internal/gate/source_fake_test.go @@ -9,15 +9,13 @@ import ( ) // fakeClock is a manually-advanced Clock — no test in this package sleeps on -// real time (§22). It is goroutine-safe (a concurrent fleet under -race must -// not trip on the double itself), but After always fires immediately, -// regardless of the requested duration or whether Advance was ever called. -// That is sufficient here: every retry/backoff test in this phase only needs -// to avoid a real sleep. It is NOT sufficient for a test that must prove a -// wait did not fire early — e.g. a P4 scheduler test asserting Due() doesn't -// return a rule before its next-due time. Use virtualClock below for that: it -// is the clock P6's recorder-loop tests needed, and it makes a wait and the -// passage of time the same event. +// real time. It is goroutine-safe (a concurrent fleet under -race must not trip +// on the double itself), but After always fires immediately, regardless of the +// requested duration or whether Advance was ever called. That is enough for the +// retry/backoff tests, which only need to avoid a real sleep. It is NOT enough +// for a test that must prove a wait did not fire early — e.g. asserting Due() +// does not return a rule before its next-due time. Use virtualClock below for +// that: it makes a wait and the passage of time the same event. type fakeClock struct { mu sync.Mutex now time.Time @@ -113,12 +111,11 @@ type scriptedObservation struct { err error } -// fakeSource is a scripted Source with no HTTP, goroutine-safe so a phase -// that polls several rules concurrently (P6) can share one instance across -// goroutines without tripping -race. P3 through at least P5 can construct -// one directly instead of talking to HTTP; a phase that needs it to behave -// like a live server under concurrent load beyond simple locking should -// verify that assumption rather than take this comment's word for it. +// fakeSource is a scripted Source with no HTTP, goroutine-safe so a test that +// polls several rules concurrently can share one instance across goroutines +// without tripping -race. A test that needs it to behave like a live server +// under concurrent load beyond simple locking should verify that assumption +// rather than take this comment's word for it. type fakeSource struct { mu sync.Mutex diff --git a/grafana-alertcheck/internal/gate/source_test.go b/grafana-alertcheck/internal/gate/source_test.go index 2998b4776..39181cfca 100644 --- a/grafana-alertcheck/internal/gate/source_test.go +++ b/grafana-alertcheck/internal/gate/source_test.go @@ -90,7 +90,7 @@ func TestCheckGrafanaVersion(t *testing.T) { } for _, want := range c.wantContains { if !strings.Contains(err.Error(), want) { - t.Errorf("CheckGrafanaVersion(%q): error %q does not mention %q (the plan requires naming both what was found and what is supported)", c.version, err.Error(), want) + t.Errorf("CheckGrafanaVersion(%q): error %q does not mention %q — it must name both what was found and what is supported", c.version, err.Error(), want) } } } @@ -170,7 +170,7 @@ func TestHTTPSource_RuleState_EmptyIsNotAnError(t *testing.T) { t.Fatalf("Rules = %+v, want empty (an authoritative 2xx is not a transport error)", obs.Rules) } if obs.GrafanaNow.IsZero() { - t.Fatalf("GrafanaNow is zero, want the response's Date header value (H4)") + t.Fatalf("GrafanaNow is zero, want the response's Date header value") } } @@ -274,7 +274,7 @@ func TestHTTPSource_MissingDateHeader(t *testing.T) { src := NewHTTPSource(srv.URL, "", clock) _, err := src.Version(context.Background()) if err == nil { - t.Fatalf("Version(): want error, got nil (H4: a missing Date header is a hard error)") + t.Fatalf("Version(): want error, got nil: a missing Date header is a hard error") } if calls.Load() != 1 { t.Fatalf("calls = %d, want 1 — a missing Date header must never be retried", calls.Load()) @@ -294,7 +294,7 @@ func TestHTTPSource_UnparseableDateHeader(t *testing.T) { src := NewHTTPSource(srv.URL, "", clock) _, err := src.Version(context.Background()) if err == nil { - t.Fatalf("Version(): want error, got nil (H4: an unparseable Date header is a hard error)") + t.Fatalf("Version(): want error, got nil: an unparseable Date header is a hard error") } if calls.Load() != 1 { t.Fatalf("calls = %d, want 1 — an unparseable Date header must never be retried", calls.Load()) @@ -343,14 +343,14 @@ func TestHTTPSource_ObservationTiming(t *testing.T) { t.Errorf("SkewBound = %v, want 1s (RTT/2 with a 2s round trip to headers)", obs.SkewBound) } if obs.Latency != 4*time.Second { - t.Errorf("Latency = %v, want 4s (send through full body read, §5.2) — not just the 2s header round trip", obs.Latency) + t.Errorf("Latency = %v, want 4s (send through full body read) — not just the 2s header round trip", obs.Latency) } }) } } -// §22.7/§16: a genuinely discriminating regression for "the gate compares -// staleness against the Date header, never the runner's clock." lastEvaluation +// A discriminating regression for "the gate compares staleness against the +// Date header, never the runner's clock". lastEvaluation // sits 100s behind Grafana's TRUE now (obs.GrafanaNow, from the Date header) // — under the 120s evalStaleAfter limit — but 130s behind the RUNNER's clock. // An implementation that leaked the runner's clock into the staleness @@ -564,8 +564,7 @@ func TestHTTPSource_NetworkFailureRetries(t *testing.T) { // it names how many failures it gave up after, and — the regression this // pins — it is never itself classified as a *TransportError. If it were, // something one layer up that also retries on *TransportError would treat an -// already-exhausted give-up as retryable again, the exact conflation §19.3 -// case 1 forbids. +// already-exhausted give-up as retryable again. func assertRetryExhausted(t *testing.T, err error, wantFailures int) { t.Helper() var reErr *RetryExhaustedError diff --git a/grafana-alertcheck/internal/gate/testdata/README.md b/grafana-alertcheck/internal/gate/testdata/README.md index fb437b3ae..1a6bf5877 100644 --- a/grafana-alertcheck/internal/gate/testdata/README.md +++ b/grafana-alertcheck/internal/gate/testdata/README.md @@ -1,6 +1,6 @@ # Fixture provenance -All fixtures are sanitized slices of the real Grafana 13.1.0 payloads captured next to the plan in +All fixtures are sanitized slices of real Grafana 13.1.0 payloads captured into `tmp/` (`tmp/state_all.json`, `tmp/ruler_all.json`, `tmp/health.json` — gitignored, never committed). Renames are consistent across files: the same real folder/rule keeps the same fake identity everywhere it appears (e.g. `folder0000002`/`rule0000002` is the same real paused rule in both @@ -23,45 +23,38 @@ here instead, for every fixture, for consistency. `rule0000002`/"Example Paused Rule". Unmodified: `isPaused:true`, zero `lastEvaluation`, `health:ok`, `state:inactive`, absent `alerts`/`labels`. - **state_health_error.json** — real `health:error` rule ("[JD] No Job Proposals", folder - `job-distributor`), highest priority per §22.1. Renamed to folder `ExampleService`/`folder0000003`, + `job-distributor`). Renamed to folder `ExampleService`/`folder0000003`, rule `rule0000003`/"Example No Data Source". Unmodified: `health:error`, `lastError` text, the single `Error` instance. - **state_health_nodata.json** — real `health:nodata` rule ("ARE test", folder `diegos_playground`). Renamed to folder `ExamplePlayground`/`folder0000004`, rule `rule0000004`/"Example NoData Rule". Unmodified: `health:nodata`, the single `NoData` instance. -- **state_reason_composite.json** — composite of two real instances combined under one rule for P1.2a - coverage: a real `"Normal (Error)"` instance (from a Flux-reconciliation rule; 14 of that state exist +- **state_reason_composite.json** — two real instances combined under one rule to cover composite + state parsing: a real `"Normal (Error)"` instance (from a Flux-reconciliation rule; 14 of that state exist in the capture) and a real `"Normal (NoData)"` instance (from a pod-liveness rule; 1091 of that state exist), plus one plain `"Normal"` instance for contrast. Renamed to folder `ExampleInfra`/`folder0000005`, rule `rule0000005`/"Example Composite Reasons". - **state_missing_optional.json** — derived from `state_one_instance.json`: `alerts`, `totals`, `totalsFiltered` and `labels` all removed. Must parse with `Instances=nil`, `Totals=nil`. -- **state_missing_health.json** — derived from `state_one_instance.json`: the required `health` key - removed. Must be a parse error (H1). -- **state_missing_lasteval.json** — derived from `state_one_instance.json`: the required - `lastEvaluation` key removed. Must be a parse error (H1). -- **state_missing_state.json** — derived from `state_one_instance.json`: the required rule-level - `state` key removed. Must be a parse error (H1). Closes must-error coverage for H1's four required - fields — a review pass found `health`/`lastEvaluation` covered but `state`/`interval` weren't, even - though the code already `req`'d them correctly. -- **state_missing_interval.json** — derived from `state_one_instance.json`: the required group-level - `interval` key removed. Must be a parse error (H1); same review-pass gap as above. +- **state_missing_health.json**, **state_missing_lasteval.json**, **state_missing_state.json**, + **state_missing_interval.json** — derived from `state_one_instance.json`, each with one of the four + required keys removed (`health`, `lastEvaluation`, rule-level `state`, group-level `interval`). Each + must be a parse error. - **state_missing_file.json** / **state_missing_name.json** — derived from `state_one_instance.json`: - the group-level `file`/`name` keys removed respectively. Not part of H1's four (those are `health`, - `state`, `lastEvaluation`, `interval`), but the code treats group identity as strict too, and the same - review pass flagged the gap — closed rather than deferred to a later §22 sweep since the fixture is - the same 10-line edit. + the group-level `file`/`name` keys removed respectively. Not among the four required fields above, + but the parser treats group identity as strict too. - **state_zerotime_unpaused.json** — derived from `state_one_instance.json`: `lastEvaluation` set to - the zero time while `isPaused` stays `false`. Must be a parse error (§2.3). + the zero time while `isPaused` stays `false`. Must be a parse error — only a paused rule may report + the zero time. - **state_unknown_state.json** — derived from `state_one_instance.json`: the instance state hand-edited to `"Weird (NoData)"`, a syntactically valid composite whose base isn't in the 5-value allowlist. Must - be a parse error (P1.2a). + be a parse error. - **state_only_active_instances.json** — derived from a real rule that genuinely had 1 `Alerting` + 22 `Normal` instances (`totals: {alerting:1, normal:22}`, rule `dfhp1t5pkosu8f`, folder `BCM`). `alerts[]` - trimmed to the single `Alerting` instance only, while `totals` is left **unchanged** — reproducing the - §3.2 violation shape (instance list says "only active" while totals disagrees). Renamed to folder - `ExampleTeam`/`folder0000001`, rule `rule0000006`. `ParseState` itself parses this fine; the §3.2 - verification lives in a later phase (P5/P9). + trimmed to the single `Alerting` instance only, while `totals` is left **unchanged** — the shape a + state endpoint that stopped returning normal instances would produce (the instance list says "only + active" while totals disagrees). Renamed to folder `ExampleTeam`/`folder0000001`, rule `rule0000006`. + `ParseState` itself parses this fine; `VerifyNormalInstancesVisible` is what rejects it. ## Ruler endpoint (`/api/ruler/grafana/api/v1/rules`) @@ -69,7 +62,7 @@ here instead, for every fixture, for consistency. - The real true 2-way title collision: namespace `CRE-BCM-Prod-Zone-A`, group `Gateway`, identical folder+group+title, distinct UIDs (`ffvabtvvbozcwf`/`efvabtwbxlvk0b`) — renamed to namespace `Example-Zone-A`, rules `rule0000006a`/`rule0000006b`, both titled "Example No Gateways Available". - Folder/Group/Title alone does **not** disambiguate this pair (§17, §22.2). + Folder/Group/Title alone does **not** disambiguate this pair; only `uid:` does. - The 3 real `is_paused:true` rules, renamed to `rule0000002`/`rule0000007`/`rule0000008`. `rule0000002` intentionally shares its identity (`folder0000002`) with `state_paused.json`. - A real `for:1d` rule (`afs438kjd4v7kd` → `rule0000009`). @@ -81,7 +74,7 @@ here instead, for every fixture, for consistency. Grafana represents a datasource-managed (native Prometheus-format) alerting rule. `ParseDefinitions` must classify it as `KindDatasourceManaged`, parse `Title` from `alert`, and leave `UID` empty (this shape has no uid at all — inventing one would be inventing shape) without rejecting the rule - (rejection is P3's job, only for rules a user actually named). + (rejection is `Resolve`'s job, and only for rules a user actually named). - **ruler_recording.json** — **DERIVED**, no recording rule exists in the capture (verified: 0 rules carry `grafana_alert.record`). Hand-built: a `grafana_alert` block with a `record` sub-object but deliberately *without* `no_data_state`/`exec_err_state`/`is_paused`/`intervalSeconds`/`namespace_uid` diff --git a/grafana-alertcheck/internal/gate/watch.go b/grafana-alertcheck/internal/gate/watch.go index f8726fec6..6f4b7fa15 100644 --- a/grafana-alertcheck/internal/gate/watch.go +++ b/grafana-alertcheck/internal/gate/watch.go @@ -14,7 +14,7 @@ import ( ) // DaemonChildFlag is the hidden flag the parent passes when it re-execs itself -// as the detached recorder (§4.4). It is deliberately absent from the CLI's +// as the detached recorder. It is deliberately absent from the CLI's // usage text: an operator never types it, and a child started by hand against // a log no parent prepared fails immediately on the header read. const DaemonChildFlag = "--daemon-child" @@ -26,7 +26,7 @@ const DaemonChildFlag = "--daemon-child" // loop — and not an assumption drawn from surviving a timer. A timer cannot // tell a healthy child from one that is about to die on a slow runner, and // getting that wrong means watch returns success over a recording that never -// happened (§4.3). +// happened. const ReadyFDFlag = "--ready-fd" // childReadyTimeout bounds that wait. Everything before the signal is local — @@ -43,7 +43,7 @@ const daemonLogTailBytes = 4096 // // It has no To field and must never gain one: watch writes the stopped // sentinel with its OWN stop time and makes no comparison against `to`, which -// only check knows (§4.5). Passing `to` here would give two components an +// only check knows. Passing `to` here would give two components an // opinion about the same comparison, and the recorder's opinion is the one // that cannot be trusted — it exits before the grace it would have to wait for. // @@ -55,18 +55,18 @@ const daemonLogTailBytes = 4096 // Header carries no States field for the same reason. type WatchConfig struct { // URL and Token are the connection details. The CLI reads both from the - // environment and never from a flag (§20.2); Token is never logged and - // never enters an error string. + // environment and never from a flag; Token is never logged and never + // enters an error string. URL, Token string - // Alerts are the operator-supplied names, one per line, in any of §17's - // forms. Empty lines are discarded by Resolve. + // Alerts are the operator-supplied names, one per line, in any of the forms + // Resolve accepts. Empty lines are discarded by Resolve. Alerts []string Folder string // Out is the JSONL log path. PidFile and DaemonLog default to // .pid and .daemon.log — the same convention check uses to find - // the recorder it must stop (P9), so nothing has to be wired by hand. + // the recorder it must stop, so nothing has to be wired by hand. Out string PidFile string DaemonLog string @@ -77,11 +77,10 @@ type WatchConfig struct { Until time.Time // PollEvery is the --poll-interval override, used verbatim for every rule - // and never clamped (§5.1). Zero means each rule polls at half its own - // evaluation interval. Whatever this resolves to is written into the header - // as the cadence actually used, and that header value — never a - // re-derivation from the definitions — is what check derives maxGap from - // (P5, "two authorities"). + // and never clamped. Zero means each rule polls at half its own evaluation + // interval. Whatever this resolves to is written into the header as the + // cadence actually used, and that header value — never a re-derivation from + // the definitions — is what check derives maxGap from. PollEvery time.Duration Concurrency int @@ -90,7 +89,7 @@ type WatchConfig struct { // Notes is where the parent prints what an operator has to see before the // deploy step runs: resolve notes, the cadence per rule, the rules it will // not wait for. nil discards them. The library prints nothing else — the - // CLI owns presentation (§20.2). + // CLI owns presentation. Notes io.Writer } @@ -138,20 +137,20 @@ func (cfg WatchConfig) validate() error { return nil } -// Watch is the record step's parent process (§4.3). It returns only once the -// window is genuinely being recorded: +// Watch is the record step's parent process. It returns only once the window +// is genuinely being recorded: // // version gate -> resolve definitions and names -> derive timings -> // open the log and write the header -> ONE observation of every non-skipped -// rule -> verify §3.2 -> check the schedule budget -> detach the child -> -// wait for the child to report that it is recording -> write the pidfile -> -// return. +// rule -> verify normal instances are visible -> check the schedule budget -> +// detach the child -> wait for the child to report that it is recording -> +// write the pidfile -> return. // // The first-observation wait is not a convenience. Returning before it would // leave the deploy inside [from, first_poll] with no evidence — the exact -// blind interval the two-phase model exists to remove — and it is also what -// surfaces auth, name-resolution and parse failures BEFORE deploy.sh runs -// rather than ten minutes later. +// blind interval the record-then-check split exists to remove — and it is +// also what surfaces auth, name-resolution and parse failures BEFORE +// deploy.sh runs rather than ten minutes later. func Watch(ctx context.Context, cfg WatchConfig) error { cfg = cfg.withDefaults() if err := cfg.validate(); err != nil { @@ -165,8 +164,8 @@ func Watch(ctx context.Context, cfg WatchConfig) error { } // Hand the log over with Close, never Stop: a sentinel here would tell - // check the recording ended before the child had even started (§4.5). - // Closing also releases the flock the child is about to take. + // check the recording ended before the child had even started. Closing + // also releases the flock the child is about to take. if err := prep.writer.Close(); err != nil { return err } @@ -181,8 +180,8 @@ func Watch(ctx context.Context, cfg WatchConfig) error { // The PARENT writes the pidfile, not the child: check must find the pid the // instant Watch returns, and a child writing its own would race the very - // next step of the pipeline. A deviation from P6's argv list, and the - // reason the child is never given --pidfile at all. + // next step of the pipeline. That is why the child is never given + // --pidfile at all. // // It is written only once the child has reported ready, so no path through // this function leaves a pidfile naming a process that is not recording. @@ -295,8 +294,8 @@ type preparedWatch struct { // prepareWatch is everything the parent does before it detaches. It takes a // Source rather than building one so the paused-rule, first-observation, -// §3.2 and budget behaviours are all testable with a scripted fake — only the -// process spawning needs a real binary. +// instance-visibility and budget behaviours are all testable with a scripted +// fake — only the process spawning needs a real binary. func prepareWatch(ctx context.Context, cfg WatchConfig, src Source) (*preparedWatch, error) { version, err := src.Version(ctx) if err != nil { @@ -325,7 +324,7 @@ func prepareWatch(ctx context.Context, cfg WatchConfig, src Source) (*preparedWa for _, d := range resolved { // A cadence of zero would make the child spin: every rule is due the // instant it was marked. It also cannot be written into the header, - // where check requires a positive value to derive maxGap from (P5). + // where check requires a positive value to derive maxGap from. if rt[d.UID].pollEvery <= 0 { return nil, fmt.Errorf("rule %q (%s) reports intervalSeconds=%d: there is no poll cadence to record at", d.Title, d.UID, d.IntervalSeconds) @@ -363,18 +362,18 @@ func openRecording(ctx context.Context, cfg WatchConfig, src Source, writer *Wri return nil, err } - // A rule whose DEFINITION says is_paused is skipped (§12): it is not - // waited for, not scheduled and never polled. Waiting for one either hangs - // forever or errors before the deploy (§4.3), and recording polls for it - // would report an in-window pause (coverage check 7) for a rule that was - // already paused when the window opened — turning §12's exit 1 into an - // exit 2. The header still names it, with is_paused true, so check reports - // it as skipped from the definitions. + // A rule whose DEFINITION says is_paused is skipped: it is not waited for, + // not scheduled and never polled. Waiting for one either hangs forever or + // errors before the deploy, and recording polls for it would report an + // in-window pause (coverage check 7) for a rule that was already paused + // when the window opened — turning a skipped rule's exit 1 into an exit 2. + // The header still names it, with is_paused true, so check reports it as + // skipped from the definitions. var active []Definition activeTimings := make(map[string]ruleTimings, len(resolved)) for _, d := range resolved { if d.IsPaused { - fmt.Fprintf(cfg.Notes, "note: rule %q (%s) is paused: recorded as skipped, not waited for (§4.3)\n", d.Title, d.UID) + fmt.Fprintf(cfg.Notes, "note: rule %q (%s) is paused: recorded as skipped, not waited for\n", d.Title, d.UID) continue } active = append(active, d) @@ -392,9 +391,9 @@ func openRecording(ctx context.Context, cfg WatchConfig, src Source, writer *Wri } } - // Budget last, on the latencies just measured — never on a fixed estimate - // (§5.2). Only the active rules count: a skipped rule is never polled and - // consumes none of the capacity. + // Budget last, on the latencies just measured — never on a fixed estimate. + // Only the active rules count: a skipped rule is never polled and consumes + // none of the capacity. if err := CheckBudget(activeTimings, measured, cfg.Concurrency); err != nil { return nil, err } @@ -404,9 +403,9 @@ func openRecording(ctx context.Context, cfg WatchConfig, src Source, writer *Wri // loggedRules snapshots the resolved definitions into the header's rule list. // Every field but PollEverySeconds is forensic — a resolve-time snapshot that -// makes an uploaded log self-describing (§21.3) — while PollEverySeconds is +// makes an uploaded log self-describing — while PollEverySeconds is // load-bearing: it is the cadence this recording actually used, and check -// derives maxGap from it rather than from the definitions (P5). +// derives maxGap from it rather than from the definitions. func loggedRules(defs []Definition, rt map[string]ruleTimings) []LoggedRule { out := make([]LoggedRule, 0, len(defs)) for _, d := range defs { @@ -427,17 +426,17 @@ func loggedRules(defs []Definition, rt map[string]ruleTimings) []LoggedRule { } // firstObservations takes one observation of every rule in active, verifies -// §3.2 against those very responses, and reduces each into the poll record -// that IS the window's first heartbeat — plus the measured latency of each, -// which is the only honest input to §5.2's budget check (a fixed estimate is -// worthless when one rule's payload is ~230x another's). +// that normal instances are visible in those very responses, and reduces each +// into the poll record that IS the window's first heartbeat — plus the measured +// latency of each, which is the only honest input to the budget check (a fixed +// estimate is worthless when one rule's payload is ~230x another's). // -// Both entry paths share it: watch's parent, before it detaches (§4.3), and +// Both entry paths share it: watch's parent, before it detaches, and // single-step check's measurement pass, which keeps the polls as evidence -// rather than writing them to a log (P9). Keeping one implementation is the -// point — the §3.2 verification and the "absent is a warning, not an error" -// rule are exactly the places where two copies would silently drift, and a -// drift in either direction is fail-open. +// rather than writing them to a log. Keeping one implementation is the point — +// the instance-visibility verification and the "absent is a warning, not an +// error" rule are exactly the places where two copies would silently drift, and +// a drift in either direction is fail-open. // // polls come back in `active` order, so a log written from them is byte-stable // for a given set of observations. @@ -456,7 +455,7 @@ func firstObservations(ctx context.Context, src Source, active []Definition, red return nil, nil, err } - // Verify §3.2 before anything downstream relies on it: if the state + // Verify this before anything downstream relies on it: if the state // endpoint ever stops returning normal instances, the reduction's "keep // the non-normal ones" silently becomes "keep everything it happened to // send" and the transition markers lose their ground truth. @@ -473,12 +472,12 @@ func firstObservations(ctx context.Context, src Source, active []Definition, red measured[d.UID] = obs.Latency poll := reducer.Reduce(d.UID, obs) if !poll.Found { - // Authoritative, not transient (P2 already retried transport - // failures): the rule resolved in the ruler API but the state - // endpoint does not serve it. Recorded as Found=false, which P7 - // check 8 turns into unobservable — a note rather than an error - // here, because the state endpoint can lag a freshly created rule - // and the coverage proof fails closed either way. + // Authoritative, not transient (the transport already retried + // every transient failure): the rule resolved in the ruler API but + // the state endpoint does not serve it. Recorded as Found=false, + // which the coverage proof turns into unobservable — a note rather + // than an error here, because the state endpoint can lag a freshly + // created rule and the coverage proof fails closed either way. fmt.Fprintf(notes, "warning: rule %q (%s) is absent from the state endpoint; recorded as not found\n", d.Title, d.UID) } polls = append(polls, poll) @@ -488,8 +487,9 @@ func firstObservations(ctx context.Context, src Source, active []Definition, red // observeAll polls every rule in uids concurrently, bounded by concurrency, // and returns one Observation per rule that answered. Every rule is polled by -// TITLE (the ?rule_name= filter, §2.8) and selected out of the response by -// UID (§14.5) — a filtered response can carry several rules sharing one title. +// TITLE (the ?rule_name= filter is a title filter) and selected out of the +// response by UID — a filtered response can carry several rules sharing one +// title. // // It returns the successful observations alongside the first error in UID // order, so a caller that wants to keep the good heartbeats can, and the error @@ -537,9 +537,9 @@ func observeAll(ctx context.Context, src Source, titles map[string]string, uids // parent already wrote — one source of truth, no parent/child drift, and it // exercises ReadLog's header path — and the connection details come from the // inherited environment. Only the run facts the header does not carry travel -// in argv (§4.4). +// in argv. type DaemonChildConfig struct { - URL, Token string // from the inherited environment, never from argv (§20.2) + URL, Token string // from the inherited environment, never from argv Out string Until time.Time Concurrency int @@ -568,7 +568,7 @@ func RunDaemonChild(ctx context.Context, cfg DaemonChildConfig) error { } // Safe to read: the parent closed its writer before spawning this process, - // and no other writer can hold the log's flock (§4.4 step 4). + // and no other writer can hold the log's flock. header, polls, sentinel, err := ReadLog(cfg.Out) if err != nil { return err @@ -576,7 +576,7 @@ func RunDaemonChild(ctx context.Context, cfg DaemonChildConfig) error { if sentinel != nil { return fmt.Errorf("log %s already carries a stopped sentinel: another recorder finished it", cfg.Out) } - // The header's URL is the log's identity (§19.1 step 3). Checking it here + // The header's URL is the log's identity. Checking it here // catches a child that inherited an environment pointing somewhere else, // before it appends a single poll from the wrong Grafana. if header.URL != cfg.URL { @@ -596,7 +596,7 @@ func RunDaemonChild(ctx context.Context, cfg DaemonChildConfig) error { reducer := NewReducer() reducer.seedFrom(polls) - // SIGTERM is how check stops the recorder (§4.4 step 1); SIGINT is the + // SIGTERM is how check stops the recorder; SIGINT is the // same request from a human at a terminal. Both are clean stops, so both // end with a sentinel. Registered before the readiness report, so a signal // arriving the moment the parent unblocks is already handled. @@ -642,10 +642,10 @@ func reportReady(fd int) error { // childSchedule derives what the child polls, and how often, from the header // alone. The cadence comes from PollEverySeconds — the cadence the recording -// actually uses — and is never re-derived from the rule's evaluation interval: -// that is P5's "two authorities", and getting it wrong is fail-open in the -// faster-override direction. Paused rules are excluded here for the same -// reason the parent never polls them (§4.3, §12). +// actually uses — and is never re-derived from the rule's evaluation interval, +// which would be a second authority for the same value and is fail-open in the +// faster-override direction. Paused rules are excluded here for the same reason +// the parent never polls them. // // It returns cadences and nothing else. maxGap, healthGrace and evalStaleAfter // are coverage thresholds applied by the pure layer at classification time, so @@ -672,7 +672,7 @@ func childSchedule(h Header) (titles map[string]string, cadence map[string]time. // watchLoopConfig is the child's working state: what to poll, how often, and // where to append it. There is no threshold in here and no policy — the child -// records and classifies nothing (H5). +// records and classifies nothing. type watchLoopConfig struct { Src Source Writer *Writer @@ -690,7 +690,7 @@ type watchLoopConfig struct { // // The sentinel policy is the load-bearing part. A clean stop (a signal, or // Until) writes it; a hard error does NOT. A recorder that died must look -// exactly like a coverage gap to check, because it is one (§4.5) — writing a +// exactly like a coverage gap to check, because it is one — writing a // sentinel on the way out of a failure would hand check a "recording finished" // claim about a window that stopped being observed. func watchLoop(ctx context.Context, cfg watchLoopConfig) error { @@ -735,8 +735,8 @@ func watchLoop(ctx context.Context, cfg watchLoopConfig) error { pollErr := cfg.pollBatch(ctx, due) if ctx.Err() != nil { // Signalled while a poll was in flight. The aborted poll's error is - // not a recorder failure, and a clean stop wins over it (§4.4 step - // 1: finish the in-flight write, then the sentinel). + // not a recorder failure, and a clean stop wins over it: finish the + // in-flight write, then the sentinel. return cfg.Writer.Stop() } if pollErr != nil { @@ -780,7 +780,7 @@ func untilNextPoll(sched *Scheduler, until, now time.Time) (time.Duration, bool) return max(next.Sub(now), 0), true } -// writePidFile records the child's pid where check looks for it (P9's +// writePidFile records the child's pid where check looks for it (its // --pidfile, default .pid). The format is the decimal pid and a newline, // so `kill $(cat log.jsonl.pid)` works and ReadPidFile stays trivial. func writePidFile(path string, pid int) error { @@ -791,7 +791,7 @@ func writePidFile(path string, pid int) error { } // ReadPidFile is the other side of that contract: the pid of the recorder -// check must stop before it may read the log (§4.4 steps 1-4). +// check must stop before it may read the log. func ReadPidFile(path string) (int, error) { b, err := os.ReadFile(path) if err != nil { diff --git a/grafana-alertcheck/internal/gate/watch_daemon_test.go b/grafana-alertcheck/internal/gate/watch_daemon_test.go index db35c505b..c3a91cd12 100644 --- a/grafana-alertcheck/internal/gate/watch_daemon_test.go +++ b/grafana-alertcheck/internal/gate/watch_daemon_test.go @@ -21,7 +21,7 @@ import ( // os.Executable(), which under `go test` is this binary, so the one integration // test below exercises the real thing — a real fork/exec, a real setsid, a real // inherited environment, a real SIGTERM — with this function standing in for -// the CLI's `watch --daemon-child` dispatch, which lands in P10. +// the CLI's `watch --daemon-child` dispatch. func TestMain(m *testing.M) { if path := os.Getenv(lockHolderEnv); path != "" { os.Exit(runTestLockHolder(path)) @@ -66,8 +66,8 @@ func runTestLockHolder(path string) int { } // runTestDaemonChild parses the child argv childArgs() writes, and reads the -// connection details from the environment — never from argv (§20.2). P10's -// `watch` FlagSet does the same four flags. +// connection details from the environment — never from argv. The CLI's `watch` +// FlagSet does the same four flags. func runTestDaemonChild(args []string) int { cfg := DaemonChildConfig{ URL: os.Getenv("GRAFANA_URL"), @@ -116,7 +116,7 @@ func runTestDaemonChild(args []string) int { } // testBearerToken is what every request to grafanaTestServer must carry. The -// child never receives it in argv (§20.2), so a request that arrives +// child never receives it in argv, so a request that arrives // authenticated is proof that the token reached the detached process through // the inherited environment — and a 401 is what a test sees if that ever // breaks. @@ -147,7 +147,7 @@ func grafanaTestServer(t *testing.T) *httptest.Server { _, _ = w.Write(ruler) case strings.HasPrefix(r.URL.Path, "/api/prometheus/"): if r.URL.Query().Get("rule_name") == "" { - // §2.8: the gate must never read the state endpoint unfiltered. + // The gate must never read the state endpoint unfiltered. http.Error(w, "unfiltered state read", http.StatusBadRequest) return } @@ -212,14 +212,14 @@ func waitFor(t *testing.T, what string, timeout time.Duration, cond func() bool) t.Fatalf("timed out after %s waiting for %s", timeout, what) } -// TestWatchSpawnsADetachedRecorder is P6's one integration test: everything -// from the version gate to the sentinel, through a real detached process. +// The one watch integration test: everything from the version gate to the +// sentinel, through a real detached process. // // It asserts the four things only a real spawn can show — the pidfile points // at a live process, that process is in its own session (setsid, not a bare // `&`), it keeps appending after Watch returned, and SIGTERM makes it finish -// the log in the §4.4 order — and it uses a 200ms --poll-interval to do it in -// about a second, which also exercises the unclamped-override path (§5.1). +// the log in the stop order — and it uses a 200ms --poll-interval to do it in +// about a second, which also exercises the unclamped-override path. func TestWatchSpawnsADetachedRecorder(t *testing.T) { srv := grafanaTestServer(t) t.Setenv("GRAFANA_URL", srv.URL) @@ -263,7 +263,7 @@ func TestWatchSpawnsADetachedRecorder(t *testing.T) { t.Errorf("recorder pgid = %d, want %d: it did not get its own session", pgid, pid) } - // The parent already wrote the first heartbeat before it returned (§4.3); + // The parent already wrote the first heartbeat before it returned; // these later ones prove the detached child is the one appending now. waitFor(t, "the detached recorder to append its own polls", 10*time.Second, func() bool { _, polls, _, err := ReadLog(out) @@ -294,7 +294,7 @@ func TestWatchSpawnsADetachedRecorder(t *testing.T) { t.Fatalf("poll %d = %+v, want a found observation of %s", i, p, watchActiveUID) } if p.GrafanaNow.IsZero() { - t.Fatalf("poll %d has no grafana_now; H4 needs the Date header of its own response", i) + t.Fatalf("poll %d has no grafana_now; every poll needs the Date header of its own response", i) } } if sentinel.Before(header.StartedAt) { diff --git a/grafana-alertcheck/internal/gate/watch_process.go b/grafana-alertcheck/internal/gate/watch_process.go index 629606188..169a4f65a 100644 --- a/grafana-alertcheck/internal/gate/watch_process.go +++ b/grafana-alertcheck/internal/gate/watch_process.go @@ -23,7 +23,7 @@ type detachedChild struct { logOffset int64 } -// spawnChild re-execs this binary as the detached recorder (§4.4). A trailing +// spawnChild re-execs this binary as the detached recorder. A trailing // `&` is NOT sufficient: the child would keep the parent's session and process // group, so it would still take the terminal's signals and, on a runner, die // with the step that started it. Setsid gives it a new session AND a new @@ -52,7 +52,7 @@ func spawnChild(cfg WatchConfig) (detachedChild, error) { logOffset = info.Size() } - // The readiness pipe (§ReadyFDFlag): the child gets the write end as + // The readiness pipe: the child gets the write end as // descriptor 3 and reports on it once it holds the log and is polling. readyRead, readyWrite, err := os.Pipe() if err != nil { @@ -64,9 +64,9 @@ func spawnChild(cfg WatchConfig) (detachedChild, error) { cmd.Stdout = logFile cmd.Stderr = logFile cmd.ExtraFiles = []*os.File{readyWrite} // descriptor 3 in the child - // The environment is how the connection details reach the child (§20.2): - // the token must never appear in argv, where it would land in the process - // table and in CI logs. + // The environment is how the connection details reach the child: the token + // must never appear in argv, where it would land in the process table and + // in CI logs. cmd.Env = os.Environ() cmd.SysProcAttr = &syscall.SysProcAttr{Setsid: true} diff --git a/grafana-alertcheck/internal/gate/watch_test.go b/grafana-alertcheck/internal/gate/watch_test.go index 5a0337a37..33c455af7 100644 --- a/grafana-alertcheck/internal/gate/watch_test.go +++ b/grafana-alertcheck/internal/gate/watch_test.go @@ -92,9 +92,9 @@ func countPolls(polls []Poll, uid string) int { return n } -// TestWatchLoopPollsEachRuleAtItsOwnCadence is §5's per-rule schedule seen -// from the recorder: a 10s rule beside a 300s one keeps its own 5s cadence -// instead of dragging the slack rule along with it or being slowed to its pace. +// The per-rule schedule seen from the recorder: a 10s rule beside a 300s one +// keeps its own 5s cadence instead of dragging the slack rule along with it or +// being slowed to its pace. func TestWatchLoopPollsEachRuleAtItsOwnCadence(t *testing.T) { const tightUID, slackUID = "tight", "slack" path := filepath.Join(t.TempDir(), "log.jsonl") @@ -148,9 +148,8 @@ func TestWatchLoopPollsEachRuleAtItsOwnCadence(t *testing.T) { } } -// TestWatchLoopHardErrorLeavesNoSentinel is §4.5's fail-closed rule from the -// recorder's side: a recorder that dies must look exactly like a coverage gap, -// so it must not sign off the log on its way out. +// Fail-closed from the recorder's side: a recorder that dies must look exactly +// like a coverage gap, so it must not sign off the log on its way out. func TestWatchLoopHardErrorLeavesNoSentinel(t *testing.T) { path := filepath.Join(t.TempDir(), "log.jsonl") clock := newVirtualClock(testNow) @@ -191,10 +190,9 @@ func TestWatchLoopHardErrorLeavesNoSentinel(t *testing.T) { } } -// TestWatchLoopSignalDuringPollIsACleanStop pins §4.4 step 1: SIGTERM arriving -// while a poll is in flight is a clean stop, so the aborted poll's error must -// not suppress the sentinel — otherwise every normal check run, which stops the -// recorder exactly this way, would end unobservable. +// SIGTERM arriving while a poll is in flight is a clean stop, so the aborted +// poll's error must not suppress the sentinel — otherwise every normal check +// run, which stops the recorder exactly this way, would end unobservable. func TestWatchLoopSignalDuringPollIsACleanStop(t *testing.T) { path := filepath.Join(t.TempDir(), "log.jsonl") clock := newVirtualClock(testNow) @@ -311,7 +309,7 @@ func TestWatchLoopPollBatchKeepsTheHeartbeatsItGot(t *testing.T) { } } -// TestReducerSeedFromKeepsMarkersAcrossTheHandoff is H2 at the one seam P6 +// The vanish-versus-clear distinction at the one seam the parent/child handoff // introduces. The parent observes a firing instance; the child starts with a // fresh Reducer and sees the instance gone. Seeded, that is a vanish — a // discontinuity. Unseeded, it is nothing at all, and the instance silently @@ -321,7 +319,7 @@ func TestReducerSeedFromKeepsMarkersAcrossTheHandoff(t *testing.T) { key := instanceKey(firing.Labels) parentPoll := Poll{RuleUID: "r1", Found: true, Abnormal: []Instance{firing}} // The child's first response: the instance is gone from the response - // entirely, which is a vanish and never a clear (§4.7). + // entirely, which is a vanish and never a clear. childObs := observation(testNow, testStateRule("r1", "Example", time.Minute, testNow)) t.Run("seeded", func(t *testing.T) { @@ -385,10 +383,9 @@ func liveObservation(grafanaNow time.Time) Observation { testInstance(StateNormal, "", "a"))) } -// TestPrepareWatchDoesNotWaitForPausedRules is §22.4's regression test: a rule -// paused in its definition is skipped, never waited for. Waiting for one either -// hangs forever or errors before the deploy — and the header must still name -// it, so check can report it as skipped rather than lose it. +// A rule paused in its definition is skipped, never waited for. Waiting for one +// either hangs forever or errors before the deploy — and the header must still +// name it, so check can report it as skipped rather than lose it. func TestPrepareWatchDoesNotWaitForPausedRules(t *testing.T) { var notes strings.Builder cfg := watchTestConfig(t, ¬es, "uid:"+watchActiveUID, "uid:"+watchPausedUID) @@ -423,16 +420,16 @@ func TestPrepareWatchDoesNotWaitForPausedRules(t *testing.T) { } // One poll, for the live rule only — and it is already in the log before - // prepareWatch returned, which is the whole point of §4.3. + // prepareWatch returned, which is the whole point of the record step. if len(polls) != 1 || polls[0].RuleUID != watchActiveUID { t.Fatalf("polls = %+v, want exactly one first observation of %s", polls, watchActiveUID) } if !polls[0].Found || !polls[0].GrafanaNow.Equal(testNow) { t.Errorf("first poll = %+v, want a found observation at %s", polls[0], testNow) } - // §22.3: "the poll record holds the state histogram. Assert that watch - // writes it" — through a real prepareWatch()/Reducer call, not just - // log_test.go's hand-built Writer/ReadLog round trip. + // The poll record holds the state histogram, asserted through a real + // prepareWatch()/Reducer call rather than log_test.go's hand-built + // Writer/ReadLog round trip. if want := map[string]int{"normal": 1}; !maps.Equal(polls[0].Histogram, want) { t.Errorf("Histogram = %v, want %v: watch must record the state histogram on every poll it writes", polls[0].Histogram, want) } @@ -441,9 +438,9 @@ func TestPrepareWatchDoesNotWaitForPausedRules(t *testing.T) { } } -// TestPrepareWatchHeaderRecordsTheOverriddenCadence is P5's "two authorities" -// from the writing side: whatever --poll-interval resolves to is what the -// header records, because that is the only value check may derive maxGap from. +// One authority for the cadence, from the writing side: whatever +// --poll-interval resolves to is what the header records, because that is the +// only value check may derive maxGap from. func TestPrepareWatchHeaderRecordsTheOverriddenCadence(t *testing.T) { var notes strings.Builder cfg := watchTestConfig(t, ¬es, "uid:"+watchActiveUID) @@ -467,8 +464,8 @@ func TestPrepareWatchHeaderRecordsTheOverriddenCadence(t *testing.T) { } } -// TestPrepareWatchFailsWhenTheScheduleDoesNotFit: the budget check runs on the -// latencies the parent just measured, before the deploy runs (§5.2). +// The budget check runs on the latencies the parent just measured, before the +// deploy runs. func TestPrepareWatchFailsWhenTheScheduleDoesNotFit(t *testing.T) { var notes strings.Builder cfg := watchTestConfig(t, ¬es, "uid:"+watchActiveUID) @@ -484,9 +481,9 @@ func TestPrepareWatchFailsWhenTheScheduleDoesNotFit(t *testing.T) { assertBudgetMessage(t, err.Error()) } -// TestPrepareWatchVerifiesNormalInstancesAreVisible is the §3.2 check at the -// one place it can still be cheap: the first observation. If the state endpoint -// stops returning normal instances, the reduction's predicate quietly inverts. +// Normal instances are verified visible at the one place it is still cheap: +// the first observation. If the state endpoint stops returning them, the +// reduction's predicate quietly inverts. func TestPrepareWatchVerifiesNormalInstancesAreVisible(t *testing.T) { var notes strings.Builder cfg := watchTestConfig(t, ¬es, "uid:"+watchActiveUID) @@ -499,8 +496,8 @@ func TestPrepareWatchVerifiesNormalInstancesAreVisible(t *testing.T) { if err == nil { t.Fatal("prepareWatch: no error when totals claim normal instances the response omitted") } - if !strings.Contains(err.Error(), "3.2") { - t.Errorf("error does not name §3.2: %v", err) + if !strings.Contains(err.Error(), "no longer returns normal instances") { + t.Errorf("error does not say the endpoint stopped returning normal instances: %v", err) } // The failure happens before any poll is appended, so the log holds a @@ -525,9 +522,9 @@ func TestPrepareWatchRejectsAnUnsupportedGrafana(t *testing.T) { } } -// TestPrepareWatchNotesAnAbsentRule: a rule that resolved in the ruler API but -// is absent from the state endpoint is recorded as Found=false — authoritative -// evidence P7 turns into unobservable — not silently dropped. +// A rule that resolved in the ruler API but is absent from the state endpoint +// is recorded as Found=false — authoritative evidence the coverage proof turns +// into unobservable — not silently dropped. func TestPrepareWatchNotesAnAbsentRule(t *testing.T) { var notes strings.Builder cfg := watchTestConfig(t, ¬es, "uid:"+watchActiveUID) @@ -601,8 +598,8 @@ func TestWatchConfigValidation(t *testing.T) { }) } -// TestChildScheduleUsesTheRecordedCadence is P5's fail-open direction, checked -// on the child's side: a log recorded at 5s on a 300s rule must schedule at 5s. +// The fail-open direction, checked on the child's side: a log recorded at 5s on +// a 300s rule must schedule at 5s. // Re-deriving from the interval would give 150s — and every real 250s hole in // that recording would pass. func TestChildScheduleUsesTheRecordedCadence(t *testing.T) { @@ -616,7 +613,7 @@ func TestChildScheduleUsesTheRecordedCadence(t *testing.T) { t.Fatalf("childSchedule: %v", err) } if _, ok := titles["paused"]; ok { - t.Error("the child scheduled a rule that was paused when the window opened (§4.3)") + t.Error("the child scheduled a rule that was paused when the window opened") } if got := cadence["fast"]; got != 5*time.Second { t.Errorf("pollEvery = %s, want 5s from the header, not %s from the interval", got, defaultPollEvery(300)) From 80d92ee7131b17fe089ddaaf49e3e1b05b6dec48 Mon Sep 17 00:00:00 2001 From: Bartek Tofel Date: Wed, 2 Sep 2026 13:29:34 +0200 Subject: [PATCH 32/43] fix: merge conflict --- grafana-alertcheck/internal/gate/classify.go | 17 ++--------------- 1 file changed, 2 insertions(+), 15 deletions(-) diff --git a/grafana-alertcheck/internal/gate/classify.go b/grafana-alertcheck/internal/gate/classify.go index 5d28c16ce..ca4a4222c 100644 --- a/grafana-alertcheck/internal/gate/classify.go +++ b/grafana-alertcheck/internal/gate/classify.go @@ -613,18 +613,6 @@ func decide(h Header, polls []Poll, sentinel *time.Time, defs []Definition, }) } -<<<<<<< HEAD - // MinObserved (§12): default len(defs) after the collapse (already done - // by Resolve before decide ever sees defs). skipped rules count against - // it unless AllowPaused says otherwise. A shortfall counts toward exit 1 - // (§9.1), never exit 2 — decide never returns an error for this — and H7 - // requires it to surface through Violations like any other fail reason, - // so a shortfall always produces at least one, even when no rule is - // paused at all (an operator-supplied MinObserved that simply exceeds - // what could ever be resolved). - counted := watchedCount - var attributable []Definition -======= // MinObserved defaults to len(defs) after duplicate names collapse // (already done by Resolve before decide ever sees defs). Skipped rules // count against it unless AllowPaused says otherwise. A shortfall counts @@ -633,9 +621,8 @@ func decide(h Header, polls []Poll, sentinel *time.Time, defs []Definition, // a shortfall always produces at least one, even when no rule is paused at // all (an operator-supplied MinObserved that simply exceeds what could ever // be resolved). - counted := observedCount - var chargeable []Definition ->>>>>>> 641701bb (chore: more concise comments) + counted := watchedCount + var attributable []Definition if pol.AllowPaused { counted += len(skippedRules) } else { From 34040dfc241a8bd2fe2adc2b815e46ad79e94eca Mon Sep 17 00:00:00 2001 From: Bartek Tofel Date: Fri, 4 Sep 2026 17:01:53 +0200 Subject: [PATCH 33/43] chore: shorten comments --- grafana-alertcheck/internal/gate/check.go | 211 ++++++------------- grafana-alertcheck/internal/gate/classify.go | 155 +++++--------- grafana-alertcheck/internal/gate/coverage.go | 149 ++++--------- grafana-alertcheck/internal/gate/log.go | 105 +++------ grafana-alertcheck/internal/gate/schedule.go | 175 ++++++--------- grafana-alertcheck/internal/gate/source.go | 57 ++--- grafana-alertcheck/internal/gate/watch.go | 122 ++++------- 7 files changed, 320 insertions(+), 654 deletions(-) diff --git a/grafana-alertcheck/internal/gate/check.go b/grafana-alertcheck/internal/gate/check.go index 0f21b4d86..eab5d931b 100644 --- a/grafana-alertcheck/internal/gate/check.go +++ b/grafana-alertcheck/internal/gate/check.go @@ -14,26 +14,21 @@ import ( // Check returns (Result, error) and no exit code: the code is a presentation // decision the CLI makes. err != nil is exit 2 unconditionally, even alongside // real violations; violations with err == nil is exit 1; neither is exit 0. -// -// Check never reads the environment either. The URL and token are read by the -// CLI and passed in as fields, and the token must never reach a *flag.FlagSet. +// Check never reads the environment — the CLI reads URL/token and passes them +// in, and the token must never reach a *flag.FlagSet. // countdownEvery is how often the collection loop reports what it is waiting -// for. A silent wait is indistinguishable from a hung process, and the wait -// after `to` is the longest silence in the whole run. +// for; a silent wait is indistinguishable from a hung process. const countdownEvery = 30 * time.Second -// recorderStopTimeout bounds the wait for the recorder's exit. Everything the -// recorder does after SIGTERM is local (finish the in-flight write, append the -// sentinel, fsync) and an in-flight poll aborts through the child's own -// context, so the real figure is milliseconds; this is loose enough for an -// overloaded runner. The timeout is a hard error rather than a longer wait — a -// log a writer may still hold cannot be read at all. +// recorderStopTimeout bounds the wait for the recorder's exit after SIGTERM. +// Everything after the signal is local (finish the in-flight write, sentinel, +// fsync), so this is loose; it stays a hard error because a log a writer still +// holds cannot be read. const recorderStopTimeout = 30 * time.Second -// recorderStopPoll is how often that wait re-checks the pid. There is no -// wait(2) available: the recorder is a detached session leader, not this -// process's child, so its exit can only be observed by polling. +// recorderStopPoll is how often the wait re-checks the lock. With no wait(2) +// on a detached session leader, its exit is observable only by polling. const recorderStopPoll = 100 * time.Millisecond // Config is check's whole input. It is the CLI's view of a run, and it is @@ -117,14 +112,10 @@ func (cfg Config) namedAlerts() []string { } // Check is the I/O shell: HTTP, signals, the pidfile, file reads, the -// countdown print. Every correctness question it touches is answered -// elsewhere — by proveCoverage and decide, which are pure — and that split is -// the most important seam in the project. Check therefore needs two -// integration tests; decide carries the suite. -// -// A pass is exactly len(Violations) == 0 && err == nil. Every error path below -// leaves err non-nil, and no path anywhere in this file converts an error into -// an empty Result with a nil error. +// countdown print. Every correctness question it touches is answered elsewhere +// (proveCoverage, decide — both pure), which is the most important seam in the +// project. A pass is exactly len(Violations) == 0 && err == nil; every error +// path leaves err non-nil. func Check(ctx context.Context, cfg Config) (Result, error) { cfg = cfg.withDefaults() if err := cfg.validate(); err != nil { @@ -160,10 +151,8 @@ func (cfg Config) validate() error { } now := cfg.Clock.Now() - // from mirrors what check() will use, so the two window checks below judge - // the window that will really be classified. The fallback is not written - // back into cfg: check() re-reads the clock at the same point, and one - // authority for that value is better than two that could disagree. + // from mirrors what check() will use, so the window checks below judge the + // window that will really be classified. from := cfg.From switch { case from.IsZero() && cfg.Log != "": @@ -185,13 +174,10 @@ func (cfg Config) validate() error { from.Format(time.RFC3339), fromFutureTolerance, now.Format(time.RFC3339)) } - // A `to` already in the past is not a special mode WITH a log: the - // collection loop's condition is simply already true and the evidence is - // classified immediately. Without one it is a different thing entirely — a - // request to prove a window that nothing observed. Refusing it is not - // pedantry: the coverage window would end before the first observation, - // every heartbeat gap inside it would measure negative, and the run would - // report a proved window it never saw. + // A `to` in the past is fine WITH a log (the collection loop is already + // done). Without one it is a request to prove a window nothing observed: + // every heartbeat gap would measure negative, and the run would report a + // proved window it never saw. if cfg.Log == "" && !cfg.To.After(now) { return fmt.Errorf("check: `to` %s has already passed and there is no recorded log: a window that ended before check started can only be classified from a recording", cfg.To.Format(time.RFC3339)) @@ -232,11 +218,9 @@ func check(ctx context.Context, cfg Config, src Source) (Result, error) { // ---- With a log, validate its identity. ------------------------------- // The header is read early — line 1 only, the one line a writer can never - // change (ReadLogHeader) — so a wrong URL or a rule that no longer - // resolves fails closed NOW rather than after the whole window has - // elapsed. It is advisory: the authoritative header comes from the single - // full ReadLog once collection is over and the writer has exited, and the - // identity is validated again against that one. + // change — so a wrong URL or an unresolvable rule fails closed NOW. It is + // advisory: the authoritative header is re-read once collection ends and + // the writer has exited. var ( resolved []Definition notes []string @@ -291,11 +275,8 @@ func check(ctx context.Context, cfg Config, src Source) (Result, error) { fmt.Fprintf(cfg.Notes, "warning: %s\n", warning) } - // The measurement pass and the budget check belong to single-step mode - // alone: in recorder mode watch already took one observation of every rule - // and checked the budget against those measured latencies before it - // detached, and repeating it here would spend a second poll of every rule - // to re-answer a question already answered. + // The measurement pass and budget check are single-step only: in recorder + // mode watch already measured and checked the budget before detaching. var ( header Header initial []Poll @@ -364,10 +345,8 @@ func check(ctx context.Context, cfg Config, src Source) (Result, error) { } collected, err := collectUntil(ctx, cfg, windowEnd, poller) if err != nil { - // The failure limit was exceeded (retryTransport already gave every - // transient failure its backoff), or the context ended. Nothing - // collected is classified — the count is there so an operator can tell - // a run that failed at once from one that failed at minute nine. + // Nothing collected is classified; the count lets an operator tell a + // run that failed at once from one that failed at minute nine. return Result{}, fmt.Errorf("collect evidence after %d poll(s): %w", len(collected), err) } @@ -385,29 +364,20 @@ func check(ctx context.Context, cfg Config, src Source) (Result, error) { return Result{}, err } header, polls, sentinel, err = ReadLog(cfg.Log) - // The lock stays held across the read, so no writer can appear between - // the proof that there was none and the read itself. Released here - // rather than deferred: everything past this point works from bytes - // already in memory, and the drain wait below can take minutes. + // Held across the read so no writer can appear mid-read, then released + // (everything past here works from memory, and the drain wait is minutes). _ = heldLog.Close() if err != nil { return Result{}, err } - // The authoritative header, validated the same way the advisory one - // was — and its result is KEPT. Everything from here on judges the - // header ReadLog returned, so nothing downstream rests on the advisory - // read having been right. That read is what it claims to be: a - // fail-fast, and no part of the verdict depends on it. + // The authoritative header wins: the advisory read was only a fail-fast. resolved, _, err = resolveFromLog(allDefs, header, cfg) if err != nil { return Result{}, err } - // rt is re-derived because it depends on the header: PollEverySeconds - // is the one load-bearing value the advisory read supplied. windowEnd - // is deliberately NOT recomputed from the gt this returns: the - // collection loop has already stopped at the earlier value, and moving - // the end of the window afterwards would prove a window this run did - // not collect. + // rt is re-derived from the authoritative header. windowEnd is NOT + // recomputed: the loop already stopped at the earlier value, and moving + // it afterwards would prove a window this run did not collect. if rt, gt, err = DeriveTimingsFromLog(header, resolved); err != nil { return Result{}, fmt.Errorf("log identity: %w", err) } @@ -453,24 +423,14 @@ func check(ctx context.Context, cfg Config, src Source) (Result, error) { return result, errors.Join(decideErr, drainErr) } -// resolveFromLog turns a log header into the resolved definitions, and is the -// log's identity check in practice. Three things are verified: the URL -// matches, the schema version matches (ReadLog/ReadLogHeader own that), and -// every header UID still resolves against the fresh ruler read. The alert set -// is TAKEN from the log, never compared — with Alerts required empty in log -// mode there is nothing to compare it against, and refusing a log recorded -// against a different alert set is exactly this URL-and-UID failure. +// resolveFromLog is the log's identity check in practice: the URL must match +// and every header UID must still resolve against a fresh ruler read. The +// alert set is TAKEN from the log, never compared against --alerts (which the +// validator requires empty in log mode). Resolving through Resolve by uid: +// keeps one implementation of the resolution rules. // -// Resolving through Resolve, by uid:, rather than by a private lookup, keeps -// one implementation of the resolution rules: a header naming a recording or -// datasource-managed rule gets the same specific refusal an operator would, -// and a header naming the same UID twice collapses with a note -// (DeriveTimingsFromLog rejects that case outright, so the note is belt and -// braces). -// -// Only the header-to-defs direction needs checking. The opposite direction -// cannot fail here: resolved is BUILT from the header, so no resolved -// definition can be absent from it. +// Only the header-to-defs direction can fail: resolved is BUILT from the +// header, so no resolved definition can be absent from it. func resolveFromLog(allDefs []Definition, h Header, cfg Config) ([]Definition, []string, error) { if h.URL != cfg.URL { return nil, nil, fmt.Errorf("log identity: %s recorded url %q but this run is configured for %q", @@ -619,32 +579,22 @@ func collectUntil(ctx context.Context, cfg Config, deadline time.Time, p *livePo } } -// stopRecorder signals the recorder and waits for it to go. Nothing here is -// best-effort: the log may not be read until the writer has provably gone, so -// every failure to reach that state is a hard error. -// -// It returns the log held under an exclusive flock. The caller must keep that -// file open across ReadLog and close it afterwards — the lock is the proof -// that no writer exists, and holding it across the read also shuts out a new -// one appearing between the proof and the read. +// stopRecorder signals the recorder and waits for it to go; the log may not be +// read until the writer has provably gone, so every failure is a hard error. +// It returns the log held under an exclusive flock, which the caller must keep +// open across ReadLog — the lock is the proof that no writer exists. // -// Two authorities, and only one of them is evidence: +// Two authorities, only one of which is evidence: // -// - The PIDFILE says whether a recording was ever started, and an absent or -// unparseable one must never read as "there was nothing to stop". The -// parent writes the pidfile only AFTER the child reports that it holds the -// log and is polling, and removes it on every failing path, so a missing -// one means watch failed and this run has no evidence at all. -// - The FLOCK says whether a writer exists RIGHT NOW. Nothing removes the -// pidfile when a recorder exits cleanly — the parent has long returned and -// the child never learns the path — so after a --until run, a supported -// flow, the pidfile names a pid nobody owns. Signalling it would SIGTERM -// whatever same-user process inherited that pid. The kernel releases a -// flock when its holder exits, crash included, so the lock cannot go -// stale that way. +// - the PIDFILE says whether a recording ever started (it is written only +// after the child reports ready, and removed on failure). +// - the FLOCK says whether a writer exists right now. A pidfile can go stale +// — nothing removes it on a clean --until stop, so it may name a pid +// somebody else now owns — but the kernel drops a flock when the holder +// exits, so the lock is always authoritative. // -// So: read the pidfile to learn that a recording happened, then ask the lock -// whether it is still running, and signal only if it is. +// So: read the pidfile to learn a recording happened, ask the lock whether it +// is still running, and signal only if it is. func stopRecorder(ctx context.Context, cfg Config) (*os.File, error) { pid, err := ReadPidFile(cfg.PidFile) if err != nil { @@ -724,27 +674,15 @@ type drainVerdict struct { note string } -// drainWait is the final instance of the liveness check, asking each rule the -// last question — did you evaluate through the end of the window? A rule that -// cannot answer within drainTimeout is unobservable, never a pass. +// drainWait is the final liveness check: did each rule evaluate through the +// end of the window? A rule that cannot answer within drainTimeout is +// unobservable, never a pass. It returns one verdict per rule it could not +// clear (keyed by UID); an error only for a hard failure of the wait itself. // -// It returns one verdict per rule it could not clear, keyed by UID, which the -// caller folds into the Result. It returns an error only for a hard failure of -// the wait itself; a rule that simply never catches up is reported, not -// raised. -// -// Two kinds of rule are excluded before the wait starts, both because draining -// them could not change a verdict: -// -// - a rule the HEADER says was already paused when the recording opened: it -// is skipped, it was not evaluating, and it never was — there is no -// evaluation to wait for. The header and not the definition, for decide's -// reason (Header.pausedAtStart): a rule the header says was active must be -// drained or faulted, because a pause somebody applied after the window is -// not evidence about the window; -// - a rule whose last poll says Found == false: the rule-absent coverage -// check already makes it unobservable, so the only thing draining it could -// add is drainTimeout of waiting before the same answer. +// Two kinds of rule are excluded up front because draining them could not +// change a verdict: a rule the HEADER says was paused at the window open (the +// header, not the late-resolved definitions — see Header.pausedAtStart), and a +// rule whose last poll says Found == false (already unobservable via rule_absent). func drainWait(ctx context.Context, cfg Config, src Source, defs []Definition, pausedAtStart map[string]bool, rt map[string]ruleTimings, polls []Poll, windowEnd time.Time, timeout time.Duration) (map[string]drainVerdict, error) { @@ -872,16 +810,11 @@ func anyPollEvaluatedThrough(polls []Poll, windowEnd time.Time) bool { return false } -// evaluatedThrough is the drain wait's one comparison, and it is cross-domain: -// lastEvaluation is a Grafana timestamp and windowEnd is runner-domain, so the -// Grafana value is translated by its own poll's skew. The skew BOUND is then -// subtracted rather than added — the pessimistic end of the uncertainty — so -// an evaluation that only might have reached the end of the window does not -// count as one that did. Understating it costs a few more seconds of waiting; -// overstating it would pass an unproven window. -// -// A zero lastEvaluation never satisfies the wait: only a paused rule may -// legitimately report it, and a paused rule has nothing to drain. +// evaluatedThrough is the drain wait's cross-domain comparison: a Grafana +// lastEvaluation is translated by its poll's skew, and the bound is SUBTRACTED +// (the pessimistic end) so an evaluation that only *might* have reached the +// window end is not counted as having reached it. A zero lastEvaluation never +// satisfies the wait. func evaluatedThrough(lastEval time.Time, skew, bound time.Duration, windowEnd time.Time) bool { if lastEval.IsZero() { return false @@ -889,16 +822,10 @@ func evaluatedThrough(lastEval time.Time, skew, bound time.Duration, windowEnd t return !lastEval.Add(-skew).Add(-bound).Before(windowEnd) } -// mergeDrainTimeouts folds the I/O drain wait's verdicts into the pure layer's -// Result. It runs immediately after decide rather than before it, because -// decide owns proveCoverage and therefore builds the Coverage map itself; that -// keeps decide a pure function of its arguments. -// -// It returns its own error rather than mutating decide's, so neither hides the -// other: a run with one rule unobservable from the coverage proof and another -// from the drain wait must name both. The error says "at the drain wait" for -// that reason — the two are joined into one message, and two counts under one -// identical phrase read as a contradiction rather than as two findings. +// mergeDrainTimeouts folds the drain wait's I/O verdicts into the pure Result, +// running after decide so that function stays pure of its arguments. It returns +// its own error rather than mutating decide's so neither hides the other: a run +// faulted by both the coverage proof and the drain wait must name both. func mergeDrainTimeouts(res Result, drained map[string]drainVerdict) (Result, error) { if len(drained) == 0 { return res, nil diff --git a/grafana-alertcheck/internal/gate/classify.go b/grafana-alertcheck/internal/gate/classify.go index ca4a4222c..9d881334a 100644 --- a/grafana-alertcheck/internal/gate/classify.go +++ b/grafana-alertcheck/internal/gate/classify.go @@ -7,19 +7,15 @@ import ( "time" ) -// ReasonNodata is decide's own unobservable reason: proveCoverage deliberately -// never sets it — health=nodata is a note there, never fatal, because -// escalating it needs Policy.NodataIsUnobservable, and the pure coverage layer -// has no Policy to consult (coverage.go, check 5). decide is the seam that DOES -// have a Policy, so the escalation lives here. +// ReasonNodata is decide's own unobservable reason: proveCoverage never sets it +// — health=nodata is a note there, never fatal — because escalating it needs +// Policy.NodataIsUnobservable, which only decide (the Policy-holding seam) has. const ReasonNodata UnobservableReason = "nodata" -// Outcome is the verdict of one instance's timeline, and — after decide takes -// the worst across a rule's instances — of the rule itself. It is a published -// JSON output: the three fail values stay distinct even though v1 maps all -// three to exit 1, because a later reason string cannot recover the -// information a single "fail" value would have thrown away, and because -// splitting them later would break a published interface for no gain. +// Outcome is the verdict of one instance's timeline, and (after decide takes +// the worst across instances) of the rule. It is a published JSON output: the +// fail values stay distinct even though v1 maps them all to exit 1, so a later +// version can split them without breaking the interface. type Outcome string const ( @@ -62,28 +58,20 @@ type Violation struct { State State Health string // raw, reporting-only, like Poll.Health LastError string - // FirstSeen is the episode's onset, in the runner domain: activeAt - // translated by its poll's own skew when the episode opened strictly - // inside the window, or `from` itself when the instance was already bad - // at window-open (preexisting) — never a raw, untranslated Grafana - // timestamp. + // FirstSeen is the episode's onset in the runner domain (translated by the + // poll's own skew), or `from` when preexisting — never a raw Grafana time. FirstSeen time.Time - // ClearedAt is zero unless the episode closed via a genuine Cleared - // event, also translated to the runner domain. + // ClearedAt is zero unless the episode closed via a genuine Cleared event. ClearedAt time.Time InstanceLabels map[string]string - // Note carries an explanation for a Violation that has no instance - // behind it — the synthetic MinObserved shortfall entry decide emits when - // the deficit exceeds what any named paused rule explains. LastError is - // reporting-only rule state from a real poll and must not double as a - // message field for a Violation that never touched one. + // Note explains a Violation with no instance behind it — decide's synthetic + // MinObserved shortfall — and must not double as LastError (reporting-only + // rule state from a real poll). Note string } -// RuleVerdict is one rule's worst-of outcome, always present for every -// resolved rule — Verdicts includes the passes, not only the failures — so a -// human reading the table sees every alert that was asked for, not only the -// ones that misbehaved. +// RuleVerdict is one rule's worst-of outcome, present for every resolved rule +// (passes included) so the table shows every alert asked for. type RuleVerdict struct { Alert, RuleUID string Outcome Outcome @@ -176,27 +164,17 @@ type instanceTimeline struct { episodes []episode } -// runnerTime translates a Grafana-domain timestamp recorded on poll p into the -// runner domain, undoing that poll's own measured skew. GrafanaNow and -// ActiveAt come from the same response, so the same poll's skew applies to -// both. This is the single implementation of that translation for the package -// (same drift argument as pollsForRule): coverage.go's window membership test -// and heartbeat boundary segments call it too, rather than each keeping its own -// copy of `p.GrafanaNow.Add(-p.Skew())` that could silently diverge from this -// one. +// runnerTime translates a Grafana-domain timestamp into the runner domain by +// undoing poll p's measured skew. The single implementation for the package — +// coverage.go's window-membership and heartbeat-boundary checks use it too. func runnerTime(p Poll, grafanaDomain time.Time) time.Time { return grafanaDomain.Add(-p.Skew()) } // classifyRule builds every instance timeline for one rule across -// [from, windowEnd] and reduces them to the rule's worst outcome, its merged -// BadFor, and the Violations the preexisting policy actually charges against -// the run. It is PURE: no I/O, no clock reads — decide supplies windowEnd -// (to + transitionGrace) rather than this function deriving it, so a test can -// pin the boundary directly. -// -// polls need not be pre-filtered to this rule, matching proveCoverage's own -// contract: selection is by def.UID. +// [from, windowEnd] and reduces them to the rule's worst outcome, merged +// BadFor, and the Violations the preexisting policy charges against the run. +// PURE: no I/O, no clock reads; polls need not be pre-filtered to this rule. func classifyRule(def Definition, polls []Poll, from, windowEnd time.Time, badStates map[State]bool, pol PreexistingPolicy) (Outcome, time.Duration, []Violation) { rulePolls := pollsForRule(polls, def.UID) inWindow := inWindowPolls(rulePolls, from, windowEnd) @@ -204,11 +182,9 @@ func classifyRule(def Definition, polls []Poll, from, windowEnd time.Time, badSt timelines := make(map[string]*instanceTimeline) order := make([]string, 0) - // get backfills labels the first time a real Instance is seen: a key can - // be created earlier by a bare Cleared/Vanished marker, which carries - // no labels of its own, and the instance later re-firing must not report - // an empty InstanceLabels just because of which event happened to create - // the timeline first. + // get backfills labels on the first real Instance: a bare Cleared/Vanished + // marker can create the timeline first (with no labels), and a later re-fire + // must not report an empty InstanceLabels. get := func(key string, labels map[string]string) *instanceTimeline { tl, ok := timelines[key] if !ok { @@ -228,30 +204,26 @@ func classifyRule(def Definition, polls []Poll, from, windowEnd time.Time, badSt tl.episodeStart = start } closeEpisode := func(tl *instanceTimeline, end time.Time, real bool) { - // inWindowPolls admits a poll whose translated time is up to its own - // skew bound PAST windowEnd (the membership test widens the boundary - // outward). Without this clamp a genuine Cleared event on such a - // poll would produce an episode.end slightly beyond windowEnd, - // contradicting the episode type's own "clamped to - // [from, windowEnd]" contract. + // inWindowPolls widens its boundary outward by the skew bound, so a + // translated end can land past windowEnd or before episodeStart; clamp + // both, otherwise mergeDurations gets an inverted span. if end.After(windowEnd) { end = windowEnd } - // Different polls can carry different measured skews. In theory a - // closing poll's translated time could land before the opening - // poll's — skew is capped at SkewHardLimit (60s), so this is remote, - // not impossible — and a negative span would feed mergeDurations a - // duration that subtracts instead of adds. Clamp rather than trust - // the arithmetic never to invert. if end.Before(tl.episodeStart) { end = tl.episodeStart } tl.episodes = append(tl.episodes, episode{start: tl.episodeStart, end: end, closedByRealClear: real}) tl.badOpen = false } +<<<<<<< HEAD // onsetOf resolves a fresh episode's start: the instance's own ActiveAt, // translated to the runner domain by this poll's skew, clamped to // [from, windowEnd]. +======= + // onsetOf is a fresh episode's start: the translate ActiveAt, clamped to + // never read as starting before the window opened. +>>>>>>> 056b9146 (chore: shorten comments) onsetOf := func(p Poll, inst Instance) time.Time { start := runnerTime(p, inst.ActiveAt) if start.Before(from) { @@ -276,12 +248,9 @@ func classifyRule(def Definition, polls []Poll, from, windowEnd time.Time, badSt case !tl.seen: tl.seen = true if bad { - // Fail-closed: only call an onset "preexisting" when - // even the worst-case skew error still puts it at or - // before `from`. An onset that might really have landed - // just inside the window must classify as a new episode, - // never earn the `recovered` benefit of the doubt it - // would get if it later clears. + // Fail-closed: "preexisting" only when even the worst-case + // skew error places the onset at or before `from`; an onset + // that might be in-window must classify as a new episode. activeAtRunner := runnerTime(p, inst.ActiveAt) tl.preexisting = !activeAtRunner.Add(p.SkewBound()).After(from) if tl.preexisting { @@ -301,11 +270,8 @@ func classifyRule(def Definition, polls []Poll, from, windowEnd time.Time, badSt for _, key := range p.Cleared { tl := get(key, nil) if !tl.seen { - // Cleared on the very first mention means the transition - // happened between the poll just before this one (possibly - // pre-window) and this one: there is no window-internal - // evidence that it was ever bad, so it is neither - // preexisting nor a new episode. + // Cleared on first mention: the transition happened pre-window, + // with no in-window evidence it was ever bad. tl.seen = true continue } @@ -315,9 +281,8 @@ func classifyRule(def Definition, polls []Poll, from, windowEnd time.Time, badSt tl.lastHealth, tl.lastError = p.Health, p.LastError } - // Vanished is a deliberate no-op: freeze whatever badOpen/preexisting - // already holds. An instance that vanishes while bad must stay bad, and - // one that vanishes while never having been bad must stay uninteresting. + // Vanished is a deliberate no-op: freeze badOpen/preexisting as-is, so a + // vanish while bad stays bad (never reading as a recovery). for _, key := range p.Vanished { tl := get(key, nil) tl.seen = true @@ -325,10 +290,8 @@ func classifyRule(def Definition, polls []Poll, from, windowEnd time.Time, badSt } } - // Multiple instances can appear for the first time within the same poll, - // and map iteration order is nondeterministic; sort so this pure - // function's Violations/BadFor output is stable across runs given the - // same input, like log.go sorts Cleared/Vanished for the same reason. + // Map iteration order is nondeterministic; sort so Violations/BadFor output + // is stable for a given input (like log.go sorts Cleared/Vanished). slices.Sort(order) var ( @@ -357,9 +320,8 @@ func classifyRule(def Definition, polls []Poll, from, windowEnd time.Time, badSt instOutcome = OutcomePersistentlyBad } default: - // A genuinely new onset always fails, whether or not it later - // clears within the window: only a PREEXISTING condition earns - // the benefit of `recovered`. + // A genuinely new onset fails whether or not it clears in-window; + // only a preexisting condition earns `recovered`. instOutcome = OutcomeNewlyBad } @@ -529,12 +491,9 @@ func decide(h Header, polls []Poll, sentinel *time.Time, defs []Definition, if s < 0 { s = -s } - // The bound travels with ITS OWN poll's skew, never the largest bound - // seen overall (Result.ClockSkewBound's doc comment) — so it is only - // ever overwritten in lockstep with ClockSkew, on the same poll. >= - // rather than > on top of skewSeen: a strict > would never assign the - // bound at all when every poll's skew is exactly 0, understating the - // real measurement uncertainty as an unearned "bound ±0s". + // The bound travels with its own poll's skew (see Result.ClockSkewBound), + // overwritten in lockstep. >= rather than > so a bound is still assigned + // when every poll's skew is exactly 0. if !skewSeen || s > result.ClockSkew { result.ClockSkew = s result.ClockSkewBound = p.SkewBound() @@ -556,13 +515,10 @@ func decide(h Header, polls []Poll, sentinel *time.Time, defs []Definition, unobservableNames []string ) - // `skipped` is decided from the header, never from defs. defs are - // resolved after the window has closed, so Definition.IsPaused describes - // the present; Header.pausedAtStart describes the moment the recording - // opened, which is the only moment "paused before the window opened" can - // mean. Reading the late definition instead let a rule that fired and was - // then paused report as skipped, with its firing never classified — and - // under AllowPaused that was a pass. + // `skipped` is decided from the header, never from defs: defs are resolved + // after the window closed, so Definition.IsPaused describes the present, + // while Header.pausedAtStart describes the window open — the only moment + // "paused before the window opened" can mean. pausedAtStart := h.pausedAtStart() for _, def := range defs { @@ -613,14 +569,9 @@ func decide(h Header, polls []Poll, sentinel *time.Time, defs []Definition, }) } - // MinObserved defaults to len(defs) after duplicate names collapse - // (already done by Resolve before decide ever sees defs). Skipped rules - // count against it unless AllowPaused says otherwise. A shortfall counts - // toward exit 1, never exit 2 — decide never returns an error for this — - // and it has to surface through Violations like any other fail reason, so - // a shortfall always produces at least one, even when no rule is paused at - // all (an operator-supplied MinObserved that simply exceeds what could ever - // be resolved). + // MinObserved defaults to len(defs) (post-collapse). A shortfall counts + // toward exit 1, never exit 2, and surfaces through Violations — so it + // always produces at least one, even when no rule is paused. counted := watchedCount var attributable []Definition if pol.AllowPaused { diff --git a/grafana-alertcheck/internal/gate/coverage.go b/grafana-alertcheck/internal/gate/coverage.go index 5c82458ee..527720bfe 100644 --- a/grafana-alertcheck/internal/gate/coverage.go +++ b/grafana-alertcheck/internal/gate/coverage.go @@ -8,25 +8,15 @@ import ( // keepLastReason is the instance Reason that check 9 watches for. const keepLastReason = "KeepLast" -// Two things this file deliberately leaves to its callers: -// -// - "from more than fromFutureTolerance ahead of the runner's clock" is a -// hard error, but it is once-per-run input validation rather than a -// per-rule coverage check, and this function has no error return. -// Config.validate (check.go) applies it; check 2 below owns only the -// "from < StartedAt" half. -// - A rule paused before the window opened is never scheduled or polled, so -// it would reach this function with zero polls and read as one large -// heartbeat_gap rather than as skipped (pinned by -// TestProveCoverage_SkippedRuleWithZeroPollsPinnedAsHeartbeatGap). decide -// returns before it ever calls proveCoverage for such a rule, reading -// skipped from the log header (Header.pausedAtStart) and NOT from -// Definition.IsPaused — the definitions are re-resolved after the window -// closed, so they cannot answer what was paused when it opened. +// Two things this file leaves to its callers: the "from too far ahead" bound is +// Config.validate's once-per-run input validation (check 2 owns only the +// "from < StartedAt" half), and a rule paused at the window open never reaches +// proveCoverage — decide reads `skipped` from Header.pausedAtStart first, so a +// paused rule's zero polls read as skipped, not as one large heartbeat gap. // UnobservableReason names why proveCoverage could not prove a rule's window. -// It is machine-readable — this reaches the JSON output, so it is a published -// vocabulary like Outcome; prose belongs in Notes. +// It reaches the JSON output, so it is a published vocabulary like Outcome; +// prose belongs in Notes. type UnobservableReason string const ( @@ -61,31 +51,18 @@ type CoverageResult struct { BlindFor time.Duration } -// proveCoverage applies the nine coverage checks to one rule's polls and is -// PURE: no HTTP, no files, no clock reads — everything it needs arrives as an -// argument, which is what lets its tests build []Poll literals instead of a -// fixture server. -// -// polls need not be pre-filtered to this rule: proveCoverage selects by -// def.UID itself, exactly as Reduce selects by UID rather than by title — a -// caller handing it a whole log's polls must not have to pre-filter to get a -// correct answer. -// -// Every check always runs, even once an earlier one has already set -// Unobservable: LargestGap and the notes are diagnostics an operator reads on -// exit 2 regardless of which check actually failed. Reason names the FIRST -// check, in the order below, that failed; a later failure still adds its own -// Note. +// proveCoverage applies the nine coverage checks to one rule's polls. PURE: no +// HTTP, no files, no clock reads — everything arrives as an argument. polls need +// not be pre-filtered to this rule (selection is by def.UID). Every check runs +// even after Unobservable is set, so LargestGap and the notes are complete on +// exit 2; Reason names only the FIRST check that failed. func proveCoverage(h Header, polls []Poll, sentinel *time.Time, t ruleTimings, def Definition, from, to time.Time, grace time.Duration) CoverageResult { windowEnd := to.Add(grace) - // pollsForRule (classify.go) is the single filter+sort implementation for - // "select one rule's polls, stably ordered by GrafanaNow" — proveCoverage - // and classifyRule must never carry two independent copies of this - // selection, or one drifting from the other becomes exactly the kind of - // silent membership mismatch this file's checks exist to prevent. + // pollsForRule (classify.go) is the single filter+sort implementation; this + // and classifyRule must not carry two independent copies. rulePolls := pollsForRule(polls, def.UID) var res CoverageResult @@ -119,10 +96,7 @@ func proveCoverage(h Header, polls []Poll, sentinel *time.Time, t ruleTimings, d "requested from %s is before recording started at %s", from.Format(time.RFC3339), h.StartedAt.Format(time.RFC3339))) } - // Filtered once, here, and threaded through every remaining check — - // ruleHeartbeatGap included — rather than re-filtered per check: two - // independent filters over the same polls would only invite one of them - // drifting from the other's membership test. + // Filtered once and threaded through every remaining check. inWindow := inWindowPolls(rulePolls, from, windowEnd) // Check 3 — heartbeat continuity. Data at both ends with a hole in between @@ -144,30 +118,19 @@ func proveCoverage(h Header, polls []Poll, sentinel *time.Time, t ruleTimings, d } } - // Check 5 — health=="nodata". Never fatal here: 96% of the fleet runs - // no_data_state:OK, so treating this as fatal by default would block - // nearly every healthy deploy in an idle environment. Escalating it under - // Policy.NodataIsUnobservable is decide's job, applied directly against - // the raw polls — this pure function has no Policy to consult and must not - // invent one. + // Check 5 — health=="nodata". Never fatal here (most of the fleet runs + // no_data_state:OK, so it would block healthy idle deploys). Escalating + // under Policy.NodataIsUnobservable is decide's job, since this pure + // function has no Policy to consult. if _, sawAny := longestHealthRun(inWindow, "nodata"); sawAny { res.Notes = append(res.Notes, fmt.Sprintf("rule %q: health=nodata observed (not fatal; see --nodata-is-unobservable)", def.Title)) } - // Check 6 — liveness. Absolute only, per poll: GrafanaNow and - // LastEvaluation are both Grafana-domain reads off the SAME response, so - // this is a same-domain comparison and uses raw values — never a delta - // against a previous poll, which reports stale on ~half the polls of a - // perfectly healthy rule (polling runs at intervalSeconds/2). - // - // Skipped only for a poll whose own flags SAY there is nothing to check: - // IsPaused (a zero LastEvaluation is legal only while paused; check 7 is - // its detector) or !Found (no rule, no evaluation; check 8 is its - // detector). Deliberately NOT skipped merely because LastEvaluation is - // zero: ReadLog does no field validation, so a corrupted or hand-edited - // log line can claim found:true, is_paused:false and still carry a zero - // LastEvaluation, and that combination must read as maximally stale - // rather than being silently waved through. + // Check 6 — liveness. Same-domain (GrafanaNow and LastEvaluation are from + // the SAME response), so raw values — never a delta against a previous + // poll, which reports stale ~half the polls of a healthy rule. Skipped only + // for IsPaused (check 7) or !Found (check 8); a zero LastEvaluation on a + // found, unpaused poll is treated as maximally stale, not waved through. var staleCount int var worstStale time.Duration var worstStaleAt time.Time @@ -236,12 +199,10 @@ func proveCoverage(h Header, polls []Poll, sentinel *time.Time, t ruleTimings, d fail(ReasonRuleAbsent, fmt.Sprintf("state endpoint returned no rule on %d poll(s), first at %s", absentCount, absentAt.Format(time.RFC3339))) } - // Check 9 — KeepLast. Two distinct notes, both non-fatal: - // - // DECLARED: the rule's own no_data_state/exec_err_state is configured as - // KeepLast — a standing blind spot whether or not it is ever exercised - // during this particular window. This reads def, not polls, so it fires - // exactly once regardless of poll content. + // Check 9 — KeepLast. Two non-fatal notes: DECLARED (the rule is configured + // with no_data_state/exec_err_state=KeepLast, read from def so it fires once), + // and OBSERVED (an instance reported KeepLast in-window; Reasons keys can be + // comma-joined, so membership via reasonsContain, never a literal index). nds, ees := def.NoDataState, def.ExecErrState for _, lr := range h.Rules { if lr.UID == def.UID { @@ -253,10 +214,6 @@ func proveCoverage(h Header, polls []Poll, sentinel *time.Time, t ruleTimings, d res.Notes = append(res.Notes, fmt.Sprintf( "rule %q: configured with no_data_state/exec_err_state=KeepLast — a stale state can continue past a real fault", def.Title)) } - // OBSERVED: an instance actually reported the KeepLast reason during the - // window. It surfaces only as an instance Reason, and Reasons keys can be - // comma-joined composites, so membership (reasonsContain) is required — - // indexing "KeepLast" directly would miss "KeepLast, MissingSeries". for _, p := range inWindow { if reasonsContain(p.Reasons, keepLastReason) { res.Notes = append(res.Notes, fmt.Sprintf( @@ -269,17 +226,11 @@ func proveCoverage(h Header, polls []Poll, sentinel *time.Time, t ruleTimings, d return res } -// inWindowPolls filters polls to those inside [from, windowEnd] using the -// CROSS-DOMAIN membership test: each poll's Grafana-domain GrafanaNow is -// translated to the runner domain by its OWN skew, and its own skew bound is -// the membership tolerance, so a poll that is genuinely inside the window is -// never excluded by ordinary clock imprecision. -// -// Everything downstream of this filter (health runs, liveness, pause, absence) -// reads the poll's raw fields: GrafanaNow paired with LastEvaluation on the -// SAME response, or one poll's GrafanaNow against the next's, are same-domain -// comparisons and need no translation. Only window membership and check 3's -// two boundary segments cross domains. +// inWindowPolls filters to polls inside [from, windowEnd] via the cross-domain +// membership test: each GrafanaNow is translated to the runner domain by its +// own skew, widened by its skew bound, so clock imprecision never excludes a +// genuinely in-window poll. Everything downstream reads same-domain raw fields; +// only this filter and check 3's boundary segments cross domains. func inWindowPolls(polls []Poll, from, windowEnd time.Time) []Poll { var out []Poll for _, p := range polls { @@ -294,21 +245,14 @@ func inWindowPolls(polls []Poll, from, windowEnd time.Time) []Poll { } // ruleHeartbeatGap finds the largest unobserved span inside [from, windowEnd], -// including the two boundary segments — which is why "data at both ends with a -// hole in the middle" still fails: the segment between the polls just inside -// each edge is exactly what this measures. in must already be filtered to this -// window (inWindowPolls) and sorted by GrafanaNow — proveCoverage computes that -// filter once and threads it through every check, this one included, rather -// than each check re-filtering. +// including the two boundary segments — which is why data at both ends with a +// hole in the middle still fails. in must be filtered (inWindowPolls) and +// sorted by GrafanaNow. // -// The two boundary segments compare a Grafana-domain poll time against the -// runner-domain from/windowEnd, so each is translated by its own poll's skew -// AND widened by that same poll's skew bound — on the side that makes the -// segment larger, never smaller, so an uncertain boundary reads as at least as -// big a gap as it might really be. Understating it by up to the bound would be -// fail-open. The spacing BETWEEN consecutive polls compares two Grafana-domain -// reads to each other — same domain — and uses the raw GrafanaNow difference, -// no bound needed. +// Boundary segments are cross-domain, so each translated poll time is widened +// by its skew bound on the side that makes the gap LARGER (never smaller — an +// uncertain boundary must read as at least as big a gap as it might be). +// Consecutive-poll spacing is same-domain and uses the raw GrafanaNow diff. func ruleHeartbeatGap(in []Poll, from, windowEnd time.Time) (largestGap time.Duration, largestGapAt time.Time) { if len(in) == 0 { return windowEnd.Sub(from), from @@ -332,15 +276,10 @@ func ruleHeartbeatGap(in []Poll, from, windowEnd time.Time) (largestGap time.Dur return largestGap, largestGapAt } -// longestHealthRun returns the longest contiguous wall-clock span during which -// polls — already sorted by GrafanaNow, same-domain spacing — read the given -// rule-level Health, and whether any poll matched it at all. -// -// It detects the span as it accumulates rather than waiting for the run to -// end, so an open-ended run that is still failing at the last poll in the -// window is measured correctly without needing data past the window: waiting -// for the run to "end" would have to assume the best case about what happens -// next, which is exactly what this gate must not do. +// longestHealthRun returns the longest contiguous span of polls reading the +// given rule-level Health, measured incrementally so a run still failing at the +// last in-window poll is measured correctly without assuming anything past the +// window. func longestHealthRun(polls []Poll, health string) (longest time.Duration, sawAny bool) { var runStart time.Time for _, p := range polls { diff --git a/grafana-alertcheck/internal/gate/log.go b/grafana-alertcheck/internal/gate/log.go index 2fdb71d5f..f8f52d01f 100644 --- a/grafana-alertcheck/internal/gate/log.go +++ b/grafana-alertcheck/internal/gate/log.go @@ -47,20 +47,15 @@ type LoggedRule struct { // definitions from the ruler API. ForSeconds float64 `json:"for_seconds"` IntervalSeconds int `json:"interval_seconds"` - // IsPaused is NOT forensic, and is the second load-bearing field here - // beside PollEverySeconds. It is the pause state at record start, which is - // the only moment `skipped` can honestly mean, and decide reads it through - // Header.pausedAtStart rather than reading Definition.IsPaused off a ruler - // read taken after the window had already closed. See that method for what - // goes wrong the other way. + // IsPaused is load-bearing (beside PollEverySeconds): the pause state at + // record start, the only moment `skipped` can honestly mean. decide reads + // it via Header.pausedAtStart, never a ruler read taken after the window. IsPaused bool `json:"is_paused"` NoDataState string `json:"no_data_state"` ExecErrState string `json:"exec_err_state"` - // PollEverySeconds is the cadence this recording ACTUALLY used, after any - // --poll-interval override. Load-bearing, not forensic: check derives - // maxGap from it and never re-derives it from the definitions. Getting - // that wrong is fail-open in the faster-override direction — a real - // recorder gap would pass silently. + // PollEverySeconds is the cadence this recording ACTUALLY used. Load-bearing: + // check derives maxGap from it, never from the definitions — getting that + // wrong is fail-open in the faster-override direction. PollEverySeconds float64 `json:"poll_every_seconds"` } @@ -128,15 +123,11 @@ type Poll struct { IsPaused bool `json:"is_paused"` Histogram map[string]int `json:"histogram,omitempty"` // written, never analysed // Reasons counts this poll's non-empty instance reasons, e.g. - // {"NoData":1091,"Error":14}; nil when none. Reporting-only, and the ONLY - // place composite states stay visible: they are canonical normal (so they - // are dropped from Abnormal) and `totals` never carries composite keys. - // - // The KEYS are raw reason strings and can be comma-joined composites - // ("KeepLast, MissingSeries") — newer Grafana versions join several - // reasons into one. So any consumer, the coverage proof's KeepLast note - // included, must test membership across the keys with reasonNames and must - // NEVER index a literal key: reasons["KeepLast"] misses every composite. + // {"NoData":1091,"Error":14}; nil when none. Reporting-only, and the only + // place composite states stay visible (they are canonical normal, dropped + // from Abnormal). Keys are raw reason strings and can be comma-joined + // composites ("KeepLast, MissingSeries"), so consumers must test membership + // via reasonNames and never index a literal key. Reasons map[string]int `json:"reasons,omitempty"` // Abnormal holds the instances whose CANONICAL state is not normal. // "Normal (NoData)" and "Normal (Error)" are canonical normal and are @@ -290,16 +281,10 @@ func (r *Reducer) seedFrom(polls []Poll) { } } -// stateRuleByUID picks one rule out of a state-endpoint response BY UID, and -// nil means the response is an authoritative "the rule is absent". -// -// Never by title: the ?rule_name= filter is a title filter, and a filtered -// response can carry several rules sharing one title (the known 2-way -// collision), so picking the first would silently watch the wrong rule. This -// is the single implementation of that selection for the package — Reduce -// above and the drain wait (check.go) both call it, for the same reason -// pollsForRule (classify.go) is shared between proveCoverage and classifyRule: -// two copies of a membership test are two chances for one to drift. +// stateRuleByUID picks one rule out of a state response BY UID (nil = the +// authoritative "rule absent"). Never by title: the ?rule_name= filter is a +// title filter and can return several rules sharing a title. The single +// selection for the package — Reduce and the drain wait both use it. func stateRuleByUID(rules []StateRule, uid string) *StateRule { for i := range rules { if rules[i].UID == uid { @@ -322,18 +307,11 @@ func reasonNames(reason, want string) bool { } // VerifyNormalInstancesVisible checks, on a first observation, that the state -// endpoint really does return normal instances and not only the abnormal ones. -// If it ever stops doing so, the reduction's "keep the non-normal instances" -// becomes "keep everything the API happened to send" and the transition markers -// lose their ground truth — a silent fail-open. So this is verified at start, -// never assumed. -// -// The counts are summed over every totals key whose LOWERCASED name is -// "normal" or "inactive". Never index one literal key: the captured -// vocabulary is mixed across rules ({"alerting":445,"normal":2004} on one, -// {"firing":2,"inactive":363} on another) and its case has already drifted -// from the original recon. Composite states never appear in totals — Grafana -// counts a "Normal (NoData)" instance under normal. +// endpoint really returns normal instances: if it ever stops, the reduction's +// "keep the non-normal" becomes "keep everything the API sent", a silent +// fail-open in the transition markers. Counts are summed over every totals key +// whose lowercased name is "normal" or "inactive" — never a literal key, since +// the vocabulary is mixed and its case has drifted. func VerifyNormalInstancesVisible(rules []StateRule) error { for _, r := range rules { var claimed int @@ -453,18 +431,12 @@ func (w *Writer) WritePoll(p Poll) error { return nil } -// Stop finishes recording in a fixed order that must not be rearranged: let -// the in-flight write finish (the mutex), append the stopped sentinel, fsync, -// then release. Any other order can leave a log whose last durable byte is a -// sentinel that was never actually preceded by the polls it vouches for. -// -// Stop writes the sentinel with the recorder's OWN stop time and makes no -// comparison against `to` — watch never knows `to` or the transition grace. -// check does that comparison, after this writer has exited. -// -// Calling Stop twice is a no-op: watch reaches it from both a signal handler -// and a defer, and a second sentinel would be indistinguishable from a second -// writer. +// Stop finishes recording in a fixed order that must not be rearranged: let the +// in-flight write finish, append the sentinel, fsync, release — any other order +// can leave a sentinel that was never preceded by the polls it vouches for. +// The sentinel uses the writer's OWN stop time; check does the `to` comparison +// after this has exited. Calling Stop twice is a no-op (watch reaches it from a +// signal handler and a defer). func (w *Writer) Stop() error { w.mu.Lock() defer w.mu.Unlock() @@ -505,18 +477,11 @@ func (w *Writer) Close() error { return nil } -// ReadLogHeader reads ONLY line 1 and is the one read of a log that a writer -// may still hold. That is safe for exactly one line and for no other: the -// header is written once, by watch's parent, before any child appends a byte, -// and the file is opened O_APPEND and never O_TRUNC, so line 1 is complete and -// immutable for the whole life of the recording. -// -// It exists so check can fail closed EARLY: the log's identity, the rule set -// and the cadences are all knowable at the start, and discovering a wrong URL -// or an unresolvable rule after a ten-minute wait helps nobody. It is advisory -// only — the authoritative read is still ReadLog, once, after the writer has -// exited, and check re-validates the identity against that header rather than -// trusting this one. +// ReadLogHeader reads ONLY line 1 — the one read safe while a writer may still +// hold the log. The header is written once by watch's parent before any child +// appends a byte, so line 1 is immutable. It lets check fail closed EARLY on a +// wrong URL or unresolvable rule; it is advisory only, and the authoritative +// identity read is still ReadLog after the writer exits. func ReadLogHeader(path string) (Header, error) { f, err := os.Open(path) if err != nil { @@ -555,11 +520,9 @@ func ReadLogHeader(path string) (Header, error) { // append to can only produce a shorter window than the one that was recorded. // // The parse rules are deliberately the crudest possible: the header must be -// line 1 with a matching schema version, and ANY unparseable line — -// including the last one, and including a last line that follows a sentinel — -// is an error, full stop. No heuristics, no discarding an untidy tail: a -// truncated log is evidence that something killed the recorder, which is -// exactly what must not pass. +// line 1 with a matching schema version, and ANY unparseable line — including +// the last, or one after a sentinel — is an error, full stop. A truncated log +// is evidence something killed the recorder, which must not pass. func ReadLog(path string) (Header, []Poll, *time.Time, error) { b, err := os.ReadFile(path) if err != nil { diff --git a/grafana-alertcheck/internal/gate/schedule.go b/grafana-alertcheck/internal/gate/schedule.go index 4541317e5..f2b9bd759 100644 --- a/grafana-alertcheck/internal/gate/schedule.go +++ b/grafana-alertcheck/internal/gate/schedule.go @@ -8,26 +8,22 @@ import ( "time" ) -// SkewHardLimit is the largest clock skew between this runner and Grafana that -// a run tolerates before it errors out. Exported so the CLI can report it -// verbatim next to a measured skew instead of keeping its own mirrored copy. +// SkewHardLimit is the largest runner↔Grafana clock skew a run tolerates. +// Exported so the CLI reports it verbatim next to a measured skew. const SkewHardLimit = 60 * time.Second -// fromFutureTolerance is how far ahead of the runner's own clock a supplied -// `from` may sit before check refuses it: the same 60s as SkewHardLimit, -// because the only legitimate reason for a `from` in the future is clock -// disagreement between the deploy step and the check step, and that is bounded -// by the same figure. It is once-per-run input validation, not a per-rule -// coverage check, so Check applies it and proveCoverage does not. +// fromFutureTolerance is how far ahead of the runner's clock a `from` may sit +// before check refuses it — the same 60s as SkewHardLimit, since a future `from` +// can only be clock disagreement. Once-per-run input validation, not a coverage +// check, so Check applies it and proveCoverage does not. const fromFutureTolerance = 60 * time.Second -// minDrainTimeout is the floor on drainTimeout, which is otherwise -// 2 x max(intervalSeconds). Without the floor, a fleet of very tight rules -// would derive a drainTimeout too short to let a healthy in-flight poll land. +// minDrainTimeout floors drainTimeout (otherwise 2 × max intervalSeconds) so a +// fleet of tight rules still lets a healthy in-flight poll land. const minDrainTimeout = 2 * time.Minute -// graceWarnFraction is the share of the requested window above which -// transitionGrace is worth warning about. +// graceWarnFraction is the share of the window above which transitionGrace is +// worth warning about. const graceWarnFraction = 0.25 // ruleTimings groups the per-rule thresholds derived from a rule's poll @@ -52,12 +48,9 @@ type globalTimings struct { drainTimeout time.Duration } -// newRuleTimings derives one rule's thresholds from its fully-resolved poll -// cadence and its evaluation interval. pollEvery arrives already resolved for -// the caller's mode — the default, the operator's --poll-interval override, or -// (in log mode) the cadence recorded in the log header. Deriving pollEvery -// inline here, instead of accepting it as an input, would let a caller in the -// wrong mode compute maxGap against the wrong authority. +// newRuleTimings derives one rule's thresholds from its resolved cadence and +// evaluation interval. pollEvery is an input — already resolved for the +// caller's mode — so no caller can compute maxGap against the wrong authority. func newRuleTimings(pollEvery time.Duration, intervalSeconds int) ruleTimings { interval := time.Duration(intervalSeconds) * time.Second maxGap := 2 * pollEvery @@ -76,14 +69,10 @@ func defaultPollEvery(intervalSeconds int) time.Duration { return time.Duration(intervalSeconds) * time.Second / 2 } -// DeriveTimings computes every resolved rule's ruleTimings, keyed by UID, plus -// the shared globalTimings, from resolved definitions and watch's optional -// --poll-interval override (0 = no override: use each rule's default of half -// its own interval). A supplied override is used verbatim for every rule and is -// never clamped down to the default even when it exceeds intervalSeconds/2 — -// that case is reported back as a note, not silently corrected or refused, -// because clamping would defeat the one knob an operator has for making a tight -// schedule fit. +// DeriveTimings computes every resolved rule's ruleTimings (keyed by UID) plus +// the shared globalTimings. A non-zero override is used verbatim for every rule +// and never clamped to the default — an override above intervalSeconds/2 widens +// maxGap and is reported as a note, not corrected. func DeriveTimings(defs []Definition, override time.Duration) (rules map[string]ruleTimings, global globalTimings, notes []string) { rules = make(map[string]ruleTimings, len(defs)) for _, d := range defs { @@ -99,10 +88,8 @@ func DeriveTimings(defs []Definition, override time.Duration) (rules map[string] } rules[d.UID] = newRuleTimings(pollEvery, d.IntervalSeconds) } - // In this mode the defs ARE the start-of-step snapshot — watch resolves - // them before it detaches, and single-step check before its first - // observation — so they can answer what was paused when the window opened. - // Only the log-mode counterpart below has to look elsewhere. + // In this mode defs ARE the start-of-step snapshot, so they answer what was + // paused at the window open; only the log-mode counterpart uses the header. return rules, deriveGlobalTimings(defs, pausedSet(defs)), notes } @@ -117,39 +104,19 @@ func pausedSet(defs []Definition) map[string]bool { return paused } -// DeriveTimingsFromLog is DeriveTimings' log-mode counterpart, and having one -// authority for the cadence is the whole reason it exists as a separate -// function. pollEvery comes from the header — the cadence the recording -// ACTUALLY used, after any --poll-interval override — and maxGap and -// healthGrace follow from it. Re-deriving pollEvery from defs here would -// compare gaps recorded at the override cadence against thresholds computed -// from the default: exit 2 on a clean window when the override was slower, and, -// worse, a real recorder gap passing silently when it was faster. +// DeriveTimingsFromLog is DeriveTimings' log-mode counterpart: pollEvery comes +// from the header (the cadence actually used), not the definitions — re-deriving +// it here would compare recorded gaps against default-cadence thresholds, an +// exit 2 on a clean window (slower override) or a silently passing recorder gap +// (faster override). evalStaleAfter still comes from defs (2 × intervalSeconds). // -// evalStaleAfter still comes from defs (2 x intervalSeconds): it is a property -// of the rule's own evaluation cadence and is unaffected by how often the gate -// polled. +// Three header shapes are hard errors rather than a best-effort derivation, +// because each would silently widen a threshold: a rule with no matching +// definition, a non-positive recorded cadence, and a UID listed twice +// (last-one-wins would widen maxGap on a corrupt log). // -// Three shapes of header are errors rather than a best-effort derivation, -// because each one would otherwise widen a threshold silently: -// -// - a rule with no matching definition — a log that names a rule nobody can -// resolve cannot have that rule's coverage proved; -// - a non-positive recorded cadence — a log that cannot say how often it was -// written cannot have maxGap derived, and defaulting the cadence would -// prove a window that was never observed; -// - the same UID twice — last-one-wins would take whichever cadence happened -// to be written last, and a slower duplicate widens maxGap. That is a -// fail-open reachable through nothing but log corruption. -// -// It checks only the header-to-defs direction. The opposite direction — a -// resolved definition absent from the header — is NOT this function's to -// judge: it belongs to Check's log-identity validation, the only caller that -// knows both sets and can name the mismatch. Without that check a definition -// simply gets no timings entry, and a downstream lookup would read a zero -// maxGap: fail-closed (every gap exceeds it) but silent, so Check must reject -// the set mismatch by name rather than let a rule fail for an unexplained -// reason. +// It checks only the header-to-defs direction. A definition absent from the +// header is Check's log-identity validation to judge, not this function's. func DeriveTimingsFromLog(h Header, defs []Definition) (rules map[string]ruleTimings, global globalTimings, err error) { byUID := make(map[string]Definition, len(defs)) for _, d := range defs { @@ -180,23 +147,15 @@ func DeriveTimingsFromLog(h Header, defs []Definition) (rules map[string]ruleTim return rules, deriveGlobalTimings(defs, h.pausedAtStart()), nil } -// deriveGlobalTimings computes transitionGrace and drainTimeout over defs. +// deriveGlobalTimings computes transitionGrace and drainTimeout over defs. A +// rule skipped at the window open is excluded from the transitionGrace max (its +// `for` can never fire in-window); drainTimeout runs over every resolved rule. // -// A rule paused before the window opened — skipped — is excluded from the -// transitionGrace max: its `for` value can never fire during the window, so -// counting it would only inflate the wait past what any watched rule actually -// needs. drainTimeout carries no such exclusion and still runs over every -// resolved rule. -// -// "Before the window opened" is the whole content of that exclusion, so the -// authority is pausedAtStart and NEVER Definition.IsPaused: in log mode the -// definitions are re-resolved after the window closed. Reading them instead -// was a fail-open, and a quiet one. transitionGrace is what lets a condition -// arising just before `to` be seen when it surfaces at to + `for`, and -// windowEnd is BOTH the classification bound and the collection deadline — so -// a rule somebody paused after `to` dropped out of the max, the grace -// collapsed, the surfacing poll was never even recorded, and the run reported -// clean. With one watched rule the shrink is total. +// The exclusion authority is pausedAtStart, never Definition.IsPaused: log-mode +// defs are re-resolved after the window closed. Reading the late definitions +// was a quiet fail-open — a rule paused after `to` would drop out of the max, +// collapse the grace past windowEnd (the classification bound AND collection +// deadline), and pass a window the surfacing poll was never recorded for. func deriveGlobalTimings(defs []Definition, pausedAtStart map[string]bool) globalTimings { var g globalTimings var maxInterval time.Duration @@ -225,17 +184,11 @@ type Scheduler struct { every map[string]time.Duration } -// NewScheduler builds a Scheduler over per-rule cadences (keyed by UID), -// staggering each rule's initial next-due time across [0, pollEvery) so the -// fleet does not start phase-aligned. The burst bound CheckBudget enforces -// depends on that: an already-staggered fleet only re-aligns by chance, -// briefly, not by construction. -// -// It takes cadences rather than whole ruleTimings on purpose: a scheduler -// decides when to poll and nothing else, so it must not be handed maxGap, -// healthGrace or evalStaleAfter. Those are coverage thresholds, they are -// applied by the pure layer at classification time, and the recorder that -// drives this scheduler never applies them at all. +// NewScheduler builds a Scheduler over per-rule cadences, staggering each +// rule's initial next-due time across [0, pollEvery) so a phase-aligned fleet +// (which would void CheckBudget's burst bound) never arises by construction. +// It takes cadences, not ruleTimings: a scheduler only decides when to poll and +// must not be handed coverage thresholds it never applies. func NewScheduler(every map[string]time.Duration, now time.Time) *Scheduler { s := &Scheduler{ next: make(map[string]time.Time, len(every)), @@ -252,13 +205,10 @@ func NewScheduler(every map[string]time.Duration, now time.Time) *Scheduler { return s } -// Due returns the UIDs whose next-due time has arrived, earliest-due-first. -// Ties (equal next-due time) break by tightest cadence first: the burst bound -// assumes a newly-due tight rule waits at most for one in-flight request, which -// only holds if a simultaneous batch serves the tightest rule ahead of slacker -// ones. A tie-break that instead followed map iteration order would silently -// void that assumption — nothing else would fail until a phase-aligned fleet -// opened a mid-run gap in production. +// Due returns the due UIDs, earliest-due-first. Ties break by tightest cadence +// first: the burst bound assumes a newly-due tight rule waits at most one +// in-flight request, which only holds if a simultaneous batch serves the +// tightest rule first. A map-order tie-break would silently void that. func (s *Scheduler) Due(now time.Time) []string { var due []string for uid, t := range s.next { @@ -293,11 +243,9 @@ func (s *Scheduler) Mark(uid string, now time.Time) error { return nil } -// earliestDue returns the earliest scheduled next-due time, and false when the -// scheduler holds no rules at all. The recorder's loop waits exactly that long -// instead of waking on a fixed tick: a fixed tick either polls a slack rule -// early — spending request budget the schedule already accounted for — or wakes -// too late for the tightest rule and opens a gap inside its own maxGap. +// earliestDue returns the earliest next-due time (false when empty). The loop +// waits exactly that long instead of a fixed tick, which would poll slack rules +// early (wasting budget) or wake late for the tightest rule (opening a gap). func (s *Scheduler) earliestDue() (time.Time, bool) { var earliest time.Time ok := false @@ -310,21 +258,18 @@ func (s *Scheduler) earliestDue() (time.Time, bool) { return earliest, ok } -// CheckBudget proves at start that a fully resolved schedule can actually be -// served. t and measured are both keyed by rule UID, and measured must carry -// every UID in t: a rule this run never measured cannot have its budget -// proved, and a silent zero-duration default would be exactly the kind of -// pass-on-an-unproven-window bug this check exists to catch. It fails when any -// of three conditions holds: +// CheckBudget proves at start that a fully resolved schedule can be served. t +// and measured are keyed by UID, and measured must carry every UID in t (a rule +// never measured cannot have its budget proved). It fails on any of three +// conditions: // -// - utilization: the long-run request rate exceeds what concurrency serves; -// - a single rule's own request cannot fit inside its own cadence; -// - the burst bound: the slowest measured request is slower than the fleet's -// tightest cadence, which — even under earliest-due-first ordering — can -// open a mid-run gap bigger than that rule's maxGap. +// - utilization — the long-run request rate exceeds the concurrency; +// - a single rule's request cannot fit inside its own cadence; +// - the burst bound — the slowest measured request is slower than the fleet's +// tightest cadence, which can open a mid-run gap beyond that rule's maxGap. // -// The message names only the three controls an operator actually has: -// concurrency, poll-interval, and the alert list. +// The message names only the three operator controls: concurrency, +// poll-interval, and the alert list. func CheckBudget(t map[string]ruleTimings, measured map[string]time.Duration, concurrency int) error { if len(t) == 0 { return nil diff --git a/grafana-alertcheck/internal/gate/source.go b/grafana-alertcheck/internal/gate/source.go index 82dcdc392..bebb1839e 100644 --- a/grafana-alertcheck/internal/gate/source.go +++ b/grafana-alertcheck/internal/gate/source.go @@ -44,11 +44,9 @@ type Observation struct { Latency time.Duration // t_send through the full body read — see requestResult.Latency } -// TransportError marks a failure worth retrying: a non-2xx response, a -// network failure, or a body that failed to parse. It is never a deleted rule -// (an authoritative 2xx with no matching rule is not this) and never a clock -// problem (a missing/unparseable Date header or an out-of-bounds skew is a -// hard error instead — see doRequest). Never conflate them. +// TransportError marks a failure worth retrying: a non-2xx response, a network +// failure, or a body that failed to parse. Not a deleted rule (an authoritative +// 2xx) and not a clock problem (a hard error — see doRequest). type TransportError struct { Err error Status int // 0 when the failure never got a status (network/transport failure) @@ -63,14 +61,10 @@ func (e *TransportError) Error() string { func (e *TransportError) Unwrap() error { return e.Err } -// RetryExhaustedError is what retryTransport returns once it gives up after -// too many sequential *TransportError failures. It deliberately does not -// implement Unwrap into the underlying *TransportError: once retries are -// exhausted the result is a hard, terminal failure, and -// errors.AsType[*TransportError] must never re-classify it as retryable. -// Cause is still exposed as a plain field (and folded into Error()'s text) so -// a caller can log or inspect it; it just cannot flow back into the retry -// classification. +// RetryExhaustedError is the hard, terminal failure retryTransport returns once +// it gives up. It deliberately omits Unwrap into *TransportError so +// errors.AsType can never re-classify it as retryable; Cause stays a plain +// field for logging only. type RetryExhaustedError struct { Failures int Cause error @@ -259,27 +253,20 @@ type requestResult struct { Skew time.Duration // serverDate - (t_send+t_headers)/2, signed SkewBound time.Duration // (t_headers-t_send)/2 — RTT/2 to the response headers // Latency spans t_send through the full body read: the budget check needs - // the whole poll's wall time, or a schedule feasibility check that only - // sees header latency goes optimistic — fail-open. It does not include the - // caller's subsequent JSON parse (ParseState/ParseDefinitions run outside - // doRequest); if the budget accounting ever needs parse time folded in too, - // extend here rather than approximating it at the call site. + // the whole poll's wall time (header-only latency would be fail-open). The + // caller's JSON parse runs outside doRequest; extend here if that ever must + // be folded in. Latency time.Duration } -// doRequest performs one HTTP GET and classifies the outcome: a network -// failure, a non-2xx status, or a body-read failure is retryable -// (*TransportError); a missing or unparseable Date header, or a skew beyond -// SkewHardLimit, is a hard error — retrying can never fix either, so neither -// may enter the backoff loop. +// doRequest performs one HTTP GET and classifies the outcome: network failure, +// non-2xx, or body-read failure is retryable (*TransportError); a missing or +// unparseable Date header or a skew beyond SkewHardLimit is a hard error — +// retrying can never fix either, so neither enters the backoff loop. // -// The Date-header/skew check runs for every endpoint this hits, including -// /api/health, and not only the state endpoint whose timestamps the gate -// actually compares. Deliberate: a skewed clock discovered only once RuleState -// starts polling is a skew that has already masked whatever /api/health and -// the ruler read reported; failing closed at the first response catches it -// before any of that is trusted, and every response comes with a Date header -// for free. +// The Date/skew check runs on every endpoint (even /api/health): a skew only +// noticed once RuleState starts polling has already masked earlier reads, so it +// fails closed on the first response. func (s *httpSource) doRequest(ctx context.Context, path string) (requestResult, error) { req, buildErr := http.NewRequestWithContext(ctx, http.MethodGet, s.baseURL+path, nil) if buildErr != nil { @@ -337,12 +324,10 @@ func (s *httpSource) doRequest(ctx context.Context, path string) (requestResult, return requestResult{Body: b, ServerDate: serverDate, Skew: signedSkew, SkewBound: bound, Latency: latency}, nil } -// retryTransport runs fn, retrying with backoff only while it fails with a -// *TransportError — any other error is a hard error and returns immediately, -// never retried. failures counts consecutive *TransportError results; -// exceeding maxFailures gives up with a wrapped hard error. The wait between -// attempts goes through clock.After so a test with a fake Clock never sleeps on -// real time. +// retryTransport runs fn, retrying with backoff only on *TransportError — any +// other error returns immediately. failures counts consecutive *TransportError +// results; exceeding maxFailures gives up with a wrapped hard error. Waits go +// through clock.After so a fake Clock never sleeps real time. func retryTransport[T any](ctx context.Context, clock Clock, maxFailures int, backoffBase, backoffCap time.Duration, fn func() (T, error)) (T, error) { var zero T failures := 0 diff --git a/grafana-alertcheck/internal/gate/watch.go b/grafana-alertcheck/internal/gate/watch.go index 6f4b7fa15..74dd3945b 100644 --- a/grafana-alertcheck/internal/gate/watch.go +++ b/grafana-alertcheck/internal/gate/watch.go @@ -20,19 +20,13 @@ import ( const DaemonChildFlag = "--daemon-child" // ReadyFDFlag names the inherited descriptor the child reports readiness on. -// The parent passes the write end of a pipe as descriptor 3 and waits for one -// byte, so "the recorder is running" is a POSITIVE signal from the child -// itself — it has read the header, taken the log's flock and entered its poll -// loop — and not an assumption drawn from surviving a timer. A timer cannot -// tell a healthy child from one that is about to die on a slow runner, and -// getting that wrong means watch returns success over a recording that never -// happened. +// "Ready" is a POSITIVE byte from the child (header read, flock taken, in its +// poll loop), never a timer heuristic — a timer can't tell a healthy child from +// one about to die on a slow runner. const ReadyFDFlag = "--ready-fd" -// childReadyTimeout bounds that wait. Everything before the signal is local — -// fork, exec, one read of a log holding a header and a handful of polls — so -// the real figure is milliseconds; this is loose enough for a badly overloaded -// runner and still fails closed rather than hanging the pipeline. +// childReadyTimeout bounds that wait. Everything before the signal is local, so +// it is loose enough for an overloaded runner yet still fails closed. const childReadyTimeout = 30 * time.Second // daemonLogTailBytes bounds how much of a dead child's output the parent @@ -41,18 +35,12 @@ const daemonLogTailBytes = 4096 // WatchConfig is the record step's whole input. // -// It has no To field and must never gain one: watch writes the stopped -// sentinel with its OWN stop time and makes no comparison against `to`, which -// only check knows. Passing `to` here would give two components an -// opinion about the same comparison, and the recorder's opinion is the one -// that cannot be trusted — it exits before the grace it would have to wait for. +// It has no To field (and must never gain one): watch writes the sentinel with +// its OWN stop time and makes no `to` comparison — only check knows `to`, and +// the recorder exits before the grace it would have to wait for. // -// It has no States field either, and watch has no --states flag: recording is -// deliberately unfiltered. The reduction keeps every non-normal instance and -// the transition markers key off the same predicate, so neither consults -// States. The payoff is real — because the log is raw evidence, one recording -// can be re-classified under different --states without re-recording — and the -// Header carries no States field for the same reason. +// It has no States field either: recording is unfiltered, so the same raw log +// can be re-classified under different --states without re-recording. type WatchConfig struct { // URL and Token are the connection details. The CLI reads both from the // environment and never from a flag; Token is never logged and never @@ -137,20 +125,11 @@ func (cfg WatchConfig) validate() error { return nil } -// Watch is the record step's parent process. It returns only once the window -// is genuinely being recorded: -// -// version gate -> resolve definitions and names -> derive timings -> -// open the log and write the header -> ONE observation of every non-skipped -// rule -> verify normal instances are visible -> check the schedule budget -> -// detach the child -> wait for the child to report that it is recording -> -// write the pidfile -> return. -// -// The first-observation wait is not a convenience. Returning before it would -// leave the deploy inside [from, first_poll] with no evidence — the exact -// blind interval the record-then-check split exists to remove — and it is -// also what surfaces auth, name-resolution and parse failures BEFORE -// deploy.sh runs rather than ten minutes later. +// Watch is the record step's parent process, returning only once the window is +// genuinely being recorded (version gate, resolve, write header, one +// observation per rule, budget check, then detach and await the child's +// readiness). The first-observation wait is what surfaces auth, name-resolution +// and parse failures before deploy.sh runs, rather than ten minutes later. func Watch(ctx context.Context, cfg WatchConfig) error { cfg = cfg.withDefaults() if err := cfg.validate(); err != nil { @@ -362,13 +341,10 @@ func openRecording(ctx context.Context, cfg WatchConfig, src Source, writer *Wri return nil, err } - // A rule whose DEFINITION says is_paused is skipped: it is not waited for, - // not scheduled and never polled. Waiting for one either hangs forever or - // errors before the deploy, and recording polls for it would report an - // in-window pause (coverage check 7) for a rule that was already paused - // when the window opened — turning a skipped rule's exit 1 into an exit 2. - // The header still names it, with is_paused true, so check reports it as - // skipped from the definitions. + // A rule whose definition says is_paused is skipped (never waited for or + // polled); polling it would report an in-window pause (check 7) for a rule + // already paused at the open. The header still names it, so check reports + // it skipped from the definitions. var active []Definition activeTimings := make(map[string]ruleTimings, len(resolved)) for _, d := range resolved { @@ -425,21 +401,11 @@ func loggedRules(defs []Definition, rt map[string]ruleTimings) []LoggedRule { return out } -// firstObservations takes one observation of every rule in active, verifies -// that normal instances are visible in those very responses, and reduces each -// into the poll record that IS the window's first heartbeat — plus the measured -// latency of each, which is the only honest input to the budget check (a fixed -// estimate is worthless when one rule's payload is ~230x another's). -// -// Both entry paths share it: watch's parent, before it detaches, and -// single-step check's measurement pass, which keeps the polls as evidence -// rather than writing them to a log. Keeping one implementation is the point — -// the instance-visibility verification and the "absent is a warning, not an -// error" rule are exactly the places where two copies would silently drift, and -// a drift in either direction is fail-open. -// -// polls come back in `active` order, so a log written from them is byte-stable -// for a given set of observations. +// firstObservations takes one observation of every active rule, verifies normal +// instances are visible in those very responses, and reduces each into the +// window's first heartbeat, plus measured latency (the only honest budget +// input). Watches parent and single-step check both share it. Polls return in +// `active` order, so a log written from them is byte-stable. func firstObservations(ctx context.Context, src Source, active []Definition, reducer *Reducer, concurrency int, notes io.Writer) ([]Poll, map[string]time.Duration, error) { @@ -486,14 +452,11 @@ func firstObservations(ctx context.Context, src Source, active []Definition, red } // observeAll polls every rule in uids concurrently, bounded by concurrency, -// and returns one Observation per rule that answered. Every rule is polled by -// TITLE (the ?rule_name= filter is a title filter) and selected out of the -// response by UID — a filtered response can carry several rules sharing one -// title. -// -// It returns the successful observations alongside the first error in UID -// order, so a caller that wants to keep the good heartbeats can, and the error -// message is the same on every run. +// returning one Observation per rule that answered. Each rule is polled by +// TITLE (the ?rule_name= filter is a title filter) and selected by UID — a +// filtered response can carry several rules sharing a title. Returns the +// successes alongside the first error in UID order, so a caller can keep the +// good heartbeats. func observeAll(ctx context.Context, src Source, titles map[string]string, uids []string, concurrency int) (map[string]Observation, error) { if concurrency < 1 { concurrency = 1 @@ -641,15 +604,10 @@ func reportReady(fd int) error { } // childSchedule derives what the child polls, and how often, from the header -// alone. The cadence comes from PollEverySeconds — the cadence the recording -// actually uses — and is never re-derived from the rule's evaluation interval, -// which would be a second authority for the same value and is fail-open in the -// faster-override direction. Paused rules are excluded here for the same reason -// the parent never polls them. -// -// It returns cadences and nothing else. maxGap, healthGrace and evalStaleAfter -// are coverage thresholds applied by the pure layer at classification time, so -// the recorder must not carry them: it would only be able to misuse them. +// alone. Cadence comes from PollEverySeconds (the cadence actually used), never +// re-derived from the evaluation interval; paused rules are excluded. It +// returns cadences only — the recorder must not carry coverage thresholds it +// has no business applying. func childSchedule(h Header) (titles map[string]string, cadence map[string]time.Duration, err error) { titles = make(map[string]string, len(h.Rules)) cadence = make(map[string]time.Duration, len(h.Rules)) @@ -684,15 +642,13 @@ type watchLoopConfig struct { Clock Clock } -// watchLoop is the child's whole working life: poll the rules that are due, -// reduce each observation to one poll record, append it, and — on a clean stop -// only — finish the log with the stopped sentinel. +// watchLoop is the child's whole working life: poll due rules, reduce each +// observation to a poll record, append it, and — on a clean stop only — finish +// the log with the stopped sentinel. // -// The sentinel policy is the load-bearing part. A clean stop (a signal, or -// Until) writes it; a hard error does NOT. A recorder that died must look -// exactly like a coverage gap to check, because it is one — writing a -// sentinel on the way out of a failure would hand check a "recording finished" -// claim about a window that stopped being observed. +// The sentinel policy is load-bearing: a clean stop (signal or Until) writes +// it; a hard error does NOT. A recorder that died must look exactly like a +// coverage gap to check, because it is one. func watchLoop(ctx context.Context, cfg watchLoopConfig) error { sched := NewScheduler(cfg.Cadence, cfg.Clock.Now()) From 75e4fc88d4de9040042c2303820b4aeeb89d6cfa Mon Sep 17 00:00:00 2001 From: Bartek Tofel Date: Mon, 7 Sep 2026 17:18:27 +0200 Subject: [PATCH 34/43] fix: resolve conflict --- grafana-alertcheck/internal/gate/classify.go | 5 ----- 1 file changed, 5 deletions(-) diff --git a/grafana-alertcheck/internal/gate/classify.go b/grafana-alertcheck/internal/gate/classify.go index 9d881334a..d5d9c8e12 100644 --- a/grafana-alertcheck/internal/gate/classify.go +++ b/grafana-alertcheck/internal/gate/classify.go @@ -216,14 +216,9 @@ func classifyRule(def Definition, polls []Poll, from, windowEnd time.Time, badSt tl.episodes = append(tl.episodes, episode{start: tl.episodeStart, end: end, closedByRealClear: real}) tl.badOpen = false } -<<<<<<< HEAD // onsetOf resolves a fresh episode's start: the instance's own ActiveAt, // translated to the runner domain by this poll's skew, clamped to // [from, windowEnd]. -======= - // onsetOf is a fresh episode's start: the translate ActiveAt, clamped to - // never read as starting before the window opened. ->>>>>>> 056b9146 (chore: shorten comments) onsetOf := func(p Poll, inst Instance) time.Time { start := runnerTime(p, inst.ActiveAt) if start.Before(from) { From 63e3242e01f089e79d2159371989dea4512a7db6 Mon Sep 17 00:00:00 2001 From: Bartek Tofel Date: Wed, 2 Sep 2026 15:55:29 +0200 Subject: [PATCH 35/43] chore: use testify's require in tests --- .../cmd/grafana-alertcheck/check_test.go | 34 +- .../cmd/grafana-alertcheck/list_test.go | 31 +- .../cmd/grafana-alertcheck/main_test.go | 39 +- .../cmd/grafana-alertcheck/table_test.go | 71 +- .../cmd/grafana-alertcheck/watch_test.go | 35 +- grafana-alertcheck/go.mod | 4 + grafana-alertcheck/go.sum | 4 + grafana-alertcheck/internal/gate/check.go | 13 + .../internal/gate/check_test.go | 659 ++++++------------ .../internal/gate/classify_test.go | 410 ++++------- .../internal/gate/coverage_test.go | 202 ++---- .../internal/gate/duration_test.go | 16 +- .../internal/gate/jsonreq_test.go | 54 +- grafana-alertcheck/internal/gate/log_test.go | 467 ++++--------- .../internal/gate/parse_ruler_test.go | 98 +-- .../internal/gate/parse_state_test.go | 266 +++---- .../internal/gate/resolve_test.go | 222 ++---- .../internal/gate/schedule_test.go | 227 ++---- .../internal/gate/source_test.go | 282 +++----- .../internal/gate/watch_daemon_test.go | 119 +--- .../internal/gate/watch_test.go | 307 +++----- 21 files changed, 1104 insertions(+), 2456 deletions(-) create mode 100644 grafana-alertcheck/go.sum diff --git a/grafana-alertcheck/cmd/grafana-alertcheck/check_test.go b/grafana-alertcheck/cmd/grafana-alertcheck/check_test.go index 99e8ebc3a..c7c797a0b 100644 --- a/grafana-alertcheck/cmd/grafana-alertcheck/check_test.go +++ b/grafana-alertcheck/cmd/grafana-alertcheck/check_test.go @@ -4,10 +4,10 @@ import ( "bytes" "errors" "os" - "strings" "testing" "github.com/smartcontractkit/chainlink-testing-framework/grafana-alertcheck/internal/gate" + "github.com/stretchr/testify/require" ) // The exit-code mapping, pinned directly against exitCode with no network @@ -27,9 +27,7 @@ func TestExitCode(t *testing.T) { } for _, tt := range tests { t.Run(tt.name, func(t *testing.T) { - if got := exitCode(tt.res, tt.err); got != tt.want { - t.Fatalf("exitCode(...) = %d, want %d", got, tt.want) - } + require.Equal(t, tt.want, exitCode(tt.res, tt.err)) }) } } @@ -37,9 +35,7 @@ func TestExitCode(t *testing.T) { func writeTempAlerts(t *testing.T) string { t.Helper() path := t.TempDir() + "/alerts.txt" - if err := os.WriteFile(path, []byte("Some Alert\n"), 0o644); err != nil { - t.Fatal(err) - } + require.NoError(t, os.WriteFile(path, []byte("Some Alert\n"), 0o644)) return path } @@ -100,12 +96,8 @@ func TestRunCheck_FlagValidation(t *testing.T) { var stdout, stderr bytes.Buffer args := append([]string{"check"}, tt.args(t)...) code := run(args, &stdout, &stderr) - if code != 2 { - t.Fatalf("code = %d, want 2; stderr = %q", code, stderr.String()) - } - if !strings.Contains(stderr.String(), tt.wantErr) { - t.Fatalf("stderr = %q, want it to contain %q", stderr.String(), tt.wantErr) - } + require.Equal(t, 2, code) + require.Contains(t, stderr.String(), tt.wantErr) }) } } @@ -121,12 +113,8 @@ func TestRunCheck_ToInPastNoLog(t *testing.T) { "--from", "1999-01-01T00:00:00Z", "--to", "2000-01-01T00:00:00Z", "--alerts", writeTempAlerts(t), }, &stdout, &stderr) - if code != 2 { - t.Fatalf("code = %d, want 2; stderr = %q", code, stderr.String()) - } - if !strings.Contains(stderr.String(), "already passed") { - t.Fatalf("stderr = %q, want the past-`to` refusal", stderr.String()) - } + require.Equal(t, 2, code) + require.Contains(t, stderr.String(), "already passed") } // --output json never writes to stdout when Check was never reached, because @@ -137,10 +125,6 @@ func TestRunCheck_NoResultOnConfigError(t *testing.T) { t.Setenv("GRAFANA_TOKEN", "") var stdout, stderr bytes.Buffer code := run([]string{"check", "--to", "2026-01-01T00:00:00Z", "--output", "json"}, &stdout, &stderr) - if code != 2 { - t.Fatalf("code = %d, want 2", code) - } - if stdout.Len() != 0 { - t.Fatalf("stdout = %q, want empty", stdout.String()) - } + require.Equal(t, 2, code) + require.Empty(t, stdout.String()) } diff --git a/grafana-alertcheck/cmd/grafana-alertcheck/list_test.go b/grafana-alertcheck/cmd/grafana-alertcheck/list_test.go index 119326c2c..62aa200d2 100644 --- a/grafana-alertcheck/cmd/grafana-alertcheck/list_test.go +++ b/grafana-alertcheck/cmd/grafana-alertcheck/list_test.go @@ -5,8 +5,9 @@ import ( "fmt" "net/http" "net/http/httptest" - "strings" "testing" + + "github.com/stretchr/testify/require" ) const rulerBody = `{ @@ -60,19 +61,11 @@ func TestRunList_HappyPath(t *testing.T) { var stdout, stderr bytes.Buffer code := run([]string{"list"}, &stdout, &stderr) - if code != 0 { - t.Fatalf("code = %d, want 0; stderr = %q", code, stderr.String()) - } + require.Equal(t, 0, code) out := stdout.String() - if !strings.Contains(out, "rule0000006a") { - t.Errorf("stdout = %q, want it to list rule0000006a", out) - } - if !strings.Contains(out, "Example No Gateways Available") { - t.Errorf("stdout = %q, want it to list the rule title", out) - } - if !strings.Contains(out, "grafana-managed") { - t.Errorf("stdout = %q, want it to name the rule kind", out) - } + require.Contains(t, out, "rule0000006a") + require.Contains(t, out, "Example No Gateways Available") + require.Contains(t, out, "grafana-managed") } func TestRunList_UnsupportedVersion(t *testing.T) { @@ -82,12 +75,8 @@ func TestRunList_UnsupportedVersion(t *testing.T) { var stdout, stderr bytes.Buffer code := run([]string{"list"}, &stdout, &stderr) - if code != 2 { - t.Fatalf("code = %d, want 2", code) - } - if !strings.Contains(stderr.String(), "12.5.0") { - t.Fatalf("stderr = %q, want it to name the unsupported version", stderr.String()) - } + require.Equal(t, 2, code) + require.Contains(t, stderr.String(), "12.5.0") } func TestRunList_RejectsArgs(t *testing.T) { @@ -96,7 +85,5 @@ func TestRunList_RejectsArgs(t *testing.T) { var stdout, stderr bytes.Buffer code := run([]string{"list", "extra"}, &stdout, &stderr) - if code != 2 { - t.Fatalf("code = %d, want 2", code) - } + require.Equal(t, 2, code) } diff --git a/grafana-alertcheck/cmd/grafana-alertcheck/main_test.go b/grafana-alertcheck/cmd/grafana-alertcheck/main_test.go index 1d7906674..34dd3b2d2 100644 --- a/grafana-alertcheck/cmd/grafana-alertcheck/main_test.go +++ b/grafana-alertcheck/cmd/grafana-alertcheck/main_test.go @@ -2,19 +2,16 @@ package main import ( "bytes" - "strings" "testing" + + "github.com/stretchr/testify/require" ) func TestRun_NoArgs(t *testing.T) { var stdout, stderr bytes.Buffer code := run(nil, &stdout, &stderr) - if code != 2 { - t.Fatalf("code = %d, want 2", code) - } - if !strings.Contains(stderr.String(), "usage") { - t.Fatalf("stderr = %q, want a usage message", stderr.String()) - } + require.Equal(t, 2, code) + require.Contains(t, stderr.String(), "usage") } func TestRun_Help(t *testing.T) { @@ -22,15 +19,9 @@ func TestRun_Help(t *testing.T) { t.Run(flag, func(t *testing.T) { var stdout, stderr bytes.Buffer code := run([]string{flag}, &stdout, &stderr) - if code != 0 { - t.Fatalf("code = %d, want 0 (requested help is not a could-not-check condition)", code) - } - if !strings.Contains(stdout.String(), "usage") { - t.Fatalf("stdout = %q, want a usage message", stdout.String()) - } - if stderr.String() != "" { - t.Fatalf("stderr = %q, want empty — help goes to stdout", stderr.String()) - } + require.Equal(t, 0, code, "requested help is not a could-not-check condition") + require.Contains(t, stdout.String(), "usage") + require.Empty(t, stderr.String(), "help goes to stdout") }) } } @@ -38,12 +29,8 @@ func TestRun_Help(t *testing.T) { func TestRun_UnknownSubcommand(t *testing.T) { var stdout, stderr bytes.Buffer code := run([]string{"bogus"}, &stdout, &stderr) - if code != 2 { - t.Fatalf("code = %d, want 2", code) - } - if !strings.Contains(stderr.String(), `"bogus"`) { - t.Fatalf("stderr = %q, want it to name the unknown subcommand", stderr.String()) - } + require.Equal(t, 2, code) + require.Contains(t, stderr.String(), `"bogus"`) } func TestRun_List_MissingEnv(t *testing.T) { @@ -51,10 +38,6 @@ func TestRun_List_MissingEnv(t *testing.T) { t.Setenv("GRAFANA_TOKEN", "") var stdout, stderr bytes.Buffer code := run([]string{"list"}, &stdout, &stderr) - if code != 2 { - t.Fatalf("code = %d, want 2", code) - } - if !strings.Contains(stderr.String(), "GRAFANA_URL") { - t.Fatalf("stderr = %q, want it to name the missing env var", stderr.String()) - } + require.Equal(t, 2, code) + require.Contains(t, stderr.String(), "GRAFANA_URL") } diff --git a/grafana-alertcheck/cmd/grafana-alertcheck/table_test.go b/grafana-alertcheck/cmd/grafana-alertcheck/table_test.go index e585e9b61..9da9ea3ba 100644 --- a/grafana-alertcheck/cmd/grafana-alertcheck/table_test.go +++ b/grafana-alertcheck/cmd/grafana-alertcheck/table_test.go @@ -2,11 +2,11 @@ package main import ( "bytes" - "strings" "testing" "time" "github.com/smartcontractkit/chainlink-testing-framework/grafana-alertcheck/internal/gate" + "github.com/stretchr/testify/require" ) // The golden table test: a fixed Result renders a deterministic, ordered rule @@ -46,68 +46,41 @@ func TestRenderTable(t *testing.T) { } var buf bytes.Buffer - if err := renderTable(&buf, res); err != nil { - t.Fatalf("renderTable: %v", err) - } + require.NoError(t, renderTable(&buf, res)) out := buf.String() // Rule table: Ape sorts before Zebra sorts before... Paused is skipped and // carries no coverage entry, so it renders "-" for PROVED. - if !strings.Contains(out, "Ape Alert") || !strings.Contains(out, "unobservable") { - t.Fatalf("out = %q, want Ape's unobservable row", out) - } - if !strings.Contains(out, "heartbeat_gap") || !strings.Contains(out, "largest gap 5m0s") { - t.Fatalf("out = %q, want the coverage reason and largest gap", out) - } - if !strings.Contains(out, "Zebra Alert") || !strings.Contains(out, "clean") { - t.Fatalf("out = %q, want Zebra's clean row", out) - } + require.Contains(t, out, "Ape Alert") + require.Contains(t, out, "unobservable") + require.Contains(t, out, "heartbeat_gap") + require.Contains(t, out, "largest gap 5m0s") + require.Contains(t, out, "Zebra Alert") + require.Contains(t, out, "clean") // The violations section must show up even without --output json, and must // carry the --allow-paused hint text verbatim. - if !strings.Contains(out, "VIOLATIONS") { - t.Fatalf("out = %q, want a VIOLATIONS section", out) - } - if !strings.Contains(out, "--allow-paused") { - t.Fatalf("out = %q, want the --allow-paused hint in the human table", out) - } - if !strings.Contains(out, "STATE") || !strings.Contains(out, "HEALTH") { - t.Fatalf("out = %q, want the violations table to have STATE and HEALTH columns", out) - } - if !strings.Contains(out, string(gate.StateFiring)) || !strings.Contains(out, "error") { - t.Fatalf("out = %q, want Ape's violation State/Health", out) - } + require.Contains(t, out, "VIOLATIONS") + require.Contains(t, out, "--allow-paused") + require.Contains(t, out, "STATE") + require.Contains(t, out, "HEALTH") + require.Contains(t, out, string(gate.StateFiring)) + require.Contains(t, out, "error") // The footer: per-rule thresholds, global thresholds, and skew with its own // bound rather than the fixed hard limit. - if !strings.Contains(out, "Ape Alert: maxGap=1m0s healthGrace=2m0s evalStaleAfter=1m0s") { - t.Fatalf("out = %q, want Ape's per-rule thresholds", out) - } - if !strings.Contains(out, "Zebra Alert: maxGap=1m0s healthGrace=1m0s evalStaleAfter=1m0s") { - t.Fatalf("out = %q, want Zebra's per-rule thresholds", out) - } - if strings.Contains(out, "Paused Alert: maxGap") { - t.Fatalf("out = %q, a skipped rule must not report thresholds it never had", out) - } - if !strings.Contains(out, "global: transitionGrace=5m0s (source: Ape Alert (for=5m)) drainTimeout=2m0s") { - t.Fatalf("out = %q, want the global thresholds line", out) - } - if !strings.Contains(out, "largest measured clock skew: 1.5s (bound ±250ms, hard limit 1m0s)") { - t.Fatalf("out = %q, want the skew and its own bound, not the hard limit misused as one", out) - } - if !strings.Contains(out, "violations: 2") { - t.Fatalf("out = %q, want the violation count", out) - } - if !strings.Contains(out, "13.1.0") { - t.Fatalf("out = %q, want the grafana version", out) - } + require.Contains(t, out, "Ape Alert: maxGap=1m0s healthGrace=2m0s evalStaleAfter=1m0s") + require.Contains(t, out, "Zebra Alert: maxGap=1m0s healthGrace=1m0s evalStaleAfter=1m0s") + require.NotContains(t, out, "Paused Alert: maxGap") + require.Contains(t, out, "global: transitionGrace=5m0s (source: Ape Alert (for=5m)) drainTimeout=2m0s") + require.Contains(t, out, "largest measured clock skew: 1.5s (bound ±250ms, hard limit 1m0s)") + require.Contains(t, out, "violations: 2") + require.Contains(t, out, "13.1.0") } // The "-" case: a rule decide never asked proveCoverage about (paused before // the window opened) has an empty CoverageResult and must not be reported as // either proved or unobservable. func TestProvedLabel_Skipped(t *testing.T) { - if got := provedLabel(gate.CoverageResult{}); got != "-" { - t.Fatalf("provedLabel(zero value) = %q, want \"-\"", got) - } + require.Equal(t, "-", provedLabel(gate.CoverageResult{})) } diff --git a/grafana-alertcheck/cmd/grafana-alertcheck/watch_test.go b/grafana-alertcheck/cmd/grafana-alertcheck/watch_test.go index 6d8c8c1a6..5c8f53c55 100644 --- a/grafana-alertcheck/cmd/grafana-alertcheck/watch_test.go +++ b/grafana-alertcheck/cmd/grafana-alertcheck/watch_test.go @@ -3,8 +3,9 @@ package main import ( "bytes" "os" - "strings" "testing" + + "github.com/stretchr/testify/require" ) // The record step's flag-validation matrix. Every case fails inside @@ -47,12 +48,8 @@ func TestRunWatch_FlagValidation(t *testing.T) { var stdout, stderr bytes.Buffer args := append([]string{"watch"}, tt.args(t)...) code := run(args, &stdout, &stderr) - if code != 2 { - t.Fatalf("code = %d, want 2; stderr = %q", code, stderr.String()) - } - if !strings.Contains(stderr.String(), tt.wantErr) { - t.Fatalf("stderr = %q, want it to contain %q", stderr.String(), tt.wantErr) - } + require.Equal(t, 2, code) + require.Contains(t, stderr.String(), tt.wantErr) }) } } @@ -68,18 +65,10 @@ func TestRunWatch_DaemonChildDispatch(t *testing.T) { // enough to prove dispatch happened without needing a real recording. missing := os.DevNull + ".missing" code := run([]string{"watch", "--daemon-child", "--out", missing, "--ready-fd", "0"}, &stdout, &stderr) - if code != 2 { - t.Fatalf("code = %d, want 2; stderr = %q", code, stderr.String()) - } - if !strings.Contains(stderr.String(), missing) { - t.Fatalf("stderr = %q, want RunDaemonChild's read failure naming %q", stderr.String(), missing) - } - if strings.Contains(watchUsage, "daemon-child") { - t.Fatalf("watchUsage = %q, must never name --daemon-child", watchUsage) - } - if strings.Contains(watchUsage, "ready-fd") { - t.Fatalf("watchUsage = %q, must never name --ready-fd", watchUsage) - } + require.Equal(t, 2, code) + require.Contains(t, stderr.String(), missing) + require.NotContains(t, watchUsage, "daemon-child") + require.NotContains(t, watchUsage, "ready-fd") } func TestRunWatch_DaemonChild_MissingEnv(t *testing.T) { @@ -88,10 +77,6 @@ func TestRunWatch_DaemonChild_MissingEnv(t *testing.T) { var stdout, stderr bytes.Buffer code := run([]string{"watch", "--daemon-child", "--out", "log.jsonl"}, &stdout, &stderr) - if code != 2 { - t.Fatalf("code = %d, want 2; stderr = %q", code, stderr.String()) - } - if !strings.Contains(stderr.String(), "GRAFANA_URL") { - t.Fatalf("stderr = %q, want it to name the missing env var", stderr.String()) - } + require.Equal(t, 2, code) + require.Contains(t, stderr.String(), "GRAFANA_URL") } diff --git a/grafana-alertcheck/go.mod b/grafana-alertcheck/go.mod index b0c8511ce..d5e0c88be 100644 --- a/grafana-alertcheck/go.mod +++ b/grafana-alertcheck/go.mod @@ -1,3 +1,7 @@ module github.com/smartcontractkit/chainlink-testing-framework/grafana-alertcheck go 1.26.6 + +require github.com/stretchr/testify v1.12.1 + +require go.yaml.in/yaml/v3 v3.0.5 // indirect diff --git a/grafana-alertcheck/go.sum b/grafana-alertcheck/go.sum new file mode 100644 index 000000000..c2336837e --- /dev/null +++ b/grafana-alertcheck/go.sum @@ -0,0 +1,4 @@ +github.com/stretchr/testify v1.12.1 h1:EuwCh5fleGS7H32xRwO3wRGT7DxrDhLAT6FF8MpWDWE= +github.com/stretchr/testify v1.12.1/go.mod h1:MDEgiDPPsNp5cuIrHPPCyornHKgEVbtFUmoNlxoYthg= +go.yaml.in/yaml/v3 v3.0.5 h1:N6y/pJk8buWs9NY5ERU2HSMfm+IuD/OtfdAnq6kESPw= +go.yaml.in/yaml/v3 v3.0.5/go.mod h1:HVTZu1O7/Vkt2N+BFy8Zza+lnLsABggaTM2ZpNIGuKg= diff --git a/grafana-alertcheck/internal/gate/check.go b/grafana-alertcheck/internal/gate/check.go index eab5d931b..2e8cd98c9 100644 --- a/grafana-alertcheck/internal/gate/check.go +++ b/grafana-alertcheck/internal/gate/check.go @@ -236,6 +236,19 @@ func check(ctx context.Context, cfg Config, src Source) (Result, error) { return Result{}, fmt.Errorf("log identity: %w", err) } logHasHdr = true + // Fail fast on a statically-knowable bound violation. `from < + // StartedAt` makes coverage unprovable no matter how healthy the polls + // that DO exist look, and StartedAt is immutable (line 1, written + // first), so this cannot disagree with the authoritative header read + // later. proveCoverage's check 2 remains the backstop against the + // authoritative header, so a bad advisory read can only ever fail + // closed, never produce a false pass. This is recorder mode only: the + // single-step branch has no header, and its own `from < startedAt` is + // a warning-and-pass (see below), not an error. + if from.Before(earlyHdr.StartedAt) { + return Result{}, fmt.Errorf("check: `from` %s is before recording started at %s", + from.Format(time.RFC3339), earlyHdr.StartedAt.Format(time.RFC3339)) + } resolved, notes, err = resolveFromLog(allDefs, earlyHdr, cfg) } else { resolved, notes, err = Resolve(allDefs, cfg.namedAlerts(), cfg.Folder) diff --git a/grafana-alertcheck/internal/gate/check_test.go b/grafana-alertcheck/internal/gate/check_test.go index d13914a8e..6029d5bb7 100644 --- a/grafana-alertcheck/internal/gate/check_test.go +++ b/grafana-alertcheck/internal/gate/check_test.go @@ -14,6 +14,8 @@ import ( "syscall" "testing" "time" + + "github.com/stretchr/testify/require" ) // The one rule every test in this file watches, unless it says otherwise: a @@ -228,12 +230,8 @@ func TestCheckValidateRejectsBadConfigurations(t *testing.T) { cfg := base() tc.mutate(&cfg) err := cfg.withDefaults().validate() - if err == nil { - t.Fatalf("validate() = nil, want an error containing %q", tc.wantErr) - } - if !strings.Contains(err.Error(), tc.wantErr) { - t.Fatalf("validate() = %q, want it to contain %q", err, tc.wantErr) - } + require.Errorf(t, err, "validate()") + require.Contains(t, err.Error(), tc.wantErr) }) } } @@ -250,12 +248,8 @@ func TestCheckValidateAcceptsAPastToWithALog(t *testing.T) { Clock: newFakeClock(testNow), }.withDefaults() - if err := cfg.validate(); err != nil { - t.Fatalf("validate() = %v, want nil", err) - } - if cfg.PidFile != "log.jsonl.pid" { - t.Errorf("PidFile = %q, want the .pid default", cfg.PidFile) - } + require.NoError(t, cfg.validate()) + require.Equal(t, "log.jsonl.pid", cfg.PidFile) } // --------------------------------------------------------------------------- @@ -270,35 +264,24 @@ func TestCheckSingleStepCleanWindowPasses(t *testing.T) { }) res, err := check(context.Background(), cfg, src) - if err != nil { - t.Fatalf("check() = %v, want nil\nnotes:\n%s", err, notesOf(cfg)) - } + require.NoError(t, err) // A pass is exactly this shape. - if len(res.Violations) != 0 { - t.Fatalf("Violations = %+v, want none", res.Violations) - } - if len(res.Verdicts) != 1 || res.Verdicts[0].Outcome != OutcomeClean { - t.Fatalf("Verdicts = %+v, want one clean verdict", res.Verdicts) - } - if cov := res.Coverage[checkUID]; !cov.Proved || cov.Unobservable { - t.Fatalf("Coverage = %+v, want proved", cov) - } + require.Empty(t, res.Violations) + require.Len(t, res.Verdicts, 1) + require.Equal(t, OutcomeClean, res.Verdicts[0].Outcome) + cov := res.Coverage[checkUID] + require.True(t, cov.Proved) + require.False(t, cov.Unobservable) // The collection loop ran to to+transitionGrace and no further. windowEnd := cfg.To.Add(checkGrace) - if clock.Now().Before(windowEnd) { - t.Errorf("stopped collecting at %s, before to+grace %s", clock.Now(), windowEnd) - } + require.False(t, clock.Now().Before(windowEnd)) // One measurement-pass poll plus one every 30s across the 6-minute // collection, plus the drain wait's own polls. The exact count depends on // the scheduler's random stagger, so assert the order of magnitude a full // window implies rather than an exact number. - if got := src.callCount(checkTitle); got < 12 { - t.Errorf("polled %d times, want at least the ~13 a full 6-minute window at 30s implies", got) - } - if notes := notesOf(cfg); !strings.Contains(notes, "planned run time") { - t.Errorf("the planned run time must be printed at start; notes were:\n%s", notes) - } + require.GreaterOrEqual(t, src.callCount(checkTitle), 12) + require.Contains(t, notesOf(cfg), "planned run time") } // resolve_test.go proves the collapse-note-plus-satisfied-MinObserved path at @@ -315,18 +298,10 @@ func TestCheckSingleStepDuplicateAlertNamesCollapseWithNote(t *testing.T) { }) res, err := check(context.Background(), cfg, src) - if err != nil { - t.Fatalf("check() = %v, want nil\nnotes:\n%s", err, notesOf(cfg)) - } - if len(res.Verdicts) != 1 { - t.Fatalf("Verdicts = %+v, want exactly one — the duplicate must collapse to a single rule", res.Verdicts) - } - if len(res.Violations) != 0 { - t.Fatalf("Violations = %+v, want none: MinObserved must be satisfied by the post-collapse count of 1", res.Violations) - } - if notes := notesOf(cfg); !strings.Contains(notes, "counted once") { - t.Errorf("want the collapse note in the run's own notes; got:\n%s", notes) - } + require.NoError(t, err) + require.Len(t, res.Verdicts, 1, "the duplicate must collapse to a single rule") + require.Empty(t, res.Violations, "MinObserved must be satisfied by the post-collapse count of 1") + require.Contains(t, notesOf(cfg), "counted once") } // A rule with health=error for the whole window is unobservable, exit 2 — @@ -337,9 +312,7 @@ func TestCheckSingleStepDuplicateAlertNamesCollapseWithNote(t *testing.T) { func TestCheckSingleStepContinuousHealthErrorIsUnobservable(t *testing.T) { body := readFixture(t, "state_health_error.json") rules, err := ParseState(body) - if err != nil { - t.Fatalf("ParseState: %v", err) - } + require.NoError(t, err) base := rules[0] def := Definition{ UID: base.UID, Title: base.Title, Folder: base.Folder, Group: base.Group, @@ -365,15 +338,10 @@ func TestCheckSingleStepContinuousHealthErrorIsUnobservable(t *testing.T) { src.defs = []Definition{def} res, err := check(context.Background(), cfg, src) - if err == nil { - t.Fatalf("check() = nil, want an error: continuous health=error must be unobservable\nnotes:\n%s", notesOf(cfg)) - } - if len(res.Verdicts) != 1 || res.Verdicts[0].Outcome != OutcomeUnobservable { - t.Fatalf("Verdicts = %+v, want one unobservable verdict", res.Verdicts) - } - if cov := res.Coverage[def.UID]; cov.Reason != ReasonHealthError { - t.Fatalf("Coverage[%s].Reason = %q, want %q", def.UID, cov.Reason, ReasonHealthError) - } + require.Error(t, err, "continuous health=error must be unobservable") + require.Len(t, res.Verdicts, 1) + require.Equal(t, OutcomeUnobservable, res.Verdicts[0].Outcome) + require.Equal(t, ReasonHealthError, res.Coverage[def.UID].Reason) } // A certain violation does not release the runner early, and it does not stop @@ -391,18 +359,10 @@ func TestCheckSingleStepFiringInstanceReportsWithoutExitingEarly(t *testing.T) { }) res, err := check(context.Background(), cfg, src) - if err != nil { - t.Fatalf("check() = %v, want nil (a violation is exit 1, not an error)", err) - } - if len(res.Violations) != 1 { - t.Fatalf("Violations = %+v, want exactly one", res.Violations) - } - if got := res.Violations[0].Outcome; got != OutcomePersistentlyBad { - t.Errorf("Outcome = %q, want %q", got, OutcomePersistentlyBad) - } - if windowEnd := cfg.To.Add(checkGrace); clock.Now().Before(windowEnd) { - t.Errorf("exited early at %s; collection must run to %s", clock.Now(), windowEnd) - } + require.NoError(t, err, "a violation is exit 1, not an error") + require.Len(t, res.Violations, 1) + require.Equal(t, OutcomePersistentlyBad, res.Violations[0].Outcome) + require.False(t, clock.Now().Before(cfg.To.Add(checkGrace)), "exited early; collection must run to to+grace") } // A newly_bad instance at from+30s gives exit 1, but ONLY after @@ -427,15 +387,11 @@ func TestCheckSingleStepNewOnsetDoesNotExitEarly(t *testing.T) { }) res, err := check(context.Background(), cfg, src) - if err != nil { - t.Fatalf("check() = %v, want nil (a violation is exit 1, not an error)", err) - } - if len(res.Violations) != 1 || res.Violations[0].Outcome != OutcomeNewlyBad { - t.Fatalf("Violations = %+v, want exactly one newly_bad", res.Violations) - } - if windowEnd := cfg.To.Add(checkGrace); clock.Now().Before(windowEnd) { - t.Errorf("exited early at %s; collection must run to %s even for a fresh onset at from+30s", clock.Now(), windowEnd) - } + require.NoError(t, err) + require.Len(t, res.Violations, 1) + require.Equal(t, OutcomeNewlyBad, res.Violations[0].Outcome) + require.False(t, clock.Now().Before(cfg.To.Add(checkGrace)), + "exited early; collection must run to to+grace even for a fresh onset at from+30s") } // An ABSENT `from` in single-step mode (as opposed to recorder mode, which @@ -451,16 +407,9 @@ func TestCheckSingleStepAbsentFromFallsBackToStepStart(t *testing.T) { }) res, err := check(context.Background(), cfg, src) - if err != nil { - t.Fatalf("check() = %v, want nil: an absent `from` in single-step mode is a fallback, not an error\nnotes:\n%s", err, notesOf(cfg)) - } - notes := notesOf(cfg) - if !strings.Contains(notes, "no `from` given") { - t.Errorf("want the step-start fallback note; notes were:\n%s", notes) - } - if !res.From.Equal(testNow) { - t.Errorf("Result.From = %s, want the step-start fallback %s", res.From, testNow) - } + require.NoError(t, err, "an absent `from` in single-step mode is a fallback, not an error") + require.Contains(t, notesOf(cfg), "no `from` given") + require.True(t, res.From.Equal(testNow)) } // In single-step mode an explicit `from` earlier than the first observation is @@ -475,18 +424,13 @@ func TestCheckSingleStepFromBeforeFirstObservationWarnsAndPasses(t *testing.T) { }) res, err := check(context.Background(), cfg, src) - if err != nil { - t.Fatalf("check() = %v, want a pass with a warning\nnotes:\n%s", err, notesOf(cfg)) - } + require.NoError(t, err) notes := notesOf(cfg) - if !strings.Contains(notes, "cannot see [") || !strings.Contains(notes, testNow.Format(time.RFC3339)) { - t.Errorf("want a warning naming the unseen interval; notes were:\n%s", notes) - } + require.Contains(t, notes, "cannot see [") + require.Contains(t, notes, testNow.Format(time.RFC3339)) // The classified window is the clamped one, and Result says so rather than // reporting a window the run never proved. - if !res.From.Equal(testNow) { - t.Errorf("Result.From = %s, want the clamped %s", res.From, testNow) - } + require.True(t, res.From.Equal(testNow)) } // The failure limit was exceeded. The measurement pass succeeds and the @@ -503,15 +447,9 @@ func TestCheckFailClosedOnExhaustedRetries(t *testing.T) { }) res, err := check(context.Background(), cfg, src) - if err == nil { - t.Fatalf("check() = nil, want the collection failure to fail closed") - } - if !strings.Contains(err.Error(), "collect evidence") { - t.Errorf("err = %q, want it to name the collection step", err) - } - if len(res.Violations) != 0 { - t.Errorf("Violations = %+v; an error must never be reported as a verdict", res.Violations) - } + require.Error(t, err, "the collection failure to fail closed") + require.Contains(t, err.Error(), "collect evidence") + require.Empty(t, res.Violations, "an error must never be reported as a verdict") } // The resolution of the definitions failed. Both shapes — the ruler read @@ -523,10 +461,9 @@ func TestCheckFailClosedOnDefinitionResolution(t *testing.T) { src := newCheckSource(nil) src.defsErr = errors.New("502 bad gateway") - if _, err := check(context.Background(), cfg, src); err == nil || - !strings.Contains(err.Error(), "read rule definitions") { - t.Fatalf("check() = %v, want a definitions-read failure", err) - } + _, err := check(context.Background(), cfg, src) + require.Error(t, err) + require.Contains(t, err.Error(), "read rule definitions") }) t.Run("unknown alert name", func(t *testing.T) { @@ -535,10 +472,9 @@ func TestCheckFailClosedOnDefinitionResolution(t *testing.T) { cfg.Alerts = []string{"No Such Rule"} src := newCheckSource(nil) - if _, err := check(context.Background(), cfg, src); err == nil || - !strings.Contains(err.Error(), "no rule matched") { - t.Fatalf("check() = %v, want a no-match failure", err) - } + _, err := check(context.Background(), cfg, src) + require.Error(t, err) + require.Contains(t, err.Error(), "no rule matched") }) } @@ -550,10 +486,9 @@ func TestCheckRefusesUnsupportedGrafanaVersion(t *testing.T) { src := newCheckSource(nil) src.version = "12.4.0" - if _, err := check(context.Background(), cfg, src); err == nil || - !strings.Contains(err.Error(), "unsupported grafana version") { - t.Fatalf("check() = %v, want the version gate to refuse 12.4.0", err) - } + _, err := check(context.Background(), cfg, src) + require.Error(t, err) + require.Contains(t, err.Error(), "unsupported grafana version") } // The budget is checked against the latencies the measurement pass actually @@ -569,13 +504,9 @@ func TestCheckSingleStepRefusesAScheduleThatDoesNotFit(t *testing.T) { }) _, err := check(context.Background(), cfg, src) - if err == nil { - t.Fatalf("check() = nil, want the budget check to refuse the schedule") - } + require.Error(t, err, "the budget check to refuse the schedule") for _, want := range []string{"raising concurrency", "raising poll-interval", "watching fewer alerts"} { - if !strings.Contains(err.Error(), want) { - t.Errorf("err = %q, want it to name the control %q", err, want) - } + require.Contains(t, err.Error(), want) } } @@ -592,9 +523,7 @@ func recordedLog(t *testing.T, dir string, url string, startedAt, start, end, se path := filepath.Join(dir, "log.jsonl") clock := newFakeClock(sentinelAt) w, err := NewWriter(path, clock) - if err != nil { - t.Fatalf("NewWriter: %v", err) - } + require.NoError(t, err) header := Header{ URL: url, GrafanaVersion: "13.1.0", @@ -605,20 +534,14 @@ func recordedLog(t *testing.T, dir string, url string, startedAt, start, end, se PollEverySeconds: checkPollEvery.Seconds(), }}, } - if err := w.WriteHeader(header); err != nil { - t.Fatalf("WriteHeader: %v", err) - } + require.NoError(t, w.WriteHeader(header)) for at := start; !at.After(end); at = at.Add(checkPollEvery) { - if err := w.WritePoll(Poll{ + require.NoError(t, w.WritePoll(Poll{ RuleUID: checkUID, GrafanaNow: at, Found: true, State: "inactive", Health: "ok", LastEvaluation: at.Add(-lastEvalLag), - }); err != nil { - t.Fatalf("WritePoll: %v", err) - } - } - if err := w.Stop(); err != nil { - t.Fatalf("Stop: %v", err) + })) } + require.NoError(t, w.Stop()) return path } @@ -628,21 +551,15 @@ func recordedLog(t *testing.T, dir string, url string, startedAt, start, end, se func deadPid(t *testing.T) int { t.Helper() cmd := exec.Command("/bin/sh", "-c", "exit 0") - if err := cmd.Start(); err != nil { - t.Fatalf("start a throwaway process: %v", err) - } + require.NoError(t, cmd.Start()) pid := cmd.Process.Pid - if err := cmd.Wait(); err != nil { - t.Fatalf("wait for the throwaway process: %v", err) - } + require.NoError(t, cmd.Wait()) return pid } func writePid(t *testing.T, path, contents string) { t.Helper() - if err := os.WriteFile(path, []byte(contents), 0o644); err != nil { - t.Fatalf("write pidfile: %v", err) - } + require.NoError(t, os.WriteFile(path, []byte(contents), 0o644)) } // recorderConfig points check at a recording of [testNow-1m, windowEnd+30s] @@ -671,28 +588,19 @@ func TestCheckRecorderModeCleanWindowPasses(t *testing.T) { // The drain wait is satisfied from the log's own evidence, so the source // must never be asked for a state — asserted by the nil responder. src := newCheckSource(func(title string, _ int) (Observation, error) { - t.Errorf("the drain wait polled %q although the log already proves the evaluations", title) + require.Fail(t, fmt.Sprintf("the drain wait polled %q although the log already proves the evaluations", title)) return Observation{}, errors.New("unexpected poll") }) res, err := check(context.Background(), cfg, src) - if err != nil { - t.Fatalf("check() = %v, want nil\nnotes:\n%s", err, notesOf(cfg)) - } - if len(res.Violations) != 0 { - t.Fatalf("Violations = %+v, want none", res.Violations) - } - if len(res.Verdicts) != 1 || res.Verdicts[0].Outcome != OutcomeClean { - t.Fatalf("Verdicts = %+v, want one clean verdict", res.Verdicts) - } - if res.GrafanaVersion != "13.1.0" { - t.Errorf("GrafanaVersion = %q, want the recorded one", res.GrafanaVersion) - } + require.NoError(t, err) + require.Empty(t, res.Violations) + require.Len(t, res.Verdicts, 1) + require.Equal(t, OutcomeClean, res.Verdicts[0].Outcome) + require.Equal(t, "13.1.0", res.GrafanaVersion) // The collection loop still waited out to+transitionGrace even though the // recorder had already finished. - if clock.Now().Before(windowEnd) { - t.Errorf("returned at %s, before to+grace %s", clock.Now(), windowEnd) - } + require.False(t, clock.Now().Before(windowEnd)) } // The identity of the log is not correct. The check runs against the header @@ -707,12 +615,9 @@ func TestCheckFailClosedOnWrongLogIdentity(t *testing.T) { clock := newVirtualClock(testNow) cfg := recorderConfig(t, clock, logPath) _, err := check(context.Background(), cfg, newCheckSource(nil)) - if err == nil || !strings.Contains(err.Error(), "log identity") { - t.Fatalf("check() = %v, want a log-identity failure", err) - } - if !clock.Now().Equal(testNow) { - t.Errorf("the identity check waited out the window (now %s); it must fail before the wait", clock.Now()) - } + require.Error(t, err) + require.Contains(t, err.Error(), "log identity") + require.True(t, clock.Now().Equal(testNow), "it must fail before the wait") }) t.Run("rule no longer resolves", func(t *testing.T) { @@ -726,12 +631,32 @@ func TestCheckFailClosedOnWrongLogIdentity(t *testing.T) { src.defs = []Definition{{UID: "somebody-else", Title: "Other", Kind: KindGrafanaManaged, IntervalSeconds: 60}} _, err := check(context.Background(), cfg, src) - if err == nil || !strings.Contains(err.Error(), "log identity") { - t.Fatalf("check() = %v, want a log-identity failure", err) - } + require.Error(t, err) + require.Contains(t, err.Error(), "log identity") }) } +// `from` before the recording's StartedAt is statically knowable from the +// header (immutable line 1), so check fails closed on it BEFORE the window's +// wait — exactly like the identity check above — rather than surfacing a +// from_before_record verdict only after the drain. +func TestCheckFailFastWhenFromPrecedesRecordStart(t *testing.T) { + dir := t.TempDir() + startedAt := testNow.Add(time.Minute) // the recording opened a minute AFTER `from` + windowEnd := testNow.Add(5*time.Minute + checkGrace) + // The poll range is irrelevant to the assertion: the fail-fast reads + // StartedAt from the header alone, before any polling would matter. + logPath := recordedLog(t, dir, "https://grafana.example.com", + startedAt, startedAt, windowEnd.Add(30*time.Second), windowEnd.Add(30*time.Second), 0) + + clock := newVirtualClock(testNow) + cfg := recorderConfig(t, clock, logPath) // From = testNow, before StartedAt + _, err := check(context.Background(), cfg, newCheckSource(nil)) + require.Error(t, err) + require.Contains(t, err.Error(), "before recording started") + require.True(t, clock.Now().Equal(testNow), "it must fail before the wait") +} + // The coverage proof failed: a hole in the middle of the recording is not // saved by healthy data at both ends. func TestCheckFailClosedOnCoverageGap(t *testing.T) { @@ -740,42 +665,28 @@ func TestCheckFailClosedOnCoverageGap(t *testing.T) { path := filepath.Join(dir, "log.jsonl") clock := newFakeClock(windowEnd.Add(30 * time.Second)) w, err := NewWriter(path, clock) - if err != nil { - t.Fatalf("NewWriter: %v", err) - } - if err := w.WriteHeader(Header{ + require.NoError(t, err) + require.NoError(t, w.WriteHeader(Header{ URL: "https://grafana.example.com", GrafanaVersion: "13.1.0", StartedAt: testNow.Add(-time.Minute), Rules: []LoggedRule{{UID: checkUID, Title: checkTitle, IntervalSeconds: 60, PollEverySeconds: checkPollEvery.Seconds()}}, - }); err != nil { - t.Fatalf("WriteHeader: %v", err) - } + })) for at := testNow.Add(-time.Minute); !at.After(windowEnd.Add(30 * time.Second)); at = at.Add(checkPollEvery) { // A three-minute hole in the middle of the window. if at.After(testNow.Add(time.Minute)) && at.Before(testNow.Add(4*time.Minute)) { continue } - if err := w.WritePoll(Poll{ + require.NoError(t, w.WritePoll(Poll{ RuleUID: checkUID, GrafanaNow: at, Found: true, State: "inactive", Health: "ok", LastEvaluation: at, - }); err != nil { - t.Fatalf("WritePoll: %v", err) - } - } - if err := w.Stop(); err != nil { - t.Fatalf("Stop: %v", err) + })) } + require.NoError(t, w.Stop()) writePid(t, path+".pid", fmt.Sprintf("%d\n", deadPid(t))) cfg := recorderConfig(t, newVirtualClock(testNow), path) res, err := check(context.Background(), cfg, newCheckSource(nil)) - if err == nil { - t.Fatalf("check() = nil, want the coverage gap to fail closed") - } - if got := res.Coverage[checkUID].Reason; got != ReasonHeartbeatGap { - t.Errorf("Reason = %q, want %q", got, ReasonHeartbeatGap) - } - if got := res.Verdicts[0].Outcome; got != OutcomeUnobservable { - t.Errorf("Outcome = %q, want %q", got, OutcomeUnobservable) - } + require.Error(t, err, "the coverage gap to fail closed") + require.Equal(t, ReasonHeartbeatGap, res.Coverage[checkUID].Reason) + require.Equal(t, OutcomeUnobservable, res.Verdicts[0].Outcome) } // An episode fully between the deploy and the start of the check: recorder @@ -790,36 +701,25 @@ func TestCheckRecorderModeFindsAGapImmediatelyAfterTheDeploy(t *testing.T) { path := filepath.Join(dir, "log.jsonl") clock := newFakeClock(windowEnd.Add(30 * time.Second)) w, err := NewWriter(path, clock) - if err != nil { - t.Fatalf("NewWriter: %v", err) - } - if err := w.WriteHeader(Header{ + require.NoError(t, err) + require.NoError(t, w.WriteHeader(Header{ URL: "https://grafana.example.com", GrafanaVersion: "13.1.0", StartedAt: testNow.Add(-time.Minute), Rules: []LoggedRule{{UID: checkUID, Title: checkTitle, IntervalSeconds: 60, PollEverySeconds: checkPollEvery.Seconds()}}, - }); err != nil { - t.Fatalf("WriteHeader: %v", err) - } + })) gapEnd := testNow.Add(3 * time.Minute) // nothing recorded from `from` (testNow) to here for at := gapEnd; !at.After(windowEnd.Add(30 * time.Second)); at = at.Add(checkPollEvery) { - if err := w.WritePoll(Poll{ + require.NoError(t, w.WritePoll(Poll{ RuleUID: checkUID, GrafanaNow: at, Found: true, State: "inactive", Health: "ok", LastEvaluation: at, - }); err != nil { - t.Fatalf("WritePoll: %v", err) - } - } - if err := w.Stop(); err != nil { - t.Fatalf("Stop: %v", err) + })) } + require.NoError(t, w.Stop()) writePid(t, path+".pid", fmt.Sprintf("%d\n", deadPid(t))) cfg := recorderConfig(t, newVirtualClock(testNow), path) res, err := check(context.Background(), cfg, newCheckSource(nil)) - if err == nil { - t.Fatalf("check() = nil, want exit 2: a hole right after the deploy hides whatever happened there just as much as one in the middle") - } - if len(res.Verdicts) != 1 || res.Verdicts[0].Outcome != OutcomeUnobservable { - t.Fatalf("Verdicts = %+v, want one unobservable verdict, never clean", res.Verdicts) - } + require.Error(t, err, "a hole right after the deploy hides whatever happened there") + require.Len(t, res.Verdicts, 1) + require.Equal(t, OutcomeUnobservable, res.Verdicts[0].Outcome, "never clean") } // The drain limit passed. The recording itself is clean, so this isolates the @@ -846,21 +746,12 @@ func TestCheckFailClosedOnDrainTimeout(t *testing.T) { }) res, err := check(context.Background(), cfg, src) - if err == nil { - t.Fatalf("check() = nil, want the drain limit to fail closed") - } - if got := res.Coverage[checkUID].Reason; got != ReasonDrainTimeout { - t.Errorf("Reason = %q, want %q", got, ReasonDrainTimeout) - } - if got := res.Verdicts[0].Outcome; got != OutcomeUnobservable { - t.Errorf("Outcome = %q, want %q", got, OutcomeUnobservable) - } - if !strings.Contains(res.Verdicts[0].Note, "drain limit") { - t.Errorf("Note = %q, want it to explain the drain limit", res.Verdicts[0].Note) - } - if waited := clock.Now().Sub(windowEnd); waited < checkDrainLimit { - t.Errorf("gave up after %s of drain wait, want the full %s", waited, checkDrainLimit) - } + require.Error(t, err, "the drain limit to fail closed") + require.Equal(t, ReasonDrainTimeout, res.Coverage[checkUID].Reason) + require.Equal(t, OutcomeUnobservable, res.Verdicts[0].Outcome) + require.Contains(t, res.Verdicts[0].Note, "drain limit") + require.GreaterOrEqual(t, clock.Now().Sub(windowEnd), checkDrainLimit, + "the rule never evaluates through the window, so the drain wait must run its full limit") } // A rule the state endpoint no longer serves is knowable on the FIRST drain @@ -884,18 +775,10 @@ func TestCheckDrainWaitNamesADeletedRuleAtOnce(t *testing.T) { }) res, err := check(context.Background(), cfg, src) - if err == nil { - t.Fatalf("check() = nil, want a deleted rule to fail closed") - } - if got := res.Coverage[checkUID].Reason; got != ReasonRuleAbsent { - t.Errorf("Reason = %q, want %q — the fault, not the wait", got, ReasonRuleAbsent) - } - if got := src.callCount(checkTitle); got != 1 { - t.Errorf("polled %d times, want exactly 1: the absence is knowable on the first poll", got) - } - if waited := clock.Now().Sub(windowEnd); waited >= checkDrainLimit { - t.Errorf("spent %s in the drain wait, want it to conclude at once", waited) - } + require.Error(t, err, "a deleted rule to fail closed") + require.Equal(t, ReasonRuleAbsent, res.Coverage[checkUID].Reason, "the fault, not the wait") + require.Equal(t, 1, src.callCount(checkTitle), "the absence is knowable on the first poll") + require.Less(t, clock.Now().Sub(windowEnd), checkDrainLimit) } // --------------------------------------------------------------------------- @@ -909,18 +792,14 @@ func pausedAfterWindowLog(t *testing.T, dir string, firesAt time.Time, end, sent t.Helper() path := filepath.Join(dir, "log.jsonl") w, err := NewWriter(path, newFakeClock(sentinelAt)) - if err != nil { - t.Fatalf("NewWriter: %v", err) - } - if err := w.WriteHeader(Header{ + require.NoError(t, err) + require.NoError(t, w.WriteHeader(Header{ URL: "https://grafana.example.com", GrafanaVersion: "13.1.0", StartedAt: testNow.Add(-time.Minute), Rules: []LoggedRule{{ UID: checkUID, Title: checkTitle, IntervalSeconds: 60, IsPaused: false, PollEverySeconds: checkPollEvery.Seconds(), }}, - }); err != nil { - t.Fatalf("WriteHeader: %v", err) - } + })) firing := Instance{ Labels: map[string]string{"alertname": checkTitle, "instance": "a"}, State: StateFiring, @@ -932,13 +811,9 @@ func pausedAfterWindowLog(t *testing.T, dir string, firesAt time.Time, end, sent p.State = "firing" p.Abnormal = []Instance{firing} } - if err := w.WritePoll(p); err != nil { - t.Fatalf("WritePoll: %v", err) - } - } - if err := w.Stop(); err != nil { - t.Fatalf("Stop: %v", err) + require.NoError(t, w.WritePoll(p)) } + require.NoError(t, w.Stop()) return path } @@ -970,19 +845,13 @@ func pausedAfterWindowCheck(t *testing.T, allowPaused bool) (Result, error, Conf // that was active at record start is classified, whatever its pause state is // by the time check resolves the definitions. func TestCheckPausingARuleAfterTheWindowDoesNotMakeItSkipped(t *testing.T) { - res, err, cfg := pausedAfterWindowCheck(t, false) - if err != nil { - t.Fatalf("check() = %v, want a classified verdict\nnotes:\n%s", err, notesOf(cfg)) - } - if got := res.Verdicts[0].Outcome; got != OutcomeNewlyBad { - t.Fatalf("Outcome = %q, want %q: the rule was active for the whole window and fired inside it", got, OutcomeNewlyBad) - } - if len(res.Violations) != 1 || res.Violations[0].Outcome != OutcomeNewlyBad { - t.Fatalf("Violations = %+v, want the firing reported", res.Violations) - } - if strings.Contains(res.Verdicts[0].Note, "paused before the window opened") { - t.Errorf("Note = %q, which the log's own polls contradict", res.Verdicts[0].Note) - } + res, err, _ := pausedAfterWindowCheck(t, false) + require.NoError(t, err) + require.Equal(t, OutcomeNewlyBad, res.Verdicts[0].Outcome, + "the rule was active for the whole window and fired inside it") + require.Len(t, res.Violations, 1) + require.Equal(t, OutcomeNewlyBad, res.Violations[0].Outcome) + require.NotContains(t, res.Verdicts[0].Note, "paused before the window opened") } // The regression pin for the loophole this fix closed. Reading skipped from @@ -990,13 +859,9 @@ func TestCheckPausingARuleAfterTheWindowDoesNotMakeItSkipped(t *testing.T) { // skipped free; and a window in which the alert fired reported exit 0. The // default message names --allow-paused, so an operator was led straight to it. func TestCheckAllowPausedCannotExcuseARulePausedAfterItFired(t *testing.T) { - res, err, cfg := pausedAfterWindowCheck(t, true) - if err != nil { - t.Fatalf("check() = %v, want a classified verdict\nnotes:\n%s", err, notesOf(cfg)) - } - if len(res.Violations) == 0 { - t.Fatalf("Violations = none with --allow-paused: the run passed over a window in which the alert fired") - } + res, err, _ := pausedAfterWindowCheck(t, true) + require.NoError(t, err) + require.NotEmpty(t, res.Violations, "the run passed over a window in which the alert fired") } // The other direction, unchanged: a rule the HEADER says was paused when the @@ -1007,23 +872,17 @@ func TestCheckHeaderPausedRuleStaysSkipped(t *testing.T) { windowEnd := testNow.Add(5*time.Minute + checkGrace) path := filepath.Join(dir, "log.jsonl") w, err := NewWriter(path, newFakeClock(windowEnd.Add(30*time.Second))) - if err != nil { - t.Fatalf("NewWriter: %v", err) - } + require.NoError(t, err) // Named in the header, is_paused true, and no poll records at all — the // shape watch writes for a rule paused before the window opened. - if err := w.WriteHeader(Header{ + require.NoError(t, w.WriteHeader(Header{ URL: "https://grafana.example.com", GrafanaVersion: "13.1.0", StartedAt: testNow.Add(-time.Minute), Rules: []LoggedRule{{ UID: checkUID, Title: checkTitle, IntervalSeconds: 60, IsPaused: true, PollEverySeconds: checkPollEvery.Seconds(), }}, - }); err != nil { - t.Fatalf("WriteHeader: %v", err) - } - if err := w.Stop(); err != nil { - t.Fatalf("Stop: %v", err) - } + })) + require.NoError(t, w.Stop()) writePid(t, path+".pid", fmt.Sprintf("%d\n", deadPid(t))) run := func(allowPaused bool) (Result, error) { @@ -1031,30 +890,22 @@ func TestCheckHeaderPausedRuleStaysSkipped(t *testing.T) { cfg.AllowPaused = allowPaused // The definition is unpaused now; the header still decides. src := newCheckSource(func(title string, _ int) (Observation, error) { - t.Errorf("the drain wait polled skipped rule %q", title) + require.Fail(t, fmt.Sprintf("the drain wait polled skipped rule %q", title)) return Observation{}, errors.New("unexpected poll") }) return check(context.Background(), cfg, src) } res, err := run(false) - if err != nil { - t.Fatalf("check() = %v, want exit-1 shape: a skipped rule is a known condition, not an inability", err) - } - if got := res.Verdicts[0].Outcome; got != OutcomeSkipped { - t.Fatalf("Outcome = %q, want %q", got, OutcomeSkipped) - } - if _, ok := res.Coverage[checkUID]; ok { - t.Errorf("Coverage[%s] present, want absent: a skipped rule has no coverage to prove", checkUID) - } - if len(res.Violations) != 1 { - t.Errorf("Violations = %+v, want the MinObserved shortfall", res.Violations) - } + require.NoError(t, err, "a skipped rule is a known condition, not an inability") + require.Equal(t, OutcomeSkipped, res.Verdicts[0].Outcome) + _, ok := res.Coverage[checkUID] + require.False(t, ok, "a skipped rule has no coverage to prove") + require.Len(t, res.Violations, 1, "the MinObserved shortfall") res, err = run(true) - if err != nil || len(res.Violations) != 0 { - t.Errorf("with --allow-paused: err = %v, Violations = %+v, want a pass", err, res.Violations) - } + require.NoError(t, err) + require.Empty(t, res.Violations, "with --allow-paused: want a pass") } // A paused rule does not evaluate, so it can never catch up: the drain wait @@ -1076,21 +927,12 @@ func TestCheckDrainWaitConcludesAtOnceOnAPausedRule(t *testing.T) { }) res, err := check(context.Background(), cfg, src) - if err == nil { - t.Fatalf("check() = nil, want a rule that stopped evaluating to fail closed") - } - if got := src.callCount(checkTitle); got != 1 { - t.Errorf("polled %d times, want exactly 1: a paused rule can never catch up", got) - } - if got := res.Coverage[checkUID].Reason; got != ReasonDrainTimeout { - t.Errorf("Reason = %q, want %q — the vocabulary is published, so the detail goes in the note", got, ReasonDrainTimeout) - } - if !strings.Contains(res.Verdicts[0].Note, "paused before it evaluated through") { - t.Errorf("Note = %q, want it to say the rule was paused", res.Verdicts[0].Note) - } - if waited := clock.Now().Sub(windowEnd); waited >= checkDrainLimit { - t.Errorf("spent %s in the drain wait, want it to conclude at once", waited) - } + require.Error(t, err, "a rule that stopped evaluating must fail closed") + require.Equal(t, 1, src.callCount(checkTitle), "a paused rule can never catch up") + require.Equal(t, ReasonDrainTimeout, res.Coverage[checkUID].Reason, + "the vocabulary is published, so the detail goes in the note") + require.Contains(t, res.Verdicts[0].Note, "paused before it evaluated through") + require.Less(t, clock.Now().Sub(windowEnd), checkDrainLimit) } // An absent or unparseable pidfile is never "there was nothing to stop". The @@ -1120,9 +962,8 @@ func TestCheckRefusesToReadALogItCannotStop(t *testing.T) { cfg := recorderConfig(t, newVirtualClock(testNow), logPath) _, err := check(context.Background(), cfg, newCheckSource(nil)) - if err == nil || !strings.Contains(err.Error(), "cannot stop the recorder") { - t.Fatalf("check() = %v, want a refusal to stop the recorder", err) - } + require.Error(t, err) + require.Contains(t, err.Error(), "cannot stop the recorder") }) } } @@ -1137,19 +978,14 @@ func startLockHolder(t *testing.T, logPath string) int { cmd.Env = append(os.Environ(), lockHolderEnv+"="+logPath) cmd.Stderr = os.Stderr stdout, err := cmd.StdoutPipe() - if err != nil { - t.Fatalf("pipe: %v", err) - } - if err := cmd.Start(); err != nil { - t.Fatalf("start the lock holder: %v", err) - } + require.NoError(t, err) + require.NoError(t, cmd.Start()) t.Cleanup(func() { _ = cmd.Process.Kill() _ = cmd.Wait() }) - if _, err := bufio.NewReader(stdout).ReadString('\n'); err != nil { - t.Fatalf("the lock holder never reported holding the lock: %v", err) - } + _, err = bufio.NewReader(stdout).ReadString('\n') + require.NoError(t, err, "the lock holder never reported holding the lock") return cmd.Process.Pid } @@ -1165,9 +1001,8 @@ func TestCheckFailsWhenTheRecorderWillNotExit(t *testing.T) { cfg := recorderConfig(t, newVirtualClock(testNow), logPath) _, err := check(context.Background(), cfg, newCheckSource(nil)) - if err == nil || !strings.Contains(err.Error(), "still holds") { - t.Fatalf("check() = %v, want the stop wait to time out on the lock", err) - } + require.Error(t, err) + require.Contains(t, err.Error(), "still holds") } // The regression pin for a stray SIGTERM. Nothing removes the pidfile when a @@ -1185,9 +1020,7 @@ func TestCheckDoesNotSignalABystanderHoldingAReusedPid(t *testing.T) { // left behind. It does not hold the log's lock, because it is not a // recorder. bystander := exec.Command("sleep", "30") - if err := bystander.Start(); err != nil { - t.Fatalf("start the bystander: %v", err) - } + require.NoError(t, bystander.Start()) t.Cleanup(func() { _ = bystander.Process.Kill() _ = bystander.Wait() @@ -1195,12 +1028,10 @@ func TestCheckDoesNotSignalABystanderHoldingAReusedPid(t *testing.T) { writePid(t, logPath+".pid", fmt.Sprintf("%d\n", bystander.Process.Pid)) cfg := recorderConfig(t, newVirtualClock(testNow), logPath) - if _, err := check(context.Background(), cfg, newCheckSource(nil)); err != nil { - t.Fatalf("check() = %v, want nil\nnotes:\n%s", err, notesOf(cfg)) - } - if err := syscall.Kill(bystander.Process.Pid, 0); err != nil { - t.Fatalf("the bystander is gone (%v): check signalled a process that was not the recorder", err) - } + _, err := check(context.Background(), cfg, newCheckSource(nil)) + require.NoError(t, err) + require.NoError(t, syscall.Kill(bystander.Process.Pid, 0), + "check signalled a process that was not the recorder") } // A dead pidfile (the recorder process has already exited, holding no flock) @@ -1212,40 +1043,29 @@ func TestCheckDeadPidWithNoSentinelIsUnobservable(t *testing.T) { windowEnd := testNow.Add(5*time.Minute + checkGrace) logPath := filepath.Join(dir, "log.jsonl") w, err := NewWriter(logPath, newFakeClock(testNow)) - if err != nil { - t.Fatalf("NewWriter: %v", err) - } - if err := w.WriteHeader(Header{ + require.NoError(t, err) + require.NoError(t, w.WriteHeader(Header{ URL: "https://grafana.example.com", GrafanaVersion: "13.1.0", StartedAt: testNow.Add(-time.Minute), Rules: []LoggedRule{{ UID: checkUID, Title: checkTitle, Folder: "F", Group: "G", IntervalSeconds: 60, NoDataState: "OK", ExecErrState: "OK", PollEverySeconds: checkPollEvery.Seconds(), }}, - }); err != nil { - t.Fatalf("WriteHeader: %v", err) - } + })) // Healthy heartbeats all the way past windowEnd — evaluatedThrough is // satisfied, so the drain wait needs no live re-poll — but no sentinel is // ever written: the recorder died before it could call Stop. for at := testNow.Add(-time.Minute); !at.After(windowEnd.Add(30 * time.Second)); at = at.Add(checkPollEvery) { - if err := w.WritePoll(Poll{RuleUID: checkUID, GrafanaNow: at, Found: true, State: "inactive", Health: "ok", LastEvaluation: at}); err != nil { - t.Fatalf("WritePoll: %v", err) - } - } - if err := w.Close(); err != nil { // no sentinel — a clean exit would call Stop - t.Fatalf("Close: %v", err) + require.NoError(t, w.WritePoll(Poll{RuleUID: checkUID, GrafanaNow: at, Found: true, State: "inactive", Health: "ok", LastEvaluation: at})) } + require.NoError(t, w.Close()) // no sentinel — a clean exit would call Stop writePid(t, logPath+".pid", fmt.Sprintf("%d\n", deadPid(t))) cfg := recorderConfig(t, newVirtualClock(testNow), logPath) res, err := check(context.Background(), cfg, newCheckSource(nil)) - if err == nil { - t.Fatalf("check() = nil, want an error: no sentinel means the recorder never proved it ran to the end") - } - if len(res.Verdicts) != 1 || res.Verdicts[0].Outcome != OutcomeUnobservable { - t.Fatalf("Verdicts = %+v, want one unobservable verdict", res.Verdicts) - } + require.Error(t, err, "no sentinel means the recorder never proved it ran to the end") + require.Len(t, res.Verdicts, 1) + require.Equal(t, OutcomeUnobservable, res.Verdicts[0].Outcome) } // An incomplete last line gives exit 2. log_test.go's TestReadLogRejectsBadLogs @@ -1263,25 +1083,19 @@ func TestCheckRecorderModeTruncatedLogFailsClosed(t *testing.T) { Rules: []LoggedRule{{UID: checkUID, Title: checkTitle, IntervalSeconds: 60, PollEverySeconds: checkPollEvery.Seconds()}}, } hb, err := json.Marshal(headerRecord{Type: RecordHeader, Header: h}) - if err != nil { - t.Fatalf("marshal header: %v", err) - } + require.NoError(t, err) pb, err := json.Marshal(pollRecord{Type: RecordPoll, Poll: Poll{ RuleUID: checkUID, GrafanaNow: testNow, Found: true, State: "inactive", Health: "ok", LastEvaluation: testNow, }}) - if err != nil { - t.Fatalf("marshal poll: %v", err) - } + require.NoError(t, err) content := string(hb) + "\n" + string(pb) + "\n" + `{"type":"poll","rule_ui` // torn mid-write - if err := os.WriteFile(path, []byte(content), 0o644); err != nil { - t.Fatalf("write log: %v", err) - } + require.NoError(t, os.WriteFile(path, []byte(content), 0o644)) writePid(t, path+".pid", fmt.Sprintf("%d\n", deadPid(t))) cfg := recorderConfig(t, newVirtualClock(testNow), path) - if _, err := check(context.Background(), cfg, newCheckSource(nil)); err == nil || !strings.Contains(err.Error(), "unparseable") { - t.Fatalf("check() = %v, want a refusal naming the unparseable tail", err) - } + _, err = check(context.Background(), cfg, newCheckSource(nil)) + require.Error(t, err) + require.Contains(t, err.Error(), "unparseable") } // One authority for the cadence, from check's side: maxGap comes from the @@ -1293,15 +1107,11 @@ func TestCheckDerivesMaxGapFromTheRecordedCadence(t *testing.T) { windowEnd := testNow.Add(5*time.Minute + checkGrace) path := filepath.Join(dir, "log.jsonl") w, err := NewWriter(path, newFakeClock(windowEnd.Add(30*time.Second))) - if err != nil { - t.Fatalf("NewWriter: %v", err) - } - if err := w.WriteHeader(Header{ + require.NoError(t, err) + require.NoError(t, w.WriteHeader(Header{ URL: "https://grafana.example.com", GrafanaVersion: "13.1.0", StartedAt: testNow.Add(-time.Minute), Rules: []LoggedRule{{UID: checkUID, Title: checkTitle, IntervalSeconds: 60, PollEverySeconds: 5}}, - }); err != nil { - t.Fatalf("WriteHeader: %v", err) - } + })) for at := testNow.Add(-time.Minute); !at.After(windowEnd.Add(30 * time.Second)); at = at.Add(5 * time.Second) { // A 20s hole: under the recorded 5s cadence maxGap is 10s and this // fails; under a cadence re-derived from intervalSeconds it would be @@ -1309,25 +1119,17 @@ func TestCheckDerivesMaxGapFromTheRecordedCadence(t *testing.T) { if at.After(testNow.Add(time.Minute)) && at.Before(testNow.Add(80*time.Second)) { continue } - if err := w.WritePoll(Poll{ + require.NoError(t, w.WritePoll(Poll{ RuleUID: checkUID, GrafanaNow: at, Found: true, State: "inactive", Health: "ok", LastEvaluation: at, - }); err != nil { - t.Fatalf("WritePoll: %v", err) - } - } - if err := w.Stop(); err != nil { - t.Fatalf("Stop: %v", err) + })) } + require.NoError(t, w.Stop()) writePid(t, path+".pid", fmt.Sprintf("%d\n", deadPid(t))) cfg := recorderConfig(t, newVirtualClock(testNow), path) res, err := check(context.Background(), cfg, newCheckSource(nil)) - if err == nil { - t.Fatalf("check() = nil; a 20s hole exceeds the 10s maxGap the recorded 5s cadence implies") - } - if got := res.Coverage[checkUID].Reason; got != ReasonHeartbeatGap { - t.Errorf("Reason = %q, want %q", got, ReasonHeartbeatGap) - } + require.Error(t, err, "a 20s hole exceeds the 10s maxGap the recorded 5s cadence implies") + require.Equal(t, ReasonHeartbeatGap, res.Coverage[checkUID].Reason) } // --------------------------------------------------------------------------- @@ -1367,9 +1169,7 @@ func TestEvaluatedThroughSpendsItsUncertaintyFailingClosed(t *testing.T) { for _, tc := range tests { t.Run(tc.name, func(t *testing.T) { - if got := evaluatedThrough(tc.lastEval, tc.skew, tc.bound, end); got != tc.wantSatisfied { - t.Errorf("evaluatedThrough() = %v, want %v", got, tc.wantSatisfied) - } + require.Equal(t, tc.wantSatisfied, evaluatedThrough(tc.lastEval, tc.skew, tc.bound, end)) }) } } @@ -1392,33 +1192,17 @@ func TestMergeDrainTimeoutsNamesEveryUnobservableRule(t *testing.T) { "a": {reason: ReasonDrainTimeout, note: "rule \"A\": did not evaluate through the end within the drain limit"}, "b": {reason: ReasonDrainTimeout, note: "rule \"B\": did not evaluate through the end within the drain limit"}, }) - if err == nil { - t.Fatal("mergeDrainTimeouts() = nil, want an error naming the newly unobservable rule") - } - // Its own shape: joined with decide's, two counts under one identical - // phrase would read as a contradiction rather than as two findings. - if !strings.Contains(err.Error(), "unobservable at the drain wait") { - t.Errorf("err = %q, want the drain wait's own error shape", err) - } + require.Error(t, err, "naming the newly unobservable rule") + require.Contains(t, err.Error(), "unobservable at the drain wait") // Only A is newly unobservable; B was already, so naming it twice would // only lengthen the message. - if !strings.Contains(err.Error(), "A ("+string(ReasonDrainTimeout)+")") { - t.Errorf("err = %q, want it to name A's drain timeout", err) - } - if strings.Contains(err.Error(), "B (") { - t.Errorf("err = %q, want it not to re-report B, which decide already reported", err) - } - if got := merged.Coverage["a"].Reason; got != ReasonDrainTimeout { - t.Errorf("Coverage[a].Reason = %q, want %q", got, ReasonDrainTimeout) - } + require.Contains(t, err.Error(), "A ("+string(ReasonDrainTimeout)+")") + require.NotContains(t, err.Error(), "B (") + require.Equal(t, ReasonDrainTimeout, merged.Coverage["a"].Reason) // B keeps the reason the coverage proof gave it — the FIRST reason wins, // as it does inside proveCoverage. - if got := merged.Coverage["b"].Reason; got != ReasonHeartbeatGap { - t.Errorf("Coverage[b].Reason = %q, want the earlier %q", got, ReasonHeartbeatGap) - } - if merged.Verdicts[0].Outcome != OutcomeUnobservable { - t.Errorf("Verdicts[0].Outcome = %q, want %q", merged.Verdicts[0].Outcome, OutcomeUnobservable) - } + require.Equal(t, ReasonHeartbeatGap, merged.Coverage["b"].Reason) + require.Equal(t, OutcomeUnobservable, merged.Verdicts[0].Outcome) } // ReadLogHeader is the one read of a log a writer may still hold, so its @@ -1429,45 +1213,30 @@ func TestReadLogHeader(t *testing.T) { t.Run("reads line 1 while the log keeps growing", func(t *testing.T) { path := filepath.Join(dir, "growing.jsonl") w, err := NewWriter(path, newFakeClock(testNow)) - if err != nil { - t.Fatalf("NewWriter: %v", err) - } + require.NoError(t, err) defer w.Close() - if err := w.WriteHeader(testHeader()); err != nil { - t.Fatalf("WriteHeader: %v", err) - } - if err := w.WritePoll(Poll{RuleUID: "rule1", GrafanaNow: testNow, Found: true}); err != nil { - t.Fatalf("WritePoll: %v", err) - } + require.NoError(t, w.WriteHeader(testHeader())) + require.NoError(t, w.WritePoll(Poll{RuleUID: "rule1", GrafanaNow: testNow, Found: true})) h, err := ReadLogHeader(path) - if err != nil { - t.Fatalf("ReadLogHeader: %v", err) - } - if h.URL != testHeader().URL || len(h.Rules) != 1 { - t.Errorf("header = %+v, want the written one", h) - } + require.NoError(t, err) + require.Equal(t, testHeader().URL, h.URL) + require.Len(t, h.Rules, 1) }) t.Run("a half-written header is not a header", func(t *testing.T) { path := filepath.Join(dir, "torn.jsonl") - if err := os.WriteFile(path, []byte(`{"type":"header","url":"htt`), 0o644); err != nil { - t.Fatalf("write: %v", err) - } - if _, err := ReadLogHeader(path); err == nil || - !strings.Contains(err.Error(), "no complete header") { - t.Fatalf("ReadLogHeader() = %v, want a refusal", err) - } + require.NoError(t, os.WriteFile(path, []byte(`{"type":"header","url":"htt`), 0o644)) + _, err := ReadLogHeader(path) + require.Error(t, err) + require.Contains(t, err.Error(), "no complete header") }) t.Run("a wrong schema version is refused", func(t *testing.T) { path := filepath.Join(dir, "old.jsonl") - if err := os.WriteFile(path, []byte(`{"type":"header","schema_version":99,"url":"u"}`+"\n"), 0o644); err != nil { - t.Fatalf("write: %v", err) - } - if _, err := ReadLogHeader(path); err == nil || - !strings.Contains(err.Error(), "schema version 99") { - t.Fatalf("ReadLogHeader() = %v, want a schema refusal", err) - } + require.NoError(t, os.WriteFile(path, []byte(`{"type":"header","schema_version":99,"url":"u"}`+"\n"), 0o644)) + _, err := ReadLogHeader(path) + require.Error(t, err) + require.Contains(t, err.Error(), "schema version 99") }) } diff --git a/grafana-alertcheck/internal/gate/classify_test.go b/grafana-alertcheck/internal/gate/classify_test.go index 017d7ae6b..ad25a6ea0 100644 --- a/grafana-alertcheck/internal/gate/classify_test.go +++ b/grafana-alertcheck/internal/gate/classify_test.go @@ -3,6 +3,8 @@ package gate import ( "testing" "time" + + "github.com/stretchr/testify/require" ) func lbl(name string) map[string]string { return map[string]string{"instance": name} } @@ -55,9 +57,9 @@ func TestClassifyRule_NoEvidenceIsClean(t *testing.T) { polls := []Poll{quietPoll("r1", from), quietPoll("r1", to)} outcome, badFor, viols := classifyRule(def, polls, from, to, defaultBad, PreexistingFailUnlessRecovered) - if outcome != OutcomeClean || badFor != 0 || len(viols) != 0 { - t.Fatalf("outcome=%v badFor=%v viols=%v, want clean/0/none", outcome, badFor, viols) - } + require.Equal(t, OutcomeClean, outcome) + require.Zero(t, badFor) + require.Empty(t, viols) } func TestClassifyRule_NewOnsetInsideWindowIsNewlyBad(t *testing.T) { @@ -72,15 +74,10 @@ func TestClassifyRule_NewOnsetInsideWindowIsNewlyBad(t *testing.T) { abnormalPoll("r1", to, StateFiring, lbl("a"), onset), } outcome, badFor, viols := classifyRule(def, polls, from, to, defaultBad, PreexistingFailUnlessRecovered) - if outcome != OutcomeNewlyBad { - t.Fatalf("outcome = %v, want newly_bad", outcome) - } - if want := to.Sub(onset); badFor != want { - t.Fatalf("badFor = %v, want %v", badFor, want) - } - if len(viols) != 1 || viols[0].Outcome != OutcomeNewlyBad { - t.Fatalf("viols = %+v, want exactly one newly_bad violation", viols) - } + require.Equal(t, OutcomeNewlyBad, outcome) + require.Equal(t, to.Sub(onset), badFor) + require.Len(t, viols, 1) + require.Equal(t, OutcomeNewlyBad, viols[0].Outcome) } // A genuinely new bad episode fails even if it clears again before the window @@ -99,12 +96,8 @@ func TestClassifyRule_NewOnsetThatClearsStillFails(t *testing.T) { quietPoll("r1", to), } outcome, _, viols := classifyRule(def, polls, from, to, defaultBad, PreexistingFailUnlessRecovered) - if outcome != OutcomeNewlyBad { - t.Fatalf("outcome = %v, want newly_bad even though it cleared", outcome) - } - if len(viols) != 1 { - t.Fatalf("viols = %+v, want one violation", viols) - } + require.Equal(t, OutcomeNewlyBad, outcome, "even though it cleared") + require.Len(t, viols, 1) } // --- recovered / persistently_bad (preexisting) --- @@ -122,15 +115,9 @@ func TestClassifyRule_PreexistingThatRecoversIsRecoveredAndNotAViolation(t *test quietPoll("r1", to), } outcome, badFor, viols := classifyRule(def, polls, from, to, defaultBad, PreexistingFailUnlessRecovered) - if outcome != OutcomeRecovered { - t.Fatalf("outcome = %v, want recovered", outcome) - } - if want := clearAt.Sub(from); badFor != want { - t.Fatalf("badFor = %v, want %v", badFor, want) - } - if len(viols) != 0 { - t.Fatalf("viols = %+v, want none: default policy passes a recovered preexisting instance", viols) - } + require.Equal(t, OutcomeRecovered, outcome) + require.Equal(t, clearAt.Sub(from), badFor) + require.Empty(t, viols, "default policy passes a recovered preexisting instance") } // The late condition: bad for 58 of a 60-minute window, clear at minute 58, @@ -149,15 +136,9 @@ func TestClassifyRule_LateRecoveryPassesRegardlessOfHowLateItIs(t *testing.T) { quietPoll("r1", to), } outcome, badFor, viols := classifyRule(def, polls, from, to, defaultBad, PreexistingFailUnlessRecovered) - if outcome != OutcomeRecovered { - t.Fatalf("outcome = %v, want recovered even 58 minutes into a 60-minute window", outcome) - } - if want := clearAt.Sub(from); badFor != want { - t.Fatalf("badFor = %v, want the full %v bad duration, not a value clamped against a deadline", badFor, want) - } - if len(viols) != 0 { - t.Fatalf("viols = %+v, want none: there is no deadline a preexisting recovery must beat", viols) - } + require.Equal(t, OutcomeRecovered, outcome, "even 58 minutes into a 60-minute window") + require.Equal(t, clearAt.Sub(from), badFor, "not a value clamped against a deadline") + require.Empty(t, viols, "there is no deadline a preexisting recovery must beat") } func TestClassifyRule_PreexistingStillBadAtWindowEndIsPersistentlyBad(t *testing.T) { @@ -170,15 +151,10 @@ func TestClassifyRule_PreexistingStillBadAtWindowEndIsPersistentlyBad(t *testing abnormalPoll("r1", to, StateFiring, lbl("a"), from.Add(-time.Hour)), } outcome, badFor, viols := classifyRule(def, polls, from, to, defaultBad, PreexistingFailUnlessRecovered) - if outcome != OutcomePersistentlyBad { - t.Fatalf("outcome = %v, want persistently_bad", outcome) - } - if badFor != to.Sub(from) { - t.Fatalf("badFor = %v, want the full window %v", badFor, to.Sub(from)) - } - if len(viols) != 1 || viols[0].Outcome != OutcomePersistentlyBad { - t.Fatalf("viols = %+v, want one persistently_bad violation", viols) - } + require.Equal(t, OutcomePersistentlyBad, outcome) + require.Equal(t, to.Sub(from), badFor) + require.Len(t, viols, 1) + require.Equal(t, OutcomePersistentlyBad, viols[0].Outcome) } // --- flapping --- @@ -196,12 +172,9 @@ func TestClassifyRule_ClearThenBadAgainIsFlapping(t *testing.T) { quietPoll("r1", to), } outcome, _, viols := classifyRule(def, polls, from, to, defaultBad, PreexistingFailUnlessRecovered) - if outcome != OutcomeFlapping { - t.Fatalf("outcome = %v, want flapping", outcome) - } - if len(viols) != 1 || viols[0].Outcome != OutcomeFlapping { - t.Fatalf("viols = %+v, want one flapping violation, always a fail regardless of policy", viols) - } + require.Equal(t, OutcomeFlapping, outcome) + require.Len(t, viols, 1) + require.Equal(t, OutcomeFlapping, viols[0].Outcome, "always a fail regardless of policy") } // A clear and then a second bad state gives flapping, wherever the second bad @@ -233,12 +206,9 @@ func TestClassifyRule_FlappingAtEveryTimingOfTheSecondOnset(t *testing.T) { quietPoll("r1", to), } outcome, _, viols := classifyRule(def, polls, from, to, defaultBad, PreexistingFailUnlessRecovered) - if outcome != OutcomeFlapping { - t.Fatalf("outcome = %v, want flapping for a second onset at %s", outcome, tc.secondOnset) - } - if len(viols) != 1 || viols[0].Outcome != OutcomeFlapping { - t.Fatalf("viols = %+v, want one flapping violation", viols) - } + require.Equalf(t, OutcomeFlapping, outcome, "second onset at %s", tc.secondOnset) + require.Len(t, viols, 1) + require.Equal(t, OutcomeFlapping, viols[0].Outcome) }) } } @@ -257,15 +227,9 @@ func TestClassifyRule_VanishedWhileBadStaysPersistentlyBad(t *testing.T) { quietPoll("r1", to), } outcome, badFor, viols := classifyRule(def, polls, from, to, defaultBad, PreexistingFailUnlessRecovered) - if outcome != OutcomePersistentlyBad { - t.Fatalf("outcome = %v, want persistently_bad: a vanish must never read as a recovery", outcome) - } - if badFor != to.Sub(from) { - t.Fatalf("badFor = %v, want the full window %v: the freeze must hold the episode open to windowEnd", badFor, to.Sub(from)) - } - if len(viols) != 1 { - t.Fatalf("viols = %+v, want one violation", viols) - } + require.Equal(t, OutcomePersistentlyBad, outcome, "a vanish must never read as a recovery") + require.Equal(t, to.Sub(from), badFor, "the freeze must hold the episode open to windowEnd") + require.Len(t, viols, 1) } func TestClassifyRule_VanishedWhileNeverBadIsUninteresting(t *testing.T) { @@ -282,9 +246,9 @@ func TestClassifyRule_VanishedWhileNeverBadIsUninteresting(t *testing.T) { quietPoll("r1", to), } outcome, badFor, viols := classifyRule(def, polls, from, to, defaultBad, PreexistingFailUnlessRecovered) - if outcome != OutcomeClean || badFor != 0 || len(viols) != 0 { - t.Fatalf("outcome=%v badFor=%v viols=%v, want clean/0/none", outcome, badFor, viols) - } + require.Equal(t, OutcomeClean, outcome) + require.Zero(t, badFor) + require.Empty(t, viols) } // --- preexisting policy --- @@ -301,12 +265,10 @@ func TestClassifyRule_PreexistingPolicyFailFailsARecoveredInstance(t *testing.T) quietPoll("r1", to), } outcome, _, viols := classifyRule(def, polls, from, to, defaultBad, PreexistingFail) - if outcome != OutcomeRecovered { - t.Fatalf("outcome = %v, want recovered — the descriptive outcome does not change under policy=fail", outcome) - } - if len(viols) != 1 || viols[0].Outcome != OutcomeRecovered { - t.Fatalf("viols = %+v, want one violation: policy=fail gives no benefit of the doubt to a preexisting instance", viols) - } + require.Equal(t, OutcomeRecovered, outcome, "the descriptive outcome does not change under policy=fail") + require.Len(t, viols, 1) + require.Equal(t, OutcomeRecovered, viols[0].Outcome, + "policy=fail gives no benefit of the doubt to a preexisting instance") } func TestClassifyRule_PreexistingPolicyIgnoreForgivesPersistentlyBad(t *testing.T) { @@ -319,12 +281,8 @@ func TestClassifyRule_PreexistingPolicyIgnoreForgivesPersistentlyBad(t *testing. abnormalPoll("r1", to, StateFiring, lbl("a"), from.Add(-time.Hour)), } outcome, _, viols := classifyRule(def, polls, from, to, defaultBad, PreexistingIgnore) - if outcome != OutcomePersistentlyBad { - t.Fatalf("outcome = %v, want persistently_bad — the descriptive outcome does not change under policy=ignore", outcome) - } - if len(viols) != 0 { - t.Fatalf("viols = %+v, want none: policy=ignore disregards a preexisting instance even if it never recovers", viols) - } + require.Equal(t, OutcomePersistentlyBad, outcome, "the descriptive outcome does not change under policy=ignore") + require.Empty(t, viols, "policy=ignore disregards a preexisting instance even if it never recovers") } func TestClassifyRule_PreexistingPolicyIgnoreStillFailsANewOnset(t *testing.T) { @@ -339,9 +297,8 @@ func TestClassifyRule_PreexistingPolicyIgnoreStillFailsANewOnset(t *testing.T) { abnormalPoll("r1", to, StateFiring, lbl("a"), onset), } outcome, _, viols := classifyRule(def, polls, from, to, defaultBad, PreexistingIgnore) - if outcome != OutcomeNewlyBad || len(viols) != 1 { - t.Fatalf("outcome=%v viols=%v, want newly_bad/1: ignore only forgives PREEXISTING badness", outcome, viols) - } + require.Equal(t, OutcomeNewlyBad, outcome) + require.Len(t, viols, 1, "ignore only forgives PREEXISTING badness") } // --- worst-of across instances --- @@ -366,12 +323,9 @@ func TestClassifyRule_WorstOfMultipleInstancesWins(t *testing.T) { }, } outcome, _, viols := classifyRule(def, polls, from, to, defaultBad, PreexistingFailUnlessRecovered) - if outcome != OutcomePersistentlyBad { - t.Fatalf("outcome = %v, want persistently_bad: the worse of {recovered, persistently_bad}", outcome) - } - if len(viols) != 1 || viols[0].Outcome != OutcomePersistentlyBad { - t.Fatalf("viols = %+v, want exactly the persistently_bad instance's violation", viols) - } + require.Equal(t, OutcomePersistentlyBad, outcome, "the worse of {recovered, persistently_bad}") + require.Len(t, viols, 1) + require.Equal(t, OutcomePersistentlyBad, viols[0].Outcome) } // --- decide(): skipped rules, unobservable, MinObserved, exit mapping --- @@ -390,15 +344,11 @@ func TestDecide_SkippedRuleNeverReachesProveCoverage(t *testing.T) { // The HEADER is what says paused — decide reads skipped from there, not // from def.IsPaused, which is a post-window reading (Header.pausedAtStart). res, err := decide(pausedHeader(from.Add(-time.Hour), "r1"), nil, nil, defs, rt, gt, pol) - if err != nil { - t.Fatalf("err = %v, want nil: a rule paused before the window is skipped, not unobservable", err) - } - if len(res.Verdicts) != 1 || res.Verdicts[0].Outcome != OutcomeSkipped { - t.Fatalf("Verdicts = %+v, want exactly one skipped verdict", res.Verdicts) - } - if _, ok := res.Coverage["r1"]; ok { - t.Fatalf("Coverage[r1] present, want absent: a skipped rule has no coverage to prove") - } + require.NoError(t, err, "a rule paused before the window is skipped, not unobservable") + require.Len(t, res.Verdicts, 1) + require.Equal(t, OutcomeSkipped, res.Verdicts[0].Outcome) + _, ok := res.Coverage["r1"] + require.False(t, ok, "a skipped rule has no coverage to prove") } func TestDecide_UnobservableRuleAlwaysReturnsAnError(t *testing.T) { @@ -413,12 +363,9 @@ func TestDecide_UnobservableRuleAlwaysReturnsAnError(t *testing.T) { // No sentinel at all: check 1 fails, so the rule is unobservable // regardless of anything else. res, err := decide(Header{StartedAt: from.Add(-time.Hour)}, nil, nil, defs, rt, gt, pol) - if err == nil { - t.Fatalf("err = nil, want non-nil: an unobservable rule must always fail the run") - } - if len(res.Verdicts) != 1 || res.Verdicts[0].Outcome != OutcomeUnobservable { - t.Fatalf("Verdicts = %+v, want exactly one unobservable verdict", res.Verdicts) - } + require.Error(t, err, "an unobservable rule must always fail the run") + require.Len(t, res.Verdicts, 1) + require.Equal(t, OutcomeUnobservable, res.Verdicts[0].Outcome) } // Any unobservable rule means exit 2, with no exception — even alongside a @@ -450,9 +397,7 @@ func TestDecide_UnobservableWinsEvenAlongsideARealViolation(t *testing.T) { sentinel := to res, err := decide(Header{StartedAt: from.Add(-time.Hour)}, polls, &sentinel, defs, rt, gt, pol) - if err == nil { - t.Fatalf("err = nil, want non-nil: one rule is unobservable") - } + require.Error(t, err, "one rule is unobservable") var gotBroken, gotBad Outcome for _, v := range res.Verdicts { switch v.RuleUID { @@ -462,15 +407,11 @@ func TestDecide_UnobservableWinsEvenAlongsideARealViolation(t *testing.T) { gotBad = v.Outcome } } - if gotBroken != OutcomeUnobservable { - t.Fatalf("broken.Outcome = %v, want unobservable", gotBroken) - } - if gotBad != OutcomeNewlyBad { - t.Fatalf("bad.Outcome = %v, want newly_bad: classification still runs and is still visible in Verdicts", gotBad) - } - if len(res.Violations) == 0 { - t.Fatalf("Violations empty, want the newly_bad instance still reported even though the run fails on the unobservable rule") - } + require.Equal(t, OutcomeUnobservable, gotBroken) + require.Equal(t, OutcomeNewlyBad, gotBad, + "classification still runs and is still visible in Verdicts") + require.NotEmpty(t, res.Violations, + "the newly_bad instance still reported even though the run fails on the unobservable rule") } // A clean verdict with a coverage gap must never give exit 0, and recovered @@ -550,9 +491,7 @@ func TestDecide_UnobservableRuleWinsOverEveryFavorableOutcome(t *testing.T) { // so it is unobservable regardless of "good". sentinel := to res, err := decide(h, tc.goodPolls, &sentinel, defs, rt, gt, pol) - if err == nil { - t.Fatalf("err = nil, want non-nil: 'broken' is unobservable regardless of 'good' being %s", tc.name) - } + require.Errorf(t, err, "'broken' is unobservable regardless of 'good' being %s", tc.name) var gotGood, gotBroken Outcome for _, v := range res.Verdicts { switch v.RuleUID { @@ -562,12 +501,8 @@ func TestDecide_UnobservableRuleWinsOverEveryFavorableOutcome(t *testing.T) { gotBroken = v.Outcome } } - if gotGood != tc.wantOutcome { - t.Errorf("good.Outcome = %v, want %v", gotGood, tc.wantOutcome) - } - if gotBroken != OutcomeUnobservable { - t.Errorf("broken.Outcome = %v, want unobservable", gotBroken) - } + require.Equal(t, tc.wantOutcome, gotGood) + require.Equal(t, OutcomeUnobservable, gotBroken) }) } } @@ -603,15 +538,10 @@ func TestDecide_RecoveredOutcomeOverriddenByItsOwnCoverageGap(t *testing.T) { sentinel := to res, err := decide(Header{StartedAt: from.Add(-time.Hour)}, polls, &sentinel, defs, rt, gt, pol) - if err == nil { - t.Fatalf("err = nil, want non-nil: r1's own coverage gap must fail the run even though it recovered") - } - if len(res.Verdicts) != 1 || res.Verdicts[0].Outcome != OutcomeUnobservable { - t.Fatalf("Verdicts = %+v, want unobservable, never recovered", res.Verdicts) - } - if cov := res.Coverage["r1"]; cov.Proved { - t.Fatalf("Coverage = %+v, want not proved", cov) - } + require.Error(t, err, "r1's own coverage gap must fail the run even though it recovered") + require.Len(t, res.Verdicts, 1) + require.Equal(t, OutcomeUnobservable, res.Verdicts[0].Outcome, "never recovered") + require.False(t, res.Coverage["r1"].Proved) } func TestDecide_CleanWindowIsAPass(t *testing.T) { @@ -630,15 +560,9 @@ func TestDecide_CleanWindowIsAPass(t *testing.T) { sentinel := to res, err := decide(Header{StartedAt: from.Add(-time.Hour)}, polls, &sentinel, defs, rt, gt, pol) - if err != nil { - t.Fatalf("err = %v, want nil", err) - } - if len(res.Violations) != 0 { - t.Fatalf("Violations = %+v, want none: a pass is exactly len(Violations)==0 && err==nil", res.Violations) - } - if res.Verdicts[0].Outcome != OutcomeClean { - t.Fatalf("Outcome = %v, want clean", res.Verdicts[0].Outcome) - } + require.NoError(t, err) + require.Empty(t, res.Violations, "a pass is exactly len(Violations)==0 && err==nil") + require.Equal(t, OutcomeClean, res.Verdicts[0].Outcome) } // A pause and then an unpause inside the window, with an episode that would @@ -678,15 +602,10 @@ func TestDecide_PauseThenUnpauseWithHiddenEpisodeGivesUnobservableNotClean(t *te sentinel := to res, err := decide(Header{StartedAt: from.Add(-time.Hour)}, polls, &sentinel, defs, rt, gt, pol) - if err == nil { - t.Fatalf("err = nil, want the pause-then-unpause blind interval to fail closed") - } - if len(res.Verdicts) != 1 || res.Verdicts[0].Outcome != OutcomeUnobservable { - t.Fatalf("Verdicts = %+v, want unobservable, never clean", res.Verdicts) - } - if cov := res.Coverage["r1"]; cov.Proved { - t.Fatalf("Coverage = %+v, want not proved", cov) - } + require.Error(t, err, "the pause-then-unpause blind interval must fail closed") + require.Len(t, res.Verdicts, 1) + require.Equal(t, OutcomeUnobservable, res.Verdicts[0].Outcome, "never clean") + require.False(t, res.Coverage["r1"].Proved) } // --- MinObserved shortfall --- @@ -713,19 +632,14 @@ func TestDecide_SkippedOnlyShortfallProducesAViolationWithoutAnError(t *testing. sentinel := to res, err := decide(pausedHeader(from.Add(-time.Hour), "paused"), polls, &sentinel, defs, rt, gt, pol) - if err != nil { - t.Fatalf("err = %v, want nil: a shortfall caused only by a skipped rule is exit 1, not exit 2", err) - } - if len(res.Violations) != 1 { - t.Fatalf("Violations = %+v, want exactly one: a shortfall must be visible through Violations like any other fail reason", res.Violations) - } - if v := res.Violations[0]; v.Outcome != OutcomeSkipped || v.RuleUID != "paused" || v.Alert != "Paused" { - t.Fatalf("Violations[0] = %+v, want Outcome=skipped naming the paused rule", v) - } - if res.Violations[0].Note == "" { - t.Fatalf("Violations[0].Note is empty, want an explanation: the shortfall reason must not be smuggled into LastError, " + - "which is reporting-only rule state from a real poll this synthetic Violation never touched") - } + require.NoError(t, err, "a shortfall caused only by a skipped rule is exit 1, not exit 2") + require.Len(t, res.Violations, 1) + v := res.Violations[0] + require.Equal(t, OutcomeSkipped, v.Outcome) + require.Equal(t, "paused", v.RuleUID) + require.Equal(t, "Paused", v.Alert) + require.NotEmpty(t, v.Note, + "the shortfall reason must not be smuggled into LastError") } // An operator-supplied MinObserved that exceeds what could ever be resolved is @@ -748,17 +662,10 @@ func TestDecide_ExplicitMinObservedShortfallWithNoPausedRuleStillProducesAViolat sentinel := to res, err := decide(Header{StartedAt: from.Add(-time.Hour)}, polls, &sentinel, defs, rt, gt, pol) - if err != nil { - t.Fatalf("err = %v, want nil: an unmet MinObserved is exit 1, never exit 2", err) - } - if len(res.Violations) != 2 { - t.Fatalf("Violations = %+v, want two: the shortfall (3-1=2) is not explained by any paused rule, "+ - "so it must surface directly rather than pass silently", res.Violations) - } + require.NoError(t, err, "an unmet MinObserved is exit 1, never exit 2") + require.Len(t, res.Violations, 2, "the shortfall (3-1=2) must surface directly rather than pass silently") for _, v := range res.Violations { - if v.Outcome != OutcomeSkipped { - t.Fatalf("Violations = %+v, want Outcome=skipped on the synthetic shortfall entries", res.Violations) - } + require.Equal(t, OutcomeSkipped, v.Outcome) } } @@ -783,12 +690,8 @@ func TestDecide_AllowPausedSuppressesTheShortfall(t *testing.T) { sentinel := to res, err := decide(pausedHeader(from.Add(-time.Hour), "paused"), polls, &sentinel, defs, rt, gt, pol) - if err != nil { - t.Fatalf("err = %v, want nil", err) - } - if len(res.Violations) != 0 { - t.Fatalf("Violations = %+v, want none: --allow-paused must suppress the shortfall entirely", res.Violations) - } + require.NoError(t, err) + require.Empty(t, res.Violations, "--allow-paused must suppress the shortfall entirely") } // --- nodata escalation (decide's own Policy-driven check) --- @@ -809,12 +712,8 @@ func TestDecide_NodataIsUnobservableEscalatesASustainedRun(t *testing.T) { sentinel := to res, err := decide(Header{StartedAt: from.Add(-time.Hour)}, polls, &sentinel, defs, rt, gt, pol) - if err == nil { - t.Fatalf("err = nil, want non-nil: a sustained nodata run must be unobservable under --nodata-is-unobservable") - } - if res.Coverage["r1"].Reason != ReasonNodata { - t.Fatalf("Reason = %q, want %q", res.Coverage["r1"].Reason, ReasonNodata) - } + require.Error(t, err, "a sustained nodata run must be unobservable under --nodata-is-unobservable") + require.Equal(t, ReasonNodata, res.Coverage["r1"].Reason) } func TestDecide_NodataIsANoteByDefault(t *testing.T) { @@ -833,12 +732,8 @@ func TestDecide_NodataIsANoteByDefault(t *testing.T) { sentinel := to res, err := decide(Header{StartedAt: from.Add(-time.Hour)}, polls, &sentinel, defs, rt, gt, pol) - if err != nil { - t.Fatalf("err = %v, want nil: 96%% of the fleet runs no_data_state:OK and must not fail by default", err) - } - if res.Coverage["r1"].Unobservable { - t.Fatalf("Coverage[r1].Unobservable = true, want false by default") - } + require.NoError(t, err, "96%% of the fleet runs no_data_state:OK and must not fail by default") + require.False(t, res.Coverage["r1"].Unobservable) } // --- preexisting is decided by ActiveAt, not poll timing --- @@ -862,16 +757,11 @@ func TestClassifyRule_OnsetBetweenFromAndFirstPollIsNewlyBadNotRecovered(t *test quietPoll("r1", to), } outcome, badFor, viols := classifyRule(def, polls, from, to, defaultBad, PreexistingFailUnlessRecovered) - if outcome != OutcomeNewlyBad { - t.Fatalf("outcome = %v, want newly_bad: the onset is after `from`, so it is not preexisting even though "+ - "the FIRST in-window poll already observes it bad", outcome) - } - if len(viols) != 1 || viols[0].Outcome != OutcomeNewlyBad { - t.Fatalf("viols = %+v, want one newly_bad violation: a policy=fail-unless-recovered default must still fail this", viols) - } - if want := clearAt.Sub(onset); badFor != want { - t.Fatalf("badFor = %v, want %v: BadFor must count from the true onset, not from `from`", badFor, want) - } + require.Equal(t, OutcomeNewlyBad, outcome, + "the onset is after `from`, so it is not preexisting even though the FIRST in-window poll already observes it bad") + require.Len(t, viols, 1) + require.Equal(t, OutcomeNewlyBad, viols[0].Outcome) + require.Equal(t, clearAt.Sub(onset), badFor, "BadFor must count from the true onset, not from `from`") } // TestClassifyRule_OnsetJustBeforeFromIsPreexisting is the mirror check: an @@ -892,15 +782,10 @@ func TestClassifyRule_OnsetJustBeforeFromIsPreexisting(t *testing.T) { quietPoll("r1", to), } outcome, badFor, viols := classifyRule(def, polls, from, to, defaultBad, PreexistingFailUnlessRecovered) - if outcome != OutcomeRecovered { - t.Fatalf("outcome = %v, want recovered: the onset is at/before `from`, genuinely preexisting", outcome) - } - if len(viols) != 0 { - t.Fatalf("viols = %+v, want none: default policy passes a recovered preexisting instance", viols) - } - if want := clearAt.Sub(from); badFor != want { - t.Fatalf("badFor = %v, want %v: a preexisting episode's BadFor is clamped to window-open, not backdated past it", badFor, want) - } + require.Equal(t, OutcomeRecovered, outcome, "the onset is at/before `from`, genuinely preexisting") + require.Empty(t, viols, "default policy passes a recovered preexisting instance") + require.Equal(t, clearAt.Sub(from), badFor, + "a preexisting episode's BadFor is clamped to window-open, not backdated past it") } // A poll carrying a nonzero skew must have its ActiveAt (and GrafanaNow) @@ -928,12 +813,9 @@ func TestClassifyRule_SkewTranslatesActiveAtAcrossTheWindowBoundary(t *testing.T stillBad.LastEvaluation = to.Add(skew) outcome, badFor, _ := classifyRule(def, []Poll{poll, stillBad}, from, to, defaultBad, PreexistingFailUnlessRecovered) - if outcome != OutcomePersistentlyBad { - t.Fatalf("outcome = %v, want persistently_bad: a +90s skew must translate ActiveAt back to exactly `from`", outcome) - } - if badFor != to.Sub(from) { - t.Fatalf("badFor = %v, want the full window %v", badFor, to.Sub(from)) - } + require.Equal(t, OutcomePersistentlyBad, outcome, + "a +90s skew must translate ActiveAt back to exactly `from`") + require.Equal(t, to.Sub(from), badFor) } // --- InstanceLabels must survive a timeline first created by a bare marker --- @@ -954,13 +836,10 @@ func TestClassifyRule_LabelsSurviveWhenTimelineStartsFromAClearedMarker(t *testi quietPoll("r1", to), } _, _, viols := classifyRule(def, polls, from, to, defaultBad, PreexistingFailUnlessRecovered) - if len(viols) != 1 { - t.Fatalf("viols = %+v, want exactly one newly_bad violation", viols) - } - if viols[0].InstanceLabels == nil || viols[0].InstanceLabels["instance"] != "a" { - t.Fatalf("InstanceLabels = %+v, want {instance: a}: labels must backfill even though the "+ - "timeline was first created by a label-less Cleared marker", viols[0].InstanceLabels) - } + require.Len(t, viols, 1) + require.NotNil(t, viols[0].InstanceLabels) + require.Equal(t, "a", viols[0].InstanceLabels["instance"], + "labels must backfill even though the timeline was first created by a label-less Cleared marker") } // FirstSeen/ClearedAt are pinned exactly, not just that a violation exists. @@ -978,19 +857,11 @@ func TestClassifyRule_ViolationFieldsArePrecise(t *testing.T) { quietPoll("r1", to), } _, _, viols := classifyRule(def, polls, from, to, defaultBad, PreexistingFailUnlessRecovered) - if len(viols) != 1 { - t.Fatalf("viols = %+v, want exactly one violation", viols) - } + require.Len(t, viols, 1) v := viols[0] - if !v.FirstSeen.Equal(onset) { - t.Fatalf("FirstSeen = %v, want %v", v.FirstSeen, onset) - } - if !v.ClearedAt.Equal(clearAt) { - t.Fatalf("ClearedAt = %v, want %v", v.ClearedAt, clearAt) - } - if v.InstanceLabels["instance"] != "a" { - t.Fatalf("InstanceLabels = %+v, want {instance: a}", v.InstanceLabels) - } + require.True(t, v.FirstSeen.Equal(onset)) + require.True(t, v.ClearedAt.Equal(clearAt)) + require.Equal(t, "a", v.InstanceLabels["instance"]) } // The episode.end clamp: inWindowPolls admits a poll up to its own skew bound @@ -1013,13 +884,9 @@ func TestClassifyRule_ClearedEventPastWindowEndClampsToWindowEnd(t *testing.T) { }, } outcome, badFor, _ := classifyRule(def, polls, from, to, defaultBad, PreexistingFailUnlessRecovered) - if outcome != OutcomeRecovered { - t.Fatalf("outcome = %v, want recovered", outcome) - } - if badFor != to.Sub(from) { - t.Fatalf("badFor = %v, want the window %v exactly: the episode end must clamp to windowEnd, "+ - "not extend to the late Cleared event's raw time", badFor, to.Sub(from)) - } + require.Equal(t, OutcomeRecovered, outcome) + require.Equal(t, to.Sub(from), badFor, + "the episode end must clamp to windowEnd, not extend to the late Cleared event's raw time") } // TestClassifyRule_OnsetJustPastWindowEndIsNewlyBadNotClean pins the fail-closed @@ -1047,16 +914,10 @@ func TestClassifyRule_OnsetJustPastWindowEndIsNewlyBadNotClean(t *testing.T) { } outcome, badFor, viols := classifyRule(def, []Poll{poll}, from, windowEnd, defaultBad, PreexistingFailUnlessRecovered) - if outcome != OutcomeNewlyBad { - t.Fatalf("outcome = %v, want newly_bad: an onset past windowEnd seen only via the skew bound must fail closed", outcome) - } - if badFor != 0 { - t.Fatalf("badFor = %v, want 0: the zero-length episode must truncate to the window end", badFor) - } - if len(viols) != 1 { - t.Fatalf("viols = %+v, want exactly one newly_bad violation", viols) - - } + require.Equal(t, OutcomeNewlyBad, outcome, + "an onset past windowEnd seen only via the skew bound must fail closed") + require.Zero(t, badFor, "the zero-length episode must truncate to the window end") + require.Len(t, viols, 1) } // A clear after `to` gives persistently_bad. classifyRule filters @@ -1075,15 +936,10 @@ func TestClassifyRule_ClearAfterWindowEndIsPersistentlyBad(t *testing.T) { clearedPoll("r1", to.Add(time.Hour), key), // far past `to`, not a boundary case } outcome, badFor, viols := classifyRule(def, polls, from, to, defaultBad, PreexistingFailUnlessRecovered) - if outcome != OutcomePersistentlyBad { - t.Fatalf("outcome = %v, want persistently_bad: a clear outside the window must not read as a recovery", outcome) - } - if badFor != to.Sub(from) { - t.Fatalf("badFor = %v, want the full window %v", badFor, to.Sub(from)) - } - if len(viols) != 1 || viols[0].Outcome != OutcomePersistentlyBad { - t.Fatalf("viols = %+v, want one persistently_bad violation", viols) - } + require.Equal(t, OutcomePersistentlyBad, outcome, "a clear outside the window must not read as a recovery") + require.Equal(t, to.Sub(from), badFor) + require.Len(t, viols, 1) + require.Equal(t, OutcomePersistentlyBad, viols[0].Outcome) } // TestClassifyRule_CloseBeforeOpenClampsToZeroNotNegative pins the @@ -1110,19 +966,11 @@ func TestClassifyRule_CloseBeforeOpenClampsToZeroNotNegative(t *testing.T) { quietPoll("r1", to), } outcome, badFor, viols := classifyRule(def, polls, from, to, defaultBad, PreexistingFailUnlessRecovered) - if outcome != OutcomeNewlyBad { - t.Fatalf("outcome = %v, want newly_bad", outcome) - } - if badFor < 0 { - t.Fatalf("badFor = %v, want a non-negative duration even though the closing poll's translated "+ - "time landed before the opening poll's", badFor) - } - if badFor != 0 { - t.Fatalf("badFor = %v, want 0: the clamp collapses the inverted span to a zero-length episode", badFor) - } - if len(viols) != 1 { - t.Fatalf("viols = %+v, want one violation", viols) - } + require.Equal(t, OutcomeNewlyBad, outcome) + require.GreaterOrEqual(t, badFor, time.Duration(0), + "a non-negative duration even though the closing poll's translated time landed before the opening poll's") + require.Zero(t, badFor, "the clamp collapses the inverted span to a zero-length episode") + require.Len(t, viols, 1) } // --- mergeDurations --- @@ -1136,13 +984,9 @@ func TestMergeDurations_OverlappingEpisodesCountOnce(t *testing.T) { } got := mergeDurations(eps) want := 8*time.Minute + 1*time.Minute // [0,8) merged = 8m, plus the disjoint 1m - if got != want { - t.Fatalf("mergeDurations = %v, want %v: two simultaneously-bad instances must not double-count their overlap", got, want) - } + require.Equal(t, want, got, "two simultaneously-bad instances must not double-count their overlap") } func TestMergeDurations_Empty(t *testing.T) { - if got := mergeDurations(nil); got != 0 { - t.Fatalf("mergeDurations(nil) = %v, want 0", got) - } + require.Zero(t, mergeDurations(nil)) } diff --git a/grafana-alertcheck/internal/gate/coverage_test.go b/grafana-alertcheck/internal/gate/coverage_test.go index 720386d65..053cc43d2 100644 --- a/grafana-alertcheck/internal/gate/coverage_test.go +++ b/grafana-alertcheck/internal/gate/coverage_test.go @@ -4,6 +4,8 @@ import ( "strings" "testing" "time" + + "github.com/stretchr/testify/require" ) func TestProveCoverage_CleanWindowIsProved(t *testing.T) { @@ -19,9 +21,9 @@ func TestProveCoverage_CleanWindowIsProved(t *testing.T) { sentinel := to res := proveCoverage(Header{StartedAt: from.Add(-time.Hour)}, polls, &sentinel, rt, def, from, to, 0) - if !res.Proved || res.Unobservable || res.Reason != "" { - t.Fatalf("res = %+v, want a clean proved window", res) - } + require.True(t, res.Proved) + require.False(t, res.Unobservable) + require.Empty(t, res.Reason) } func TestProveCoverage_FiltersPollsByUID(t *testing.T) { @@ -40,9 +42,7 @@ func TestProveCoverage_FiltersPollsByUID(t *testing.T) { sentinel := to res := proveCoverage(Header{StartedAt: from.Add(-time.Hour)}, polls, &sentinel, rt, def, from, to, 0) - if !res.Proved { - t.Fatalf("res = %+v, want proved: a different rule's broken polls must not affect this rule's verdict", res) - } + require.True(t, res.Proved, "a different rule's broken polls must not affect this rule's verdict") } // --- Check 1: sentinel --- @@ -54,9 +54,8 @@ func TestProveCoverage_NoSentinelIsUnobservable(t *testing.T) { def := Definition{UID: "r1", Title: "R1"} res := proveCoverage(Header{StartedAt: from.Add(-time.Hour)}, nil, nil, rt, def, from, to, 0) - if res.Proved || res.Reason != ReasonNoSentinel { - t.Fatalf("res = %+v, want unobservable/no_sentinel: an absent sentinel must never be a pass", res) - } + require.False(t, res.Proved) + require.Equal(t, ReasonNoSentinel, res.Reason, "an absent sentinel must never be a pass") } func TestProveCoverage_SentinelBeforeGraceIsUnobservable(t *testing.T) { @@ -68,12 +67,9 @@ func TestProveCoverage_SentinelBeforeGraceIsUnobservable(t *testing.T) { sentinel := to.Add(grace).Add(-time.Second) // one second short of to+grace res := proveCoverage(Header{StartedAt: from.Add(-time.Hour)}, nil, &sentinel, rt, def, from, to, grace) - if res.Reason != ReasonSentinelEarly { - t.Fatalf("Reason = %q, want sentinel_early", res.Reason) - } - if !res.Unobservable || res.Proved { - t.Fatalf("res = %+v, want Unobservable and not Proved — a reason string with no consequence is not a coverage failure", res) - } + require.Equal(t, ReasonSentinelEarly, res.Reason) + require.True(t, res.Unobservable) + require.False(t, res.Proved, "a reason string with no consequence is not a coverage failure") // The consequence: decide() must turn this into exit 2, never a pass. defs := []Definition{def} @@ -81,12 +77,9 @@ func TestProveCoverage_SentinelBeforeGraceIsUnobservable(t *testing.T) { gt := globalTimings{transitionGrace: grace} pol := Policy{From: from, To: to} dres, err := decide(Header{StartedAt: from.Add(-time.Hour)}, nil, &sentinel, defs, drt, gt, pol) - if err == nil { - t.Fatalf("decide() err = nil, want non-nil: a sentinel short of to+grace must fail the run") - } - if len(dres.Verdicts) != 1 || dres.Verdicts[0].Outcome != OutcomeUnobservable { - t.Fatalf("Verdicts = %+v, want one unobservable verdict", dres.Verdicts) - } + require.Error(t, err, "a sentinel short of to+grace must fail the run") + require.Len(t, dres.Verdicts, 1) + require.Equal(t, OutcomeUnobservable, dres.Verdicts[0].Outcome) } func TestProveCoverage_SentinelExactlyAtGraceIsFine(t *testing.T) { @@ -104,9 +97,7 @@ func TestProveCoverage_SentinelExactlyAtGraceIsFine(t *testing.T) { sentinel := windowEnd res := proveCoverage(Header{StartedAt: from.Add(-time.Hour)}, polls, &sentinel, rt, def, from, to, grace) - if !res.Proved { - t.Fatalf("Proved = false, want true: sentinel exactly at to+grace must satisfy check 1: %+v", res) - } + require.True(t, res.Proved, "sentinel exactly at to+grace must satisfy check 1") } // --- Check 2: from bounds --- @@ -120,12 +111,9 @@ func TestProveCoverage_FromBeforeRecordIsUnobservable(t *testing.T) { sentinel := to res := proveCoverage(Header{StartedAt: started}, nil, &sentinel, rt, def, from, to, 0) - if res.Reason != ReasonFromBeforeRecord { - t.Fatalf("Reason = %q, want from_before_record", res.Reason) - } - if !res.Unobservable || res.Proved { - t.Fatalf("res = %+v, want Unobservable and not Proved — a reason string with no consequence is not a coverage failure", res) - } + require.Equal(t, ReasonFromBeforeRecord, res.Reason) + require.True(t, res.Unobservable) + require.False(t, res.Proved) // The consequence: decide() must turn this into exit 2, never a pass. defs := []Definition{def} @@ -133,12 +121,9 @@ func TestProveCoverage_FromBeforeRecordIsUnobservable(t *testing.T) { gt := globalTimings{} pol := Policy{From: from, To: to} dres, err := decide(Header{StartedAt: started}, nil, &sentinel, defs, drt, gt, pol) - if err == nil { - t.Fatalf("decide() err = nil, want non-nil: `from` before the recording started must fail the run") - } - if len(dres.Verdicts) != 1 || dres.Verdicts[0].Outcome != OutcomeUnobservable { - t.Fatalf("Verdicts = %+v, want one unobservable verdict", dres.Verdicts) - } + require.Error(t, err, "`from` before the recording started must fail the run") + require.Len(t, dres.Verdicts, 1) + require.Equal(t, OutcomeUnobservable, dres.Verdicts[0].Outcome) } // --- Check 3: heartbeat continuity --- @@ -158,18 +143,11 @@ func TestProveCoverage_HeartbeatGapBetweenBoundariesIsUnobservable(t *testing.T) sentinel := to res := proveCoverage(Header{StartedAt: from.Add(-time.Hour)}, polls, &sentinel, rt, def, from, to, 0) - if res.Reason != ReasonHeartbeatGap { - t.Fatalf("Reason = %q, want heartbeat_gap: healthy edges with a hole in the middle must still fail", res.Reason) - } + require.Equal(t, ReasonHeartbeatGap, res.Reason, "healthy edges with a hole in the middle must still fail") // The gap is the SPACING between the two polls (598s), not either // boundary segment (1s each) — pin the actual values, not just the verdict. - if res.LargestGap != 598*time.Second { - t.Fatalf("LargestGap = %s, want 598s (the spacing between the two polls, not a boundary segment)", res.LargestGap) - } - wantAt := from.Add(time.Second) - if !res.LargestGapAt.Equal(wantAt) { - t.Fatalf("LargestGapAt = %s, want %s (where the gap starts, at the first poll)", res.LargestGapAt, wantAt) - } + require.Equal(t, 598*time.Second, res.LargestGap) + require.True(t, res.LargestGapAt.Equal(from.Add(time.Second))) } // --- Check 4/5: health --- @@ -192,12 +170,8 @@ func TestProveCoverage_HealthErrorShortBlipPassesWithNote(t *testing.T) { sentinel := to res := proveCoverage(Header{StartedAt: from.Add(-time.Hour)}, polls, &sentinel, rt, def, from, to, 0) - if !res.Proved { - t.Fatalf("Proved = false, want true: one failed evaluation must not fail an otherwise clean window: %+v", res) - } - if !anyContains(res.Notes, "health=error") { - t.Fatalf("Notes = %v, want a health=error note even though it did not fail the window", res.Notes) - } + require.True(t, res.Proved, "one failed evaluation must not fail an otherwise clean window") + require.True(t, anyContains(res.Notes, "health=error"), "want a health=error note even though it did not fail the window") } func TestProveCoverage_HealthErrorSustainedIsUnobservable(t *testing.T) { @@ -218,9 +192,7 @@ func TestProveCoverage_HealthErrorSustainedIsUnobservable(t *testing.T) { sentinel := to res := proveCoverage(Header{StartedAt: from.Add(-time.Hour)}, polls, &sentinel, rt, def, from, to, 0) - if res.Reason != ReasonHealthError { - t.Fatalf("Reason = %q, want health_error for a run that outlasts healthGrace", res.Reason) - } + require.Equal(t, ReasonHealthError, res.Reason, "a run that outlasts healthGrace") } func TestProveCoverage_HealthNodataNeverFatalHere(t *testing.T) { @@ -236,13 +208,8 @@ func TestProveCoverage_HealthNodataNeverFatalHere(t *testing.T) { sentinel := to res := proveCoverage(Header{StartedAt: from.Add(-time.Hour)}, polls, &sentinel, rt, def, from, to, 0) - if !res.Proved { - t.Fatalf("Proved = false, want true: health=nodata for the WHOLE window must still not be fatal by itself "+ - "(escalating it is Policy.NodataIsUnobservable's job, applied by decide): %+v", res) - } - if !anyContains(res.Notes, "health=nodata") { - t.Fatalf("Notes = %v, want a health=nodata note", res.Notes) - } + require.True(t, res.Proved, "health=nodata for the WHOLE window must still not be fatal by itself") + require.True(t, anyContains(res.Notes, "health=nodata")) } // --- Check 6: liveness --- @@ -270,13 +237,9 @@ func TestProveCoverage_LivenessAbsoluteNeverFalseStale(t *testing.T) { sentinel := windowEnd res := proveCoverage(Header{StartedAt: from.Add(-time.Hour)}, polls, &sentinel, rt, def, from, windowEnd, 0) - if res.Reason == ReasonStaleEvaluation || res.BlindFor != 0 { - t.Fatalf("proveCoverage flagged staleness on a healthy rule polled at intervalSeconds/2 — liveness must be absolute, "+ - "never a delta against a previous poll: %+v", res) - } - if !res.Proved { - t.Fatalf("Proved = false, want true: %+v (notes: %v)", res, res.Notes) - } + require.NotEqual(t, ReasonStaleEvaluation, res.Reason, "liveness must be absolute, never a delta against a previous poll") + require.Zero(t, res.BlindFor) + require.True(t, res.Proved) } func TestProveCoverage_StaleEvaluationIsUnobservable(t *testing.T) { @@ -297,12 +260,8 @@ func TestProveCoverage_StaleEvaluationIsUnobservable(t *testing.T) { sentinel := to res := proveCoverage(Header{StartedAt: from.Add(-time.Hour)}, polls, &sentinel, rt, def, from, to, 0) - if res.Reason != ReasonStaleEvaluation { - t.Fatalf("Reason = %q, want stale_evaluation", res.Reason) - } - if res.BlindFor != 3*time.Minute { - t.Fatalf("BlindFor = %s, want 3m", res.BlindFor) - } + require.Equal(t, ReasonStaleEvaluation, res.Reason) + require.Equal(t, 3*time.Minute, res.BlindFor) } func TestProveCoverage_ZeroLastEvaluationNeverFalseStale(t *testing.T) { @@ -319,9 +278,7 @@ func TestProveCoverage_ZeroLastEvaluationNeverFalseStale(t *testing.T) { sentinel := to res := proveCoverage(Header{StartedAt: from.Add(-time.Hour)}, polls, &sentinel, rt, def, from, to, 0) - if res.Reason == ReasonStaleEvaluation { - t.Fatalf("a zero lastEvaluation on a paused poll must not trigger check 6: %+v", res) - } + require.NotEqual(t, ReasonStaleEvaluation, res.Reason, "a zero lastEvaluation on a paused poll must not trigger check 6") } // --- Check 7: isPaused in-window --- @@ -345,9 +302,7 @@ func TestProveCoverage_PausedInWindowIsUnobservable(t *testing.T) { sentinel := to res := proveCoverage(Header{StartedAt: from.Add(-time.Hour)}, polls, &sentinel, rt, def, from, to, 0) - if res.Reason != ReasonPausedInWindow { - t.Fatalf("Reason = %q, want paused_in_window", res.Reason) - } + require.Equal(t, ReasonPausedInWindow, res.Reason) } // TestProveCoverage_PausedAfterWindowIsFine pins check 7's respect for the @@ -369,12 +324,8 @@ func TestProveCoverage_PausedAfterWindowIsFine(t *testing.T) { sentinel := to res := proveCoverage(Header{StartedAt: from.Add(-time.Hour)}, polls, &sentinel, rt, def, from, to, 0) - if res.Reason == ReasonPausedInWindow { - t.Fatalf("a paused poll after windowEnd tripped check 7: %+v", res.Notes) - } - if !res.Proved { - t.Fatalf("Proved = false, want a clean window: %+v", res.Notes) - } + require.NotEqual(t, ReasonPausedInWindow, res.Reason, "a paused poll after windowEnd tripped check 7") + require.True(t, res.Proved) } // --- Check 8: rule absent --- @@ -399,9 +350,7 @@ func TestProveCoverage_RuleAbsentIsUnobservable(t *testing.T) { sentinel := to res := proveCoverage(Header{StartedAt: from.Add(-time.Hour)}, polls, &sentinel, rt, def, from, to, 0) - if res.Reason != ReasonRuleAbsent { - t.Fatalf("Reason = %q, want rule_absent", res.Reason) - } + require.Equal(t, ReasonRuleAbsent, res.Reason) } // denseHealthyPolls builds a clean poll sequence at a fixed cadence, with @@ -436,12 +385,8 @@ func TestProveCoverage_KeepLastObservedIsNoteOnly(t *testing.T) { sentinel := to res := proveCoverage(Header{StartedAt: from.Add(-time.Hour)}, polls, &sentinel, rt, def, from, to, 0) - if !res.Proved { - t.Fatalf("Proved = false, want true: KeepLast is a note, never fatal: %+v", res) - } - if !anyContains(res.Notes, "KeepLast") { - t.Fatalf("Notes = %v, want a KeepLast note (comma-joined membership, not a literal-key match)", res.Notes) - } + require.True(t, res.Proved, "KeepLast is a note, never fatal") + require.True(t, anyContains(res.Notes, "KeepLast"), "comma-joined membership, not a literal-key match") } // KeepLast in the CONFIGURATION gives a note — a different claim from the @@ -468,12 +413,9 @@ func TestProveCoverage_KeepLastConfiguredIsNoteOnly(t *testing.T) { sentinel := to res := proveCoverage(Header{StartedAt: from.Add(-time.Hour)}, polls, &sentinel, rt, tc.def, from, to, 0) - if !res.Proved { - t.Fatalf("Proved = false, want true: a declared KeepLast is a note, never fatal: %+v", res) - } - if !anyContains(res.Notes, "KeepLast") { - t.Fatalf("Notes = %v, want a KeepLast note from the definition alone, with zero KeepLast reasons observed", res.Notes) - } + require.True(t, res.Proved, "a declared KeepLast is a note, never fatal") + require.True(t, anyContains(res.Notes, "KeepLast"), + "from the definition alone, with zero KeepLast reasons observed") }) } } @@ -503,9 +445,7 @@ func TestProveCoverage_SkewTranslationAtWindowBoundary(t *testing.T) { sentinel := to res := proveCoverage(Header{StartedAt: from.Add(-time.Hour)}, polls, &sentinel, rt, def, from, to, 0) - if !res.Proved { - t.Fatalf("res = %+v, want proved: a constant clock skew must not itself read as a coverage gap", res) - } + require.True(t, res.Proved, "a constant clock skew must not itself read as a coverage gap") } // --- Override round-trip: one authority for the cadence --- @@ -524,9 +464,7 @@ func TestProveCoverage_OverrideRoundTrip(t *testing.T) { } defs := []Definition{{UID: "r1", Title: "R1", IntervalSeconds: 60}} rt, _, err := DeriveTimingsFromLog(h, defs) - if err != nil { - t.Fatalf("DeriveTimingsFromLog: %v", err) - } + require.NoError(t, err) var polls []Poll for ts := from; !ts.After(windowEnd); ts = ts.Add(120 * time.Second) { @@ -535,9 +473,8 @@ func TestProveCoverage_OverrideRoundTrip(t *testing.T) { sentinel := windowEnd res := proveCoverage(h, polls, &sentinel, rt["r1"], defs[0], from, windowEnd, 0) - if !res.Proved { - t.Fatalf("Proved = false, want true (maxGap must come from the recorded 120s cadence, not the 30s default): %+v", res) - } + require.True(t, res.Proved, + "maxGap must come from the recorded 120s cadence, not the 30s default") }) t.Run("faster override still catches a real recorder gap", func(t *testing.T) { @@ -548,9 +485,7 @@ func TestProveCoverage_OverrideRoundTrip(t *testing.T) { } defs := []Definition{{UID: "r1", Title: "R1", IntervalSeconds: 300}} rt, _, err := DeriveTimingsFromLog(h, defs) - if err != nil { - t.Fatalf("DeriveTimingsFromLog: %v", err) - } + require.NoError(t, err) var polls []Poll ts := from @@ -571,10 +506,8 @@ func TestProveCoverage_OverrideRoundTrip(t *testing.T) { sentinel := windowEnd res := proveCoverage(h, polls, &sentinel, rt["r1"], defs[0], from, windowEnd, 0) - if res.Reason != ReasonHeartbeatGap { - t.Fatalf("Reason = %q, want heartbeat_gap: if maxGap had been re-derived from the 300s definition instead of "+ - "the recorded 5s cadence, this 250s gap would pass silently — the fail-open direction", res.Reason) - } + require.Equal(t, ReasonHeartbeatGap, res.Reason, + "if maxGap had been re-derived from the 300s definition, this 250s gap would pass silently") }) } @@ -612,10 +545,8 @@ func TestProveCoverage_ZeroLastEvaluationWithoutPauseIsStale(t *testing.T) { sentinel := to res := proveCoverage(Header{StartedAt: from.Add(-time.Hour)}, polls, &sentinel, rt, def, from, to, 0) - if res.Reason != ReasonStaleEvaluation { - t.Fatalf("Reason = %q, want stale_evaluation: a zero lastEvaluation on a found, non-paused poll must fail "+ - "closed, not be silently skipped as if it were a legitimately paused observation", res.Reason) - } + require.Equal(t, ReasonStaleEvaluation, res.Reason, + "a zero lastEvaluation on a found, non-paused poll must fail closed") } // A lastEvaluation in the future of grafana_now (corrupted log) must fail closed. @@ -663,10 +594,8 @@ func TestProveCoverage_BoundaryGapWidensBySkewBound(t *testing.T) { sentinel := to res := proveCoverage(Header{StartedAt: from.Add(-time.Hour)}, polls, &sentinel, rt, def, from, to, 0) - if res.Reason != ReasonHeartbeatGap { - t.Fatalf("Reason = %q, want heartbeat_gap: the leading boundary segment sits at EXACTLY maxGap (60s) before "+ - "widening; the poll's own %s skew bound must push it past the threshold, not just the skew translation", res.Reason, bound) - } + require.Equal(t, ReasonHeartbeatGap, res.Reason, + "the poll's own %s skew bound must push it past the threshold", bound) } // --- Multi-failure contract --- @@ -698,16 +627,10 @@ func TestProveCoverage_MultipleFailuresReasonIsFirstButAllNoted(t *testing.T) { sentinel := to res := proveCoverage(Header{StartedAt: from.Add(-time.Hour)}, polls, &sentinel, rt, def, from, to, 0) - if res.Reason != ReasonPausedInWindow { - t.Fatalf("Reason = %q, want paused_in_window: the FIRST check to fail names the reason", res.Reason) - } - if !anyContains(res.Notes, "paused") { - t.Fatalf("Notes = %v, want a note about the pause", res.Notes) - } - if !anyContains(res.Notes, "no rule") { - t.Fatalf("Notes = %v, want a note about the absence too — a later failure must still be recorded, "+ - "not swallowed once Reason is already set", res.Notes) - } + require.Equal(t, ReasonPausedInWindow, res.Reason, "the FIRST check to fail names the reason") + require.True(t, anyContains(res.Notes, "paused")) + require.True(t, anyContains(res.Notes, "no rule"), + "a later failure must still be recorded, not swallowed once Reason is already set") } // --- Skipped rules --- @@ -728,9 +651,6 @@ func TestProveCoverage_SkippedRuleWithZeroPollsPinnedAsHeartbeatGap(t *testing.T sentinel := to res := proveCoverage(Header{StartedAt: from.Add(-time.Hour)}, nil, &sentinel, rt, def, from, to, 0) - if res.Reason != ReasonHeartbeatGap { - t.Fatalf("Reason = %q, want heartbeat_gap (pinned, not the desired end state): proveCoverage has no "+ - "'skipped' concept, so decide must handle a skipped rule's classification itself, before or "+ - "instead of calling this function", res.Reason) - } + require.Equal(t, ReasonHeartbeatGap, res.Reason, + "proveCoverage has no 'skipped' concept, so decide must handle a skipped rule's classification itself") } diff --git a/grafana-alertcheck/internal/gate/duration_test.go b/grafana-alertcheck/internal/gate/duration_test.go index ba3a1629a..6262c713e 100644 --- a/grafana-alertcheck/internal/gate/duration_test.go +++ b/grafana-alertcheck/internal/gate/duration_test.go @@ -3,6 +3,8 @@ package gate import ( "testing" "time" + + "github.com/stretchr/testify/require" ) func TestParsePromDuration(t *testing.T) { @@ -35,13 +37,8 @@ func TestParsePromDuration(t *testing.T) { } for _, c := range cases { got, err := ParsePromDuration(c.in) - if err != nil { - t.Errorf("ParsePromDuration(%q): unexpected error: %v", c.in, err) - continue - } - if got != c.want { - t.Errorf("ParsePromDuration(%q) = %v, want %v", c.in, got, c.want) - } + require.NoErrorf(t, err, "ParsePromDuration(%q)", c.in) + require.Equalf(t, c.want, got, "ParsePromDuration(%q)", c.in) } } @@ -60,8 +57,7 @@ func TestParsePromDuration_Errors(t *testing.T) { "carrot", // completely invalid } for _, in := range cases { - if _, err := ParsePromDuration(in); err == nil { - t.Errorf("ParsePromDuration(%q): expected an error, got none", in) - } + _, err := ParsePromDuration(in) + require.Errorf(t, err, "ParsePromDuration(%q): expected an error, got none", in) } } diff --git a/grafana-alertcheck/internal/gate/jsonreq_test.go b/grafana-alertcheck/internal/gate/jsonreq_test.go index bf3881e4b..72b446503 100644 --- a/grafana-alertcheck/internal/gate/jsonreq_test.go +++ b/grafana-alertcheck/internal/gate/jsonreq_test.go @@ -3,14 +3,14 @@ package gate import ( "encoding/json" "testing" + + "github.com/stretchr/testify/require" ) func rawMap(t *testing.T, jsonObj string) map[string]json.RawMessage { t.Helper() var m map[string]json.RawMessage - if err := json.Unmarshal([]byte(jsonObj), &m); err != nil { - t.Fatalf("rawMap: %v", err) - } + require.NoError(t, json.Unmarshal([]byte(jsonObj), &m)) return m } @@ -19,34 +19,24 @@ func TestReq(t *testing.T) { t.Run("present key decodes", func(t *testing.T) { var s string - if err := req(m, "present", &s); err != nil { - t.Fatalf("unexpected error: %v", err) - } - if s != "hello" { - t.Errorf("got %q, want hello", s) - } + require.NoError(t, req(m, "present", &s)) + require.Equal(t, "hello", s) }) t.Run("absent key errors", func(t *testing.T) { var s string - if err := req(m, "missing", &s); err == nil { - t.Fatalf("expected an error, got none") - } + require.Error(t, req(m, "missing", &s)) }) t.Run("wrong type errors", func(t *testing.T) { var s string - if err := req(m, "wrongtype", &s); err == nil { - t.Fatalf("expected an error, got none") - } + require.Error(t, req(m, "wrongtype", &s)) }) t.Run("explicit JSON null errors, never a zero value", func(t *testing.T) { var s string err := req(m, "nullval", &s) - if err == nil { - t.Fatalf("expected an error, got none (s=%q) — a null required field must not silently become a zero value", s) - } + require.Error(t, err, "a null required field must not silently become a zero value") }) } @@ -55,38 +45,24 @@ func TestOpt(t *testing.T) { t.Run("present key decodes", func(t *testing.T) { var s string - if err := opt(m, "present", &s); err != nil { - t.Fatalf("unexpected error: %v", err) - } - if s != "hello" { - t.Errorf("got %q, want hello", s) - } + require.NoError(t, opt(m, "present", &s)) + require.Equal(t, "hello", s) }) t.Run("absent key leaves dst untouched", func(t *testing.T) { s := "unchanged" - if err := opt(m, "missing", &s); err != nil { - t.Fatalf("unexpected error: %v", err) - } - if s != "unchanged" { - t.Errorf("got %q, want unchanged", s) - } + require.NoError(t, opt(m, "missing", &s)) + require.Equal(t, "unchanged", s) }) t.Run("wrong type errors", func(t *testing.T) { var s string - if err := opt(m, "wrongtype", &s); err == nil { - t.Fatalf("expected an error, got none") - } + require.Error(t, opt(m, "wrongtype", &s)) }) t.Run("explicit JSON null leaves dst at its zero value", func(t *testing.T) { var s string - if err := opt(m, "nullval", &s); err != nil { - t.Fatalf("unexpected error: %v", err) - } - if s != "" { - t.Errorf("got %q, want empty string", s) - } + require.NoError(t, opt(m, "nullval", &s)) + require.Equal(t, "", s) }) } diff --git a/grafana-alertcheck/internal/gate/log_test.go b/grafana-alertcheck/internal/gate/log_test.go index 179adb5fe..8ef3863e1 100644 --- a/grafana-alertcheck/internal/gate/log_test.go +++ b/grafana-alertcheck/internal/gate/log_test.go @@ -5,17 +5,18 @@ import ( "fmt" "os" "path/filepath" - "reflect" "strings" "sync" "testing" "time" + + "github.com/stretchr/testify/require" ) // Every time literal in this file is UTC and built with time.Date, so it // carries no monotonic reading and survives a JSON round trip byte-identical — -// which is what lets the round-trip tests below use reflect.DeepEqual on whole -// Poll values instead of comparing field by field. +// which is what lets the round-trip tests below compare whole Poll values +// instead of comparing field by field. var testNow = time.Date(2026, 8, 31, 9, 0, 0, 0, time.UTC) func testInstance(state State, reason, instanceLabel string) Instance { @@ -57,30 +58,20 @@ func TestLogReduceKeepsOnlyAbnormalInstances(t *testing.T) { p := NewReducer().Reduce("rule1", observation(testNow, rule)) - if !p.Found { - t.Fatalf("Found = false, want true") - } - if len(p.Abnormal) != 1 || p.Abnormal[0].Labels["instance"] != "b" { - t.Errorf("Abnormal = %+v, want only the firing instance b", p.Abnormal) - } - if want := map[string]int{"NoData": 1, "Error": 1}; !reflect.DeepEqual(p.Reasons, want) { - t.Errorf("Reasons = %v, want %v", p.Reasons, want) - } + require.True(t, p.Found) + require.Len(t, p.Abnormal, 1) + require.Equal(t, "b", p.Abnormal[0].Labels["instance"]) + require.Equal(t, map[string]int{"NoData": 1, "Error": 1}, p.Reasons) // The histogram is a verbatim copy of the response totals — raw keys, no // normalization. - if want := map[string]int{"alerting": 1, "normal": 2}; !reflect.DeepEqual(p.Histogram, want) { - t.Errorf("Histogram = %v, want %v", p.Histogram, want) - } + require.Equal(t, map[string]int{"alerting": 1, "normal": 2}, p.Histogram) // Rule-level state and health stay raw and unnormalized. - if p.State != "firing" || p.Health != "ok" { - t.Errorf("State/Health = %q/%q, want firing/ok", p.State, p.Health) - } - if p.Skew() != 1500*time.Millisecond || p.SkewBound() != 40*time.Millisecond || p.Latency() != 1800*time.Millisecond { - t.Errorf("durations = %s/%s/%s, want 1.5s/40ms/1.8s", p.Skew(), p.SkewBound(), p.Latency()) - } - if p.Reasons["MissingSeries"] != 0 { - t.Errorf("unexpected MissingSeries count") - } + require.Equal(t, "firing", p.State) + require.Equal(t, "ok", p.Health) + require.Equal(t, 1500*time.Millisecond, p.Skew()) + require.Equal(t, 40*time.Millisecond, p.SkewBound()) + require.Equal(t, 1800*time.Millisecond, p.Latency()) + require.Zero(t, p.Reasons["MissingSeries"]) } // A filtered response can hold several rules sharing one title, so the reducer @@ -94,9 +85,8 @@ func TestLogReduceSelectsRuleByUID(t *testing.T) { p := NewReducer().Reduce("ruleB", observation(testNow, first, second)) - if p.Health != "error" || len(p.Abnormal) != 1 { - t.Errorf("reduced the wrong rule: %+v", p) - } + require.Equal(t, "error", p.Health) + require.Len(t, p.Abnormal, 1) } func TestLogReduceRuleAbsentIsAuthoritative(t *testing.T) { @@ -104,20 +94,14 @@ func TestLogReduceRuleAbsentIsAuthoritative(t *testing.T) { p := NewReducer().Reduce("rule1", observation(testNow, other)) - if p.Found { - t.Errorf("Found = true, want false for a rule absent from an authoritative 2xx") - } - if p.RuleUID != "rule1" { - t.Errorf("RuleUID = %q, want rule1 — an absent rule is still attributed", p.RuleUID) - } + require.False(t, p.Found, "a rule absent from an authoritative 2xx") + require.Equal(t, "rule1", p.RuleUID, "an absent rule is still attributed") // The heartbeat still exists: a not-found poll is evidence that Grafana // answered at this time, which the coverage proof reads. - if !p.GrafanaNow.Equal(testNow) || p.Latency() == 0 { - t.Errorf("absent-rule poll lost its timing evidence: %+v", p) - } - if p.Health != "" || p.Abnormal != nil { - t.Errorf("absent-rule poll carries rule fields: %+v", p) - } + require.True(t, p.GrafanaNow.Equal(testNow)) + require.NotZero(t, p.Latency()) + require.Empty(t, p.Health) + require.Nil(t, p.Abnormal) } // An instance that leaves the abnormal set is resolved against the SAME @@ -175,19 +159,14 @@ func TestTransitionMarkersClearedVersusVanished(t *testing.T) { Instances: []Instance{testInstance(StateFiring, "", "b")}, } first := r.Reduce("rule1", observation(testNow, firing)) - if first.Cleared != nil || first.Vanished != nil { - t.Fatalf("first poll produced markers with no previous poll: %+v", first) - } + require.Nil(t, first.Cleared, "first poll produced cleared markers with no previous poll") + require.Nil(t, first.Vanished, "first poll produced vanished markers with no previous poll") next := StateRule{UID: "rule1", Health: "ok", State: "inactive", LastEvaluation: testNow, Instances: c.second} p := r.Reduce("rule1", observation(testNow.Add(30*time.Second), next)) - if !reflect.DeepEqual(p.Cleared, c.wantCleared) { - t.Errorf("Cleared = %q, want %q", p.Cleared, c.wantCleared) - } - if !reflect.DeepEqual(p.Vanished, c.wantVanished) { - t.Errorf("Vanished = %q, want %q", p.Vanished, c.wantVanished) - } + require.Equal(t, c.wantCleared, p.Cleared) + require.Equal(t, c.wantVanished, p.Vanished) }) } } @@ -204,15 +183,12 @@ func TestTransitionMarkersSurviveAnAbsentPoll(t *testing.T) { r.Reduce("rule1", observation(testNow, firing)) absent := r.Reduce("rule1", observation(testNow.Add(30*time.Second))) - if absent.Vanished != nil || absent.Cleared != nil { - t.Fatalf("an absent rule produced markers: %+v", absent) - } + require.Nil(t, absent.Vanished, "an absent rule produced vanished markers") + require.Nil(t, absent.Cleared, "an absent rule produced cleared markers") back := StateRule{UID: "rule1", Health: "ok", State: "inactive", LastEvaluation: testNow} p := r.Reduce("rule1", observation(testNow.Add(60*time.Second), back)) - if len(p.Vanished) != 1 { - t.Errorf("Vanished = %q, want the instance that disappeared across the absent poll", p.Vanished) - } + require.Len(t, p.Vanished, 1, "want the instance that disappeared across the absent poll") } func TestTransitionMarkersAreSortedAndPerRule(t *testing.T) { @@ -234,19 +210,14 @@ func TestTransitionMarkersAreSortedAndPerRule(t *testing.T) { clearedOne := StateRule{UID: "rule1", Health: "ok", State: "inactive", LastEvaluation: testNow} p := r.Reduce("rule1", observation(testNow.Add(time.Minute), clearedOne, ruleTwo)) - if len(p.Vanished) != 3 { - t.Fatalf("Vanished = %q, want 3 keys", p.Vanished) - } + require.Len(t, p.Vanished, 3) for i := 1; i < len(p.Vanished); i++ { - if p.Vanished[i-1] > p.Vanished[i] { - t.Errorf("Vanished is not sorted: %q", p.Vanished) - } + require.Less(t, p.Vanished[i-1], p.Vanished[i], "Vanished is not sorted: %q", p.Vanished) } // rule2's own abnormal set is untouched by rule1's transitions. q := r.Reduce("rule2", observation(testNow.Add(time.Minute), clearedOne, ruleTwo)) - if q.Cleared != nil || q.Vanished != nil { - t.Errorf("rule2 picked up rule1's transitions: %+v", q) - } + require.Nil(t, q.Cleared, "rule2 picked up rule1's transitions") + require.Nil(t, q.Vanished, "rule2 picked up rule1's transitions") } // The reduction depends on the state endpoint returning normal instances. If it @@ -267,22 +238,14 @@ func TestLogVerifyNormalInstancesVisible(t *testing.T) { for _, c := range cases { t.Run(c.fixture, func(t *testing.T) { rules, err := ParseState(readFixture(t, c.fixture)) - if err != nil { - t.Fatalf("ParseState: %v", err) - } + require.NoError(t, err) err = VerifyNormalInstancesVisible(rules) if c.wantError { - if err == nil { - t.Fatalf("VerifyNormalInstancesVisible: want an error, got nil") - } - if !strings.Contains(err.Error(), "no longer returns normal instances") { - t.Errorf("error does not say the endpoint stopped returning normal instances: %v", err) - } + require.Error(t, err) + require.Contains(t, err.Error(), "no longer returns normal instances") return } - if err != nil { - t.Fatalf("VerifyNormalInstancesVisible: unexpected error: %v", err) - } + require.NoError(t, err) }) } } @@ -310,9 +273,7 @@ func TestLogVerifyNormalInstancesVisibleVocabularies(t *testing.T) { Instances: []Instance{testInstance(StateFiring, "", "b")}, }} err := VerifyNormalInstancesVisible(rules) - if (err != nil) != c.wantError { - t.Errorf("VerifyNormalInstancesVisible: error = %v, want error = %v", err, c.wantError) - } + require.Equal(t, c.wantError, err != nil) }) } } @@ -328,37 +289,25 @@ func TestLogModeCadenceComesFromTheHeader(t *testing.T) { h.Rules[0].PollEverySeconds = 5 // an operator override far tighter than the default 150s rt, _, err := DeriveTimingsFromLog(h, defs) - if err != nil { - t.Fatalf("DeriveTimingsFromLog: %v", err) - } + require.NoError(t, err) got := rt["rule1"] - if got.pollEvery != 5*time.Second { - t.Errorf("pollEvery = %s, want the header's 5s, not the default 150s", got.pollEvery) - } + require.Equal(t, 5*time.Second, got.pollEvery, "the header's 5s, not the default 150s") // maxGap and healthGrace follow the recorded cadence; without this a 250s // hole in a log recorded at 5s would pass silently. - if got.maxGap != 10*time.Second { - t.Errorf("maxGap = %s, want 10s (2 x the recorded cadence)", got.maxGap) - } - if got.healthGrace != 300*time.Second { - t.Errorf("healthGrace = %s, want 300s (max(maxGap, interval))", got.healthGrace) - } + require.Equal(t, 10*time.Second, got.maxGap) + require.Equal(t, 300*time.Second, got.healthGrace) // evalStaleAfter is a rule fact, so it stays 2 x intervalSeconds from the // definitions regardless of how often the gate polled. - if got.evalStaleAfter != 600*time.Second { - t.Errorf("evalStaleAfter = %s, want 600s from the definition's interval", got.evalStaleAfter) - } + require.Equal(t, 600*time.Second, got.evalStaleAfter) // A log that cannot say how often it was written cannot have its coverage // proved, and neither can one naming a rule that no longer resolves. missingCadence := testHeader() missingCadence.Rules[0].PollEverySeconds = 0 - if _, _, err := DeriveTimingsFromLog(missingCadence, defs); err == nil { - t.Errorf("a header with no recorded cadence was accepted") - } - if _, _, err := DeriveTimingsFromLog(testHeader(), nil); err == nil { - t.Errorf("a header naming an unresolvable rule was accepted") - } + require.Error(t, func() error { _, _, err := DeriveTimingsFromLog(missingCadence, defs); return err }(), + "a header with no recorded cadence was accepted") + require.Error(t, func() error { _, _, err := DeriveTimingsFromLog(testHeader(), nil); return err }(), + "a header naming an unresolvable rule was accepted") // A duplicated UID must not resolve last-one-wins: the slower duplicate // would widen maxGap, which is fail-open through log corruption alone. @@ -366,9 +315,8 @@ func TestLogModeCadenceComesFromTheHeader(t *testing.T) { slower := duplicated.Rules[0] slower.PollEverySeconds = 600 duplicated.Rules = append(duplicated.Rules, slower) - if _, _, err := DeriveTimingsFromLog(duplicated, defs); err == nil { - t.Errorf("a header naming one rule twice was accepted") - } + require.Error(t, func() error { _, _, err := DeriveTimingsFromLog(duplicated, defs); return err }(), + "a header naming one rule twice was accepted") } // watch polls a fleet concurrently through one Reducer, so the marker @@ -398,9 +346,8 @@ func TestLogReduceIsSafeForConcurrentUse(t *testing.T) { // transition — the concurrency must not corrupt the per-rule state either. for _, rule := range rules { p := r.Reduce(rule.UID, obs) - if p.Cleared != nil || p.Vanished != nil { - t.Errorf("rule %s: markers after concurrent reduction: %+v", rule.UID, p) - } + require.Nilf(t, p.Cleared, "rule %s: cleared markers after concurrent reduction", rule.UID) + require.Nilf(t, p.Vanished, "rule %s: vanished markers after concurrent reduction", rule.UID) } } @@ -409,30 +356,18 @@ func TestLogReduceIsSafeForConcurrentUse(t *testing.T) { func TestLogPollOmitsTheZeroEvaluationTime(t *testing.T) { absent := NewReducer().Reduce("rule1", observation(testNow)) b, err := json.Marshal(pollRecord{Type: RecordPoll, Poll: absent}) - if err != nil { - t.Fatalf("marshal: %v", err) - } - if strings.Contains(string(b), "0001-01-01") { - t.Errorf("a not-found poll wrote the zero time: %s", b) - } - if strings.Contains(string(b), "last_evaluation") { - t.Errorf("a not-found poll wrote last_evaluation at all: %s", b) - } + require.NoError(t, err) + require.NotContains(t, string(b), "0001-01-01") + require.NotContains(t, string(b), "last_evaluation") // A real evaluation time still round-trips. found := StateRule{UID: "rule1", Health: "ok", State: "inactive", LastEvaluation: testNow} p := NewReducer().Reduce("rule1", observation(testNow, found)) b, err = json.Marshal(pollRecord{Type: RecordPoll, Poll: p}) - if err != nil { - t.Fatalf("marshal: %v", err) - } + require.NoError(t, err) var back pollRecord - if err := json.Unmarshal(b, &back); err != nil { - t.Fatalf("unmarshal: %v", err) - } - if !back.LastEvaluation.Equal(testNow) { - t.Errorf("last_evaluation = %s, want %s", back.LastEvaluation, testNow) - } + require.NoError(t, json.Unmarshal(b, &back)) + require.True(t, back.LastEvaluation.Equal(testNow)) } func testHeader() Header { @@ -452,9 +387,7 @@ func newTestWriter(t *testing.T, path string) (*Writer, *fakeClock) { t.Helper() clock := newFakeClock(testNow) w, err := NewWriter(path, clock) - if err != nil { - t.Fatalf("NewWriter: %v", err) - } + require.NoError(t, err) return w, clock } @@ -463,9 +396,7 @@ func TestWriterReadLogRoundTrip(t *testing.T) { w, clock := newTestWriter(t, path) h := testHeader() - if err := w.WriteHeader(h); err != nil { - t.Fatalf("WriteHeader: %v", err) - } + require.NoError(t, w.WriteHeader(h)) r := NewReducer() firing := StateRule{ @@ -479,35 +410,21 @@ func TestWriterReadLogRoundTrip(t *testing.T) { r.Reduce("rule1", observation(testNow.Add(time.Minute), cleared)), } for _, p := range want { - if err := w.WritePoll(p); err != nil { - t.Fatalf("WritePoll: %v", err) - } + require.NoError(t, w.WritePoll(p)) } clock.Advance(2 * time.Minute) - if err := w.Stop(); err != nil { - t.Fatalf("Stop: %v", err) - } + require.NoError(t, w.Stop()) gotHeader, gotPolls, sentinel, err := ReadLog(path) - if err != nil { - t.Fatalf("ReadLog: %v", err) - } + require.NoError(t, err) h.SchemaVersion = LogSchemaVersion // WriteHeader stamps it - if !reflect.DeepEqual(gotHeader, h) { - t.Errorf("header round trip:\n got %+v\nwant %+v", gotHeader, h) - } - if !reflect.DeepEqual(gotPolls, want) { - t.Errorf("poll round trip:\n got %+v\nwant %+v", gotPolls, want) - } - if sentinel == nil { - t.Fatalf("sentinel is nil after Stop") - } + require.Equal(t, h, gotHeader, "header round trip") + require.Equal(t, want, gotPolls, "poll round trip") + require.NotNil(t, sentinel, "sentinel is nil after Stop") // Stop stamps the recorder's own stop time and makes no comparison // against `to` — watch never knows it. - if !sentinel.Equal(testNow.Add(2 * time.Minute)) { - t.Errorf("sentinel = %s, want the writer's stop time %s", sentinel, testNow.Add(2*time.Minute)) - } + require.True(t, sentinel.Equal(testNow.Add(2*time.Minute))) } // The log is append-only. A second run against the same path must never @@ -515,44 +432,25 @@ func TestWriterReadLogRoundTrip(t *testing.T) { func TestWriterAppendsAndNeverTruncates(t *testing.T) { path := filepath.Join(t.TempDir(), "log.jsonl") w, _ := newTestWriter(t, path) - if err := w.WriteHeader(testHeader()); err != nil { - t.Fatalf("WriteHeader: %v", err) - } - if err := w.WritePoll(Poll{RuleUID: "rule1", Found: true, GrafanaNow: testNow}); err != nil { - t.Fatalf("WritePoll: %v", err) - } - if err := w.Close(); err != nil { - t.Fatalf("Close: %v", err) - } + require.NoError(t, w.WriteHeader(testHeader())) + require.NoError(t, w.WritePoll(Poll{RuleUID: "rule1", Found: true, GrafanaNow: testNow})) + require.NoError(t, w.Close()) before, err := os.ReadFile(path) - if err != nil { - t.Fatalf("read: %v", err) - } + require.NoError(t, err) // The handoff: the parent wrote the header and closed; the child // reopens the same path and appends without a second header. child, _ := newTestWriter(t, path) - if err := child.WritePoll(Poll{RuleUID: "rule1", Found: true, GrafanaNow: testNow.Add(time.Minute)}); err != nil { - t.Fatalf("child WritePoll: %v", err) - } - if err := child.Stop(); err != nil { - t.Fatalf("child Stop: %v", err) - } + require.NoError(t, child.WritePoll(Poll{RuleUID: "rule1", Found: true, GrafanaNow: testNow.Add(time.Minute)})) + require.NoError(t, child.Stop()) after, err := os.ReadFile(path) - if err != nil { - t.Fatalf("read: %v", err) - } - if !strings.HasPrefix(string(after), string(before)) { - t.Fatalf("reopening the log rewrote earlier records:\n%s", after) - } + require.NoError(t, err) + require.True(t, strings.HasPrefix(string(after), string(before)), "reopening the log rewrote earlier records") _, polls, sentinel, err := ReadLog(path) - if err != nil { - t.Fatalf("ReadLog: %v", err) - } - if len(polls) != 2 || sentinel == nil { - t.Errorf("got %d polls, sentinel %v; want 2 polls and a sentinel", len(polls), sentinel) - } + require.NoError(t, err) + require.Len(t, polls, 2) + require.NotNil(t, sentinel) } // Two recorders on one log means one of them is recording a window nobody @@ -570,67 +468,41 @@ func TestWriterSecondWriterFails(t *testing.T) { select { case err := <-done: - if err == nil { - t.Fatalf("a second writer took the lock") - } - if !strings.Contains(err.Error(), "another writer") { - t.Errorf("error does not name the conflict: %v", err) - } + require.Error(t, err, "a second writer took the lock") + require.Contains(t, err.Error(), "another writer") case <-time.After(5 * time.Second): - t.Fatalf("the second NewWriter blocked instead of failing immediately") + require.Fail(t, "the second NewWriter blocked instead of failing immediately") } } func TestWriterHeaderRefusesANonEmptyLog(t *testing.T) { path := filepath.Join(t.TempDir(), "log.jsonl") w, _ := newTestWriter(t, path) - if err := w.WriteHeader(testHeader()); err != nil { - t.Fatalf("WriteHeader: %v", err) - } - if err := w.WriteHeader(testHeader()); err == nil { - t.Fatalf("a second header was accepted") - } - if err := w.Close(); err != nil { - t.Fatalf("Close: %v", err) - } + require.NoError(t, w.WriteHeader(testHeader())) + require.Error(t, w.WriteHeader(testHeader()), "a second header was accepted") + require.NoError(t, w.Close()) reopened, _ := newTestWriter(t, path) defer reopened.Close() - if err := reopened.WriteHeader(testHeader()); err == nil { - t.Fatalf("a header was accepted on a non-empty log") - } + require.Error(t, reopened.WriteHeader(testHeader()), "a header was accepted on a non-empty log") } func TestSentinelStopIsIdempotentAndLast(t *testing.T) { path := filepath.Join(t.TempDir(), "log.jsonl") w, _ := newTestWriter(t, path) - if err := w.WriteHeader(testHeader()); err != nil { - t.Fatalf("WriteHeader: %v", err) - } - if err := w.Stop(); err != nil { - t.Fatalf("Stop: %v", err) - } + require.NoError(t, w.WriteHeader(testHeader())) + require.NoError(t, w.Stop()) // watch reaches Stop from both a signal handler and a defer; a second // sentinel would be indistinguishable from a second writer. - if err := w.Stop(); err != nil { - t.Errorf("second Stop: %v", err) - } + require.NoError(t, w.Stop()) // Nothing may be appended after the sentinel — not even by the same writer. - if err := w.WritePoll(Poll{RuleUID: "rule1"}); err == nil { - t.Errorf("WritePoll after Stop was accepted") - } + require.Error(t, w.WritePoll(Poll{RuleUID: "rule1"}), "WritePoll after Stop was accepted") b, err := os.ReadFile(path) - if err != nil { - t.Fatalf("read: %v", err) - } + require.NoError(t, err) lines := strings.Split(strings.TrimSuffix(string(b), "\n"), "\n") - if len(lines) != 2 { - t.Fatalf("got %d lines, want header + one sentinel:\n%s", len(lines), b) - } - if !strings.Contains(lines[1], `"type":"stopped"`) { - t.Errorf("last line is not the sentinel: %s", lines[1]) - } + require.Len(t, lines, 2, "header + one sentinel") + require.Contains(t, lines[1], `"type":"stopped"`) } // Close is the parent's handoff path: a sentinel there would tell check the @@ -638,23 +510,13 @@ func TestSentinelStopIsIdempotentAndLast(t *testing.T) { func TestSentinelCloseWritesNone(t *testing.T) { path := filepath.Join(t.TempDir(), "log.jsonl") w, _ := newTestWriter(t, path) - if err := w.WriteHeader(testHeader()); err != nil { - t.Fatalf("WriteHeader: %v", err) - } - if err := w.Close(); err != nil { - t.Fatalf("Close: %v", err) - } + require.NoError(t, w.WriteHeader(testHeader())) + require.NoError(t, w.Close()) _, polls, sentinel, err := ReadLog(path) - if err != nil { - t.Fatalf("ReadLog: %v", err) - } - if sentinel != nil { - t.Errorf("Close wrote a sentinel: %s", sentinel) - } - if polls != nil { - t.Errorf("polls = %+v, want none", polls) - } + require.NoError(t, err) + require.Nil(t, sentinel, "Close wrote a sentinel") + require.Nil(t, polls) } // An unfinished recording reads cleanly with a nil sentinel — ReadLog reports @@ -664,23 +526,14 @@ func TestSentinelCloseWritesNone(t *testing.T) { func TestReadLogWithoutASentinel(t *testing.T) { path := filepath.Join(t.TempDir(), "log.jsonl") w, _ := newTestWriter(t, path) - if err := w.WriteHeader(testHeader()); err != nil { - t.Fatalf("WriteHeader: %v", err) - } - if err := w.WritePoll(Poll{RuleUID: "rule1", Found: true, GrafanaNow: testNow}); err != nil { - t.Fatalf("WritePoll: %v", err) - } - if err := w.Close(); err != nil { - t.Fatalf("Close: %v", err) - } + require.NoError(t, w.WriteHeader(testHeader())) + require.NoError(t, w.WritePoll(Poll{RuleUID: "rule1", Found: true, GrafanaNow: testNow})) + require.NoError(t, w.Close()) _, polls, sentinel, err := ReadLog(path) - if err != nil { - t.Fatalf("ReadLog: %v", err) - } - if sentinel != nil || len(polls) != 1 { - t.Errorf("got %d polls, sentinel %v; want 1 poll and no sentinel", len(polls), sentinel) - } + require.NoError(t, err) + require.Nil(t, sentinel) + require.Len(t, polls, 1) } // The read rules are deliberately the crudest possible: any unparseable @@ -691,23 +544,17 @@ func TestReadLogRejectsBadLogs(t *testing.T) { h := testHeader() h.SchemaVersion = version b, err := json.Marshal(headerRecord{Type: RecordHeader, Header: h}) - if err != nil { - t.Fatalf("marshal header: %v", err) - } + require.NoError(t, err, "marshal header") return string(b) } poll := func() string { b, err := json.Marshal(pollRecord{Type: RecordPoll, Poll: Poll{RuleUID: "rule1", Found: true, GrafanaNow: testNow}}) - if err != nil { - t.Fatalf("marshal poll: %v", err) - } + require.NoError(t, err, "marshal poll") return string(b) } sentinel := func() string { b, err := json.Marshal(stoppedRecord{Type: RecordStopped, At: testNow}) - if err != nil { - t.Fatalf("marshal sentinel: %v", err) - } + require.NoError(t, err, "marshal sentinel") return string(b) } @@ -758,25 +605,17 @@ func TestReadLogRejectsBadLogs(t *testing.T) { for _, c := range cases { t.Run(c.name, func(t *testing.T) { path := filepath.Join(t.TempDir(), "log.jsonl") - if err := os.WriteFile(path, []byte(c.content), 0o600); err != nil { - t.Fatalf("write: %v", err) - } + require.NoError(t, os.WriteFile(path, []byte(c.content), 0o600)) _, _, _, err := ReadLog(path) - if err == nil { - t.Fatalf("ReadLog: want an error, got nil") - } - if !strings.Contains(err.Error(), c.wantIn) { - t.Errorf("error %q does not contain %q", err, c.wantIn) - } + require.Error(t, err) + require.Contains(t, err.Error(), c.wantIn) }) } } func TestReadLogMissingFile(t *testing.T) { _, _, _, err := ReadLog(filepath.Join(t.TempDir(), "absent.jsonl")) - if err == nil { - t.Fatalf("ReadLog on a missing log: want an error, got nil") - } + require.Error(t, err) } // Per-poll log size must not grow across polls on a high-cardinality @@ -786,69 +625,46 @@ func TestReadLogMissingFile(t *testing.T) { func TestLogSizeIsFlatAcrossPollsOnAHighCardinalityRule(t *testing.T) { body := synthesizeHighCardinalityState(t, 1, 2445) rules, err := ParseState(body) - if err != nil { - t.Fatalf("ParseState: %v", err) - } - if len(rules[0].Instances) != 2446 { - t.Fatalf("got %d instances, want 2446", len(rules[0].Instances)) - } + require.NoError(t, err) + require.Len(t, rules[0].Instances, 2446) uid := rules[0].UID path := filepath.Join(t.TempDir(), "log.jsonl") w, _ := newTestWriter(t, path) - if err := w.WriteHeader(testHeader()); err != nil { - t.Fatalf("WriteHeader: %v", err) - } + require.NoError(t, w.WriteHeader(testHeader())) r := NewReducer() var sizes []int64 // Measure from the end of the header line, so sizes[0] is the first poll // record alone rather than the header plus it. info, err := os.Stat(path) - if err != nil { - t.Fatalf("stat: %v", err) - } + require.NoError(t, err) previous := info.Size() for i := range 5 { p := r.Reduce(uid, observation(testNow.Add(time.Duration(i)*30*time.Second), rules[0])) - if len(p.Abnormal) != 1 { - t.Fatalf("poll %d: Abnormal = %d instances, want the single firing one", i, len(p.Abnormal)) - } - if got := p.Abnormal[0].Labels["instance"]; got != "alerting-0" { - t.Fatalf("poll %d: the firing instance lost its identity: %q", i, got) - } - if err := w.WritePoll(p); err != nil { - t.Fatalf("WritePoll: %v", err) - } + require.Lenf(t, p.Abnormal, 1, "poll %d", i) + require.Equalf(t, "alerting-0", p.Abnormal[0].Labels["instance"], "poll %d: the firing instance lost its identity", i) + require.NoError(t, w.WritePoll(p)) info, err := os.Stat(path) - if err != nil { - t.Fatalf("stat: %v", err) - } + require.NoError(t, err) sizes = append(sizes, info.Size()-previous) previous = info.Size() } for i := 1; i < len(sizes); i++ { - if sizes[i] != sizes[0] { - t.Errorf("per-poll size grew across polls: %v", sizes) - } + require.Equal(t, sizes[0], sizes[i], "per-poll size grew across polls: %v", sizes) } // One firing instance among 2446 costs a few hundred bytes, against the // ~600 KB the unreduced response carries. - if sizes[0] > 2048 { - t.Errorf("per-poll size %d bytes is not a reduction of a %d-byte response", sizes[0], len(body)) - } + require.LessOrEqual(t, sizes[0], int64(2048)) // When the firing instance clears, the record collapses further and the // transition is still attributed. rules[0].Instances[0].State = StateNormal p := r.Reduce(uid, observation(testNow.Add(5*30*time.Second), rules[0])) - if len(p.Cleared) != 1 || len(p.Abnormal) != 0 { - t.Errorf("cleared poll = %+v, want exactly one cleared key and no abnormal instances", p) - } - if err := w.Stop(); err != nil { - t.Fatalf("Stop: %v", err) - } + require.Len(t, p.Cleared, 1) + require.Empty(t, p.Abnormal) + require.NoError(t, w.Stop()) } // The log must stay readable by anything that reads JSONL, one flat object per @@ -857,39 +673,22 @@ func TestLogSizeIsFlatAcrossPollsOnAHighCardinalityRule(t *testing.T) { func TestLogRecordsAreFlatOneLineObjects(t *testing.T) { path := filepath.Join(t.TempDir(), "log.jsonl") w, _ := newTestWriter(t, path) - if err := w.WriteHeader(testHeader()); err != nil { - t.Fatalf("WriteHeader: %v", err) - } - if err := w.WritePoll(Poll{RuleUID: "rule1", Found: true, GrafanaNow: testNow}); err != nil { - t.Fatalf("WritePoll: %v", err) - } - if err := w.Stop(); err != nil { - t.Fatalf("Stop: %v", err) - } + require.NoError(t, w.WriteHeader(testHeader())) + require.NoError(t, w.WritePoll(Poll{RuleUID: "rule1", Found: true, GrafanaNow: testNow})) + require.NoError(t, w.Stop()) b, err := os.ReadFile(path) - if err != nil { - t.Fatalf("read: %v", err) - } + require.NoError(t, err) lines := strings.Split(strings.TrimSuffix(string(b), "\n"), "\n") wantTypes := []RecordType{RecordHeader, RecordPoll, RecordStopped} - if len(lines) != len(wantTypes) { - t.Fatalf("got %d lines, want %d:\n%s", len(lines), len(wantTypes), b) - } + require.Len(t, lines, len(wantTypes)) for i, line := range lines { var m map[string]json.RawMessage - if err := json.Unmarshal([]byte(line), &m); err != nil { - t.Fatalf("line %d is not one JSON object: %v", i+1, err) - } + require.NoErrorf(t, json.Unmarshal([]byte(line), &m), "line %d is not one JSON object", i+1) var gotType RecordType - if err := json.Unmarshal(m["type"], &gotType); err != nil { - t.Fatalf("line %d has no type tag: %v", i+1, err) - } - if gotType != wantTypes[i] { - t.Errorf("line %d type = %q, want %q", i+1, gotType, wantTypes[i]) - } - if _, nested := m["header"]; nested { - t.Errorf("line %d wraps its payload instead of being flat: %s", i+1, line) - } + require.NoErrorf(t, json.Unmarshal(m["type"], &gotType), "line %d has no type tag", i+1) + require.Equalf(t, wantTypes[i], gotType, "line %d type", i+1) + _, nested := m["header"] + require.Falsef(t, nested, "line %d wraps its payload instead of being flat", i+1) } } diff --git a/grafana-alertcheck/internal/gate/parse_ruler_test.go b/grafana-alertcheck/internal/gate/parse_ruler_test.go index b225ffc2c..b60dffa68 100644 --- a/grafana-alertcheck/internal/gate/parse_ruler_test.go +++ b/grafana-alertcheck/internal/gate/parse_ruler_test.go @@ -3,117 +3,83 @@ package gate import ( "testing" "time" + + "github.com/stretchr/testify/require" ) func TestParseDefinitions_RulerRules(t *testing.T) { defs, err := ParseDefinitions(readFixture(t, "ruler_rules.json")) - if err != nil { - t.Fatalf("ParseDefinitions: unexpected error: %v", err) - } + require.NoError(t, err) byUID := map[string]Definition{} for _, d := range defs { - if d.Kind != KindGrafanaManaged { - t.Errorf("rule %q: Kind = %v, want KindGrafanaManaged", d.UID, d.Kind) - } + require.Equalf(t, KindGrafanaManaged, d.Kind, "rule %q: Kind", d.UID) byUID[d.UID] = d } // The real 2-way duplicate title: same folder, same group, same title, // distinct UIDs — only uid: can tell them apart. a, ok := byUID["rule0000006a"] - if !ok { - t.Fatalf("missing rule0000006a") - } + require.True(t, ok, "missing rule0000006a") b, ok := byUID["rule0000006b"] - if !ok { - t.Fatalf("missing rule0000006b") - } - if a.Title != b.Title || a.Folder != b.Folder || a.Group != b.Group { - t.Errorf("duplicate-title pair should share Title/Folder/Group: a=%+v b=%+v", a, b) - } - if a.UID == b.UID { - t.Errorf("duplicate-title pair should have distinct UIDs") - } + require.True(t, ok, "missing rule0000006b") + require.Equal(t, a.Title, b.Title, "duplicate-title pair should share Title") + require.Equal(t, a.Folder, b.Folder, "duplicate-title pair should share Folder") + require.Equal(t, a.Group, b.Group, "duplicate-title pair should share Group") + require.NotEqual(t, a.UID, b.UID, "duplicate-title pair should have distinct UIDs") // The 3 real paused rules. pausedUIDs := []string{"rule0000002", "rule0000007", "rule0000008"} for _, uid := range pausedUIDs { d, ok := byUID[uid] - if !ok { - t.Fatalf("missing paused rule %q", uid) - } - if !d.IsPaused { - t.Errorf("rule %q: IsPaused = false, want true", uid) - } + require.True(t, ok, "missing paused rule %q", uid) + require.Truef(t, d.IsPaused, "rule %q: IsPaused = false, want true", uid) } // for:1d and the derived for:1w rule. dayRule, ok := byUID["rule0000009"] - if !ok || dayRule.For != 24*time.Hour { - t.Fatalf("rule0000009: For = %v, want 24h (ok=%v)", dayRule.For, ok) - } + require.Truef(t, ok, "missing rule0000009") + require.Equal(t, 24*time.Hour, dayRule.For) weekRule, ok := byUID["rule0000010"] - if !ok || weekRule.For != 7*24*time.Hour { - t.Fatalf("rule0000010: For = %v, want 168h (ok=%v)", weekRule.For, ok) - } + require.Truef(t, ok, "missing rule0000010") + require.Equal(t, 7*24*time.Hour, weekRule.For) // Identity shared with testdata/state_paused.json. shared := byUID["rule0000002"] - if shared.FolderUID != "folder0000002" { - t.Errorf("rule0000002: FolderUID = %q, want folder0000002", shared.FolderUID) - } + require.Equal(t, "folder0000002", shared.FolderUID) } func TestParseDefinitions_DatasourceManaged(t *testing.T) { defs, err := ParseDefinitions(readFixture(t, "ruler_datasource_managed.json")) - if err != nil { - t.Fatalf("ParseDefinitions: unexpected error: %v", err) - } - if len(defs) != 1 { - t.Fatalf("got %d definitions, want 1", len(defs)) - } - if defs[0].Kind != KindDatasourceManaged { - t.Errorf("Kind = %v, want KindDatasourceManaged", defs[0].Kind) - } - if defs[0].For != 5*time.Minute { - t.Errorf("For = %v, want 5m", defs[0].For) - } + require.NoError(t, err) + require.Len(t, defs, 1) + require.Equal(t, KindDatasourceManaged, defs[0].Kind) + require.Equal(t, 5*time.Minute, defs[0].For) // A datasource-managed rule has no uid in this shape; its only identity // is the Prometheus "alert" name — a synthetic UID would be invented // shape, and an empty Title would make Resolve's refusal-by-name // unreachable. - if defs[0].Title != "ExampleTargetDown" { - t.Errorf("Title = %q, want ExampleTargetDown", defs[0].Title) - } - if defs[0].UID != "" { - t.Errorf("UID = %q, want empty (this shape has no uid)", defs[0].UID) - } + require.Equal(t, "ExampleTargetDown", defs[0].Title) + require.Empty(t, defs[0].UID, "this shape has no uid") } func TestParseDefinitions_Recording(t *testing.T) { defs, err := ParseDefinitions(readFixture(t, "ruler_recording.json")) - if err != nil { - t.Fatalf("ParseDefinitions: unexpected error: %v", err) - } - if len(defs) != 1 { - t.Fatalf("got %d definitions, want 1", len(defs)) - } + require.NoError(t, err) + require.Len(t, defs, 1) d := defs[0] - if d.Kind != KindRecording { - t.Errorf("Kind = %v, want KindRecording", d.Kind) - } - if d.UID != "rule0000011" { - t.Errorf("UID = %q, want rule0000011", d.UID) - } + require.Equal(t, KindRecording, d.Kind) + require.Equal(t, "rule0000011", d.UID) // The fixture deliberately omits no_data_state/exec_err_state/is_paused/ // intervalSeconds/namespace_uid — alerting-only concepts a recording // rule may not carry. Requiring them would brick ParseDefinitions for // every named rule in the same response over one recording rule // elsewhere in the fleet; they must come back as zero values, not errors. - if d.NoDataState != "" || d.ExecErrState != "" || d.IsPaused || d.IntervalSeconds != 0 || d.FolderUID != "" { - t.Errorf("expected zero-valued alert-only fields for a recording rule, got %+v", d) - } + require.Empty(t, d.NoDataState) + require.Empty(t, d.ExecErrState) + require.False(t, d.IsPaused) + require.Zero(t, d.IntervalSeconds) + require.Empty(t, d.FolderUID) } // A datasource-managed rule with no alert/record name must fail parsing. diff --git a/grafana-alertcheck/internal/gate/parse_state_test.go b/grafana-alertcheck/internal/gate/parse_state_test.go index 4ccca1391..9e2ed899c 100644 --- a/grafana-alertcheck/internal/gate/parse_state_test.go +++ b/grafana-alertcheck/internal/gate/parse_state_test.go @@ -1,23 +1,20 @@ package gate import ( - "bytes" "encoding/json" "fmt" - "maps" "os" "path/filepath" - "strings" "testing" "time" + + "github.com/stretchr/testify/require" ) func readFixture(t *testing.T, name string) []byte { t.Helper() b, err := os.ReadFile(filepath.Join("testdata", name)) - if err != nil { - t.Fatalf("reading fixture %s: %v", name, err) - } + require.NoErrorf(t, err, "reading fixture %s", name) return b } @@ -33,24 +30,15 @@ func TestParseState_HappyPaths(t *testing.T) { wantRules: 1, wantInstances: 1, checkFirst: func(t *testing.T, r StateRule) { - if r.UID != "rule0000001" { - t.Errorf("UID = %q, want rule0000001", r.UID) - } - if r.Folder != "ExampleTeam" || r.Group != "Example Service - Prod" { - t.Errorf("Folder/Group = %q/%q, want ExampleTeam/Example Service - Prod", r.Folder, r.Group) - } - if r.Health != "ok" || r.State != "inactive" { - t.Errorf("Health/State = %q/%q, want ok/inactive", r.Health, r.State) - } - if r.Interval.Seconds() != 60 { - t.Errorf("Interval = %v, want 60s", r.Interval) - } - if r.IsPaused { - t.Errorf("IsPaused = true, want false") - } - if len(r.Instances) != 1 || r.Instances[0].State != StateNormal { - t.Fatalf("Instances = %+v, want one normal instance", r.Instances) - } + require.Equal(t, "rule0000001", r.UID) + require.Equal(t, "ExampleTeam", r.Folder) + require.Equal(t, "Example Service - Prod", r.Group) + require.Equal(t, "ok", r.Health) + require.Equal(t, "inactive", r.State) + require.Equal(t, float64(60), r.Interval.Seconds()) + require.False(t, r.IsPaused) + require.Len(t, r.Instances, 1) + require.Equal(t, StateNormal, r.Instances[0].State) inst := r.Instances[0] wantLabels := map[string]string{ @@ -62,19 +50,11 @@ func TestParseState_HappyPaths(t *testing.T) { "severity": "critical", "team": "example-team", } - if !maps.Equal(inst.Labels, wantLabels) { - t.Errorf("Labels = %+v, want %+v", inst.Labels, wantLabels) - } + require.Equal(t, wantLabels, inst.Labels) wantActiveAt, err := time.Parse(time.RFC3339, "2026-08-31T08:02:50Z") - if err != nil { - t.Fatalf("test setup: %v", err) - } - if !inst.ActiveAt.Equal(wantActiveAt) { - t.Errorf("ActiveAt = %v, want %v", inst.ActiveAt, wantActiveAt) - } - if inst.Value != "" { - t.Errorf("Value = %q, want empty string", inst.Value) - } + require.NoError(t, err, "test setup") + require.True(t, inst.ActiveAt.Equal(wantActiveAt)) + require.Empty(t, inst.Value) }, }, { @@ -82,15 +62,10 @@ func TestParseState_HappyPaths(t *testing.T) { wantRules: 1, wantInstances: 0, checkFirst: func(t *testing.T, r StateRule) { - if !r.IsPaused { - t.Errorf("IsPaused = false, want true") - } - if !r.LastEvaluation.IsZero() { - t.Errorf("LastEvaluation = %v, want zero time", r.LastEvaluation) - } - if r.Health != "ok" || r.State != "inactive" { - t.Errorf("Health/State = %q/%q, want ok/inactive", r.Health, r.State) - } + require.True(t, r.IsPaused) + require.True(t, r.LastEvaluation.IsZero()) + require.Equal(t, "ok", r.Health) + require.Equal(t, "inactive", r.State) }, }, { @@ -98,15 +73,10 @@ func TestParseState_HappyPaths(t *testing.T) { wantRules: 1, wantInstances: 1, checkFirst: func(t *testing.T, r StateRule) { - if r.Health != "error" { - t.Errorf("Health = %q, want error", r.Health) - } - if r.LastError == "" { - t.Errorf("LastError is empty, want a message") - } - if len(r.Instances) != 1 || r.Instances[0].State != StateError { - t.Fatalf("Instances = %+v, want one error instance", r.Instances) - } + require.Equal(t, "error", r.Health) + require.NotEmpty(t, r.LastError) + require.Len(t, r.Instances, 1) + require.Equal(t, StateError, r.Instances[0].State) }, }, { @@ -114,12 +84,9 @@ func TestParseState_HappyPaths(t *testing.T) { wantRules: 1, wantInstances: 1, checkFirst: func(t *testing.T, r StateRule) { - if r.Health != "nodata" { - t.Errorf("Health = %q, want nodata", r.Health) - } - if len(r.Instances) != 1 || r.Instances[0].State != StateNodata { - t.Fatalf("Instances = %+v, want one nodata instance", r.Instances) - } + require.Equal(t, "nodata", r.Health) + require.Len(t, r.Instances, 1) + require.Equal(t, StateNodata, r.Instances[0].State) }, }, { @@ -132,17 +99,14 @@ func TestParseState_HappyPaths(t *testing.T) { byReason[inst.Reason] = inst } errInst, ok := byReason["Error"] - if !ok || errInst.State != StateNormal { - t.Errorf(`want an instance with State=normal Reason="Error", got %+v`, byReason["Error"]) - } + require.True(t, ok, `want an instance with Reason="Error"`) + require.Equal(t, StateNormal, errInst.State) nodataInst, ok := byReason["NoData"] - if !ok || nodataInst.State != StateNormal { - t.Errorf(`want an instance with State=normal Reason="NoData", got %+v`, byReason["NoData"]) - } + require.True(t, ok, `want an instance with Reason="NoData"`) + require.Equal(t, StateNormal, nodataInst.State) plain, ok := byReason[""] - if !ok || plain.State != StateNormal { - t.Errorf(`want a plain State=normal Reason="" instance, got %+v`, byReason[""]) - } + require.True(t, ok, `want a plain Reason="" instance`) + require.Equal(t, StateNormal, plain.State) }, }, { @@ -150,12 +114,8 @@ func TestParseState_HappyPaths(t *testing.T) { wantRules: 1, wantInstances: 0, checkFirst: func(t *testing.T, r StateRule) { - if r.Instances != nil { - t.Errorf("Instances = %+v, want nil", r.Instances) - } - if r.Totals != nil { - t.Errorf("Totals = %+v, want nil", r.Totals) - } + require.Nil(t, r.Instances) + require.Nil(t, r.Totals) }, }, { @@ -163,12 +123,10 @@ func TestParseState_HappyPaths(t *testing.T) { wantRules: 1, wantInstances: 1, checkFirst: func(t *testing.T, r StateRule) { - if len(r.Instances) != 1 || r.Instances[0].State != StateFiring { - t.Fatalf("Instances = %+v, want one firing instance", r.Instances) - } - if r.Totals["normal"] == 0 { - t.Errorf(`Totals["normal"] = 0, want >0 (the totals/instances mismatch this fixture exists to capture)`) - } + require.Len(t, r.Instances, 1) + require.Equal(t, StateFiring, r.Instances[0].State) + require.NotZero(t, r.Totals["normal"], + "the totals/instances mismatch this fixture exists to capture") }, }, } @@ -176,15 +134,9 @@ func TestParseState_HappyPaths(t *testing.T) { for _, c := range cases { t.Run(c.fixture, func(t *testing.T) { rules, err := ParseState(readFixture(t, c.fixture)) - if err != nil { - t.Fatalf("ParseState(%s): unexpected error: %v", c.fixture, err) - } - if len(rules) != c.wantRules { - t.Fatalf("ParseState(%s): got %d rules, want %d", c.fixture, len(rules), c.wantRules) - } - if got := len(rules[0].Instances); got != c.wantInstances { - t.Fatalf("ParseState(%s): got %d instances, want %d", c.fixture, got, c.wantInstances) - } + require.NoErrorf(t, err, "ParseState(%s)", c.fixture) + require.Lenf(t, rules, c.wantRules, "ParseState(%s)", c.fixture) + require.Lenf(t, rules[0].Instances, c.wantInstances, "ParseState(%s)", c.fixture) if c.checkFirst != nil { c.checkFirst(t, rules[0]) } @@ -214,13 +166,9 @@ func TestParseState_MustError(t *testing.T) { for _, c := range cases { t.Run(c.fixture, func(t *testing.T) { _, err := ParseState(readFixture(t, c.fixture)) - if err == nil { - t.Fatalf("ParseState(%s): expected an error, got none", c.fixture) - } + require.Errorf(t, err, "ParseState(%s): expected an error, got none", c.fixture) for _, want := range c.wantContains { - if !strings.Contains(err.Error(), want) { - t.Errorf("ParseState(%s): error %q does not mention %q", c.fixture, err.Error(), want) - } + require.Containsf(t, err.Error(), want, "ParseState(%s): error", c.fixture) } }) } @@ -248,49 +196,35 @@ func TestParseNormalizeInstanceState(t *testing.T) { for _, c := range cases { state, reason, err := normalizeInstanceState(c.in) if c.wantErr { - if err == nil { - t.Errorf("normalizeInstanceState(%q): expected an error, got none", c.in) - } - continue - } - if err != nil { - t.Errorf("normalizeInstanceState(%q): unexpected error: %v", c.in, err) + require.Errorf(t, err, "normalizeInstanceState(%q)", c.in) continue } - if state != c.wantState || reason != c.wantReason { - t.Errorf("normalizeInstanceState(%q) = (%q, %q), want (%q, %q)", c.in, state, reason, c.wantState, c.wantReason) - } + require.NoErrorf(t, err, "normalizeInstanceState(%q)", c.in) + require.Equalf(t, c.wantState, state, "normalizeInstanceState(%q)", c.in) + require.Equalf(t, c.wantReason, reason, "normalizeInstanceState(%q)", c.in) } } func TestInstanceKey(t *testing.T) { a := instanceKey(map[string]string{"b": "2", "a": "1"}) b := instanceKey(map[string]string{"a": "1", "b": "2"}) - if a != b { - t.Errorf("instanceKey order-independence: %q != %q", a, b) - } - if a != `{"a":"1","b":"2"}` { - t.Errorf("instanceKey = %q, want %q", a, `{"a":"1","b":"2"}`) - } + require.Equal(t, b, a, "instanceKey order-independence") + require.Equal(t, `{"a":"1","b":"2"}`, a) diff := instanceKey(map[string]string{"a": "1", "b": "3"}) - if a == diff { - t.Errorf("instanceKey should differ when a label value differs") - } + require.NotEqual(t, a, diff, "instanceKey should differ when a label value differs") - if instanceKey(nil) != "null" { - t.Errorf("instanceKey(nil) = %q, want \"null\"", instanceKey(nil)) - } + require.Equal(t, "null", instanceKey(nil), "instanceKey(nil) should be \"null\"") } +<<<<<<< HEAD // Label values may contain "\n" or "="; the JSON encoding must keep them distinct. +======= +>>>>>>> c276546b (chore: use testify's require in tests) func TestInstanceKey_NoCollision(t *testing.T) { - if instanceKey(map[string]string{"a": "1\nb=2"}) == instanceKey(map[string]string{"a": "1", "b": "2"}) { - t.Errorf("instanceKey collided for sets {a:1\\nb=2} and {a:1,b:2}") - } - if instanceKey(map[string]string{"a": "1=b"}) == instanceKey(map[string]string{"a": "1", "b": ""}) { - t.Errorf("instanceKey collided for a value containing '='") - } + require.NotEqual(t, instanceKey(map[string]string{"a": "1\nb=2"}), instanceKey(map[string]string{"a": "1", "b": "2"}), "instanceKey should not collide for sets {a:1\\nb=2} and {a:1,b:2}") + + require.NotEqual(t, instanceKey(map[string]string{"a": "1=b"}), instanceKey(map[string]string{"a": "1", "b": ""}), "instanceKey should not collide for sets {a:1=b} and {a:1,b:}") } // minimalStateBody is the smallest legal state response: one group, one @@ -319,12 +253,8 @@ func TestParseState_KeepFiringForIsOptional(t *testing.T) { for _, tc := range tests { t.Run(tc.name, func(t *testing.T) { rules, err := ParseState(minimalStateBody(tc.extra)) - if err != nil { - t.Fatalf("ParseState: %v", err) - } - if len(rules) != 1 { - t.Fatalf("rules = %+v, want one", rules) - } + require.NoError(t, err) + require.Len(t, rules, 1) }) } } @@ -335,15 +265,10 @@ func TestParseState_KeepFiringForIsOptional(t *testing.T) { func TestParseState_InstanceWithoutLabelsParses(t *testing.T) { body := minimalStateBody(`,"alerts":[{"state":"Normal","activeAt":"2026-01-01T00:00:00Z"}]`) rules, err := ParseState(body) - if err != nil { - t.Fatalf("ParseState: %v", err) - } - if len(rules) != 1 || len(rules[0].Instances) != 1 { - t.Fatalf("rules = %+v, want one rule with one instance", rules) - } - if got := rules[0].Instances[0].Labels; len(got) != 0 { - t.Errorf("Instance.Labels = %v, want empty/nil", got) - } + require.NoError(t, err) + require.Len(t, rules, 1) + require.Len(t, rules[0].Instances, 1) + require.Empty(t, rules[0].Instances[0].Labels) } // synthesizeHighCardinalityState builds a state response with a single rule @@ -357,25 +282,15 @@ func synthesizeHighCardinalityState(t *testing.T, alerting, normal int) []byte { base := readFixture(t, "state_one_instance.json") var top map[string]json.RawMessage - if err := json.Unmarshal(base, &top); err != nil { - t.Fatalf("synthesize: %v", err) - } + require.NoError(t, json.Unmarshal(base, &top)) var data map[string]json.RawMessage - if err := json.Unmarshal(top["data"], &data); err != nil { - t.Fatalf("synthesize: %v", err) - } + require.NoError(t, json.Unmarshal(top["data"], &data)) var groups []map[string]json.RawMessage - if err := json.Unmarshal(data["groups"], &groups); err != nil { - t.Fatalf("synthesize: %v", err) - } + require.NoError(t, json.Unmarshal(data["groups"], &groups)) var rules []map[string]json.RawMessage - if err := json.Unmarshal(groups[0]["rules"], &rules); err != nil { - t.Fatalf("synthesize: %v", err) - } + require.NoError(t, json.Unmarshal(groups[0]["rules"], &rules)) var alerts []map[string]json.RawMessage - if err := json.Unmarshal(rules[0]["alerts"], &alerts); err != nil { - t.Fatalf("synthesize: %v", err) - } + require.NoError(t, json.Unmarshal(rules[0]["alerts"], &alerts)) template := alerts[0] newAlerts := make([]map[string]json.RawMessage, 0, alerting+normal) @@ -400,9 +315,7 @@ func synthesizeHighCardinalityState(t *testing.T, alerting, normal int) []byte { top["data"] = mustRaw(t, data) out, err := json.Marshal(top) - if err != nil { - t.Fatalf("synthesize: %v", err) - } + require.NoError(t, err) return out } @@ -419,29 +332,19 @@ func cloneRawMap(m map[string]json.RawMessage) map[string]json.RawMessage { func mustRaw(t *testing.T, v any) json.RawMessage { t.Helper() b, err := json.Marshal(v) - if err != nil { - t.Fatalf("marshal: %v", err) - } + require.NoError(t, err) return json.RawMessage(b) } func TestParseState_HighCardinality(t *testing.T) { body := synthesizeHighCardinalityState(t, 445, 2004) - if !bytes.Contains(body, []byte("alerting-0")) { - t.Fatalf("synthesized body missing expected content") - } + require.Contains(t, string(body), "alerting-0") rules, err := ParseState(body) - if err != nil { - t.Fatalf("ParseState: unexpected error: %v", err) - } - if len(rules) != 1 { - t.Fatalf("got %d rules, want 1", len(rules)) - } + require.NoError(t, err) + require.Len(t, rules, 1) r := rules[0] - if len(r.Instances) != 445+2004 { - t.Fatalf("got %d instances, want %d", len(r.Instances), 445+2004) - } + require.Len(t, r.Instances, 445+2004) var firing, normal int for _, inst := range r.Instances { @@ -451,28 +354,21 @@ func TestParseState_HighCardinality(t *testing.T) { case StateNormal: normal++ default: - t.Fatalf("unexpected instance state %q", inst.State) + require.Fail(t, fmt.Sprintf("unexpected instance state %q", inst.State)) } } - if firing != 445 || normal != 2004 { - t.Fatalf("got firing=%d normal=%d, want firing=445 normal=2004", firing, normal) - } + require.Equal(t, 445, firing) + require.Equal(t, 2004, normal) // Each synthesized instance carries a distinct "instance" label; confirm // Labels actually made it through parsing (not just State) by checking // instanceKey produces one unique key per instance, with no collisions. seen := make(map[string]bool, len(r.Instances)) for _, inst := range r.Instances { - if inst.Labels == nil { - t.Fatalf("instance has nil Labels") - } + require.NotNil(t, inst.Labels) k := instanceKey(inst.Labels) - if seen[k] { - t.Fatalf("duplicate instance key %q", k) - } + require.Falsef(t, seen[k], "duplicate instance key %q", k) seen[k] = true } - if len(seen) != 445+2004 { - t.Fatalf("got %d unique instance keys, want %d", len(seen), 445+2004) - } + require.Len(t, seen, 445+2004) } diff --git a/grafana-alertcheck/internal/gate/resolve_test.go b/grafana-alertcheck/internal/gate/resolve_test.go index d58ec73bd..33214516e 100644 --- a/grafana-alertcheck/internal/gate/resolve_test.go +++ b/grafana-alertcheck/internal/gate/resolve_test.go @@ -2,128 +2,89 @@ package gate import ( "fmt" - "strings" "testing" + + "github.com/stretchr/testify/require" ) func rulerDefs(t *testing.T) []Definition { t.Helper() defs, err := ParseDefinitions(readFixture(t, "ruler_rules.json")) - if err != nil { - t.Fatalf("ParseDefinitions: unexpected error: %v", err) - } + require.NoError(t, err) return defs } func TestResolve_SingleMatch(t *testing.T) { defs := rulerDefs(t) resolved, notes, err := Resolve(defs, []string{"example_workflow_paused_rule"}, "") - if err != nil { - t.Fatalf("Resolve: unexpected error: %v", err) - } - if len(notes) != 0 { - t.Errorf("notes = %v, want none", notes) - } - if len(resolved) != 1 || resolved[0].UID != "rule0000007" { - t.Fatalf("resolved = %+v, want [rule0000007]", resolved) - } + require.NoError(t, err) + require.Empty(t, notes) + require.Len(t, resolved, 1) + require.Equal(t, "rule0000007", resolved[0].UID) } func TestResolve_UIDForm(t *testing.T) { defs := rulerDefs(t) resolved, _, err := Resolve(defs, []string{"uid:rule0000006a"}, "") - if err != nil { - t.Fatalf("Resolve: unexpected error: %v", err) - } - if len(resolved) != 1 || resolved[0].UID != "rule0000006a" { - t.Fatalf("resolved = %+v, want [rule0000006a]", resolved) - } + require.NoError(t, err) + require.Len(t, resolved, 1) + require.Equal(t, "rule0000006a", resolved[0].UID) } func TestResolve_FolderTitleForm(t *testing.T) { defs := rulerDefs(t) resolved, _, err := Resolve(defs, []string{"ExampleFeeds/TEMP - Example depeg alert"}, "") - if err != nil { - t.Fatalf("Resolve: unexpected error: %v", err) - } - if len(resolved) != 1 || resolved[0].UID != "rule0000008" { - t.Fatalf("resolved = %+v, want [rule0000008]", resolved) - } + require.NoError(t, err) + require.Len(t, resolved, 1) + require.Equal(t, "rule0000008", resolved[0].UID) } func TestResolve_FolderGroupTitleForm(t *testing.T) { defs := rulerDefs(t) resolved, _, err := Resolve(defs, []string{"Example-Zone-A/Gateway/Example No Gateways Available"}, "") - if err == nil { - t.Fatalf("Resolve: want ambiguous error (real 2-way collision), got resolved=%+v", resolved) - } - if !strings.Contains(err.Error(), "matches 2 rules") { - t.Fatalf("Resolve: error = %q, want it to report 2 matches", err) - } - if !strings.Contains(err.Error(), "uid:rule0000006a") || !strings.Contains(err.Error(), "uid:rule0000006b") { - t.Fatalf("Resolve: error = %q, want both candidate uids listed", err) - } + require.Error(t, err, "want ambiguous error (real 2-way collision), got resolved=%+v", resolved) + require.Contains(t, err.Error(), "matches 2 rules") + require.Contains(t, err.Error(), "uid:rule0000006a") + require.Contains(t, err.Error(), "uid:rule0000006b") } func TestResolve_TrueCollisionResolvesByUID(t *testing.T) { defs := rulerDefs(t) resolved, _, err := Resolve(defs, []string{"uid:rule0000006a", "uid:rule0000006b"}, "") - if err != nil { - t.Fatalf("Resolve: unexpected error: %v", err) - } - if len(resolved) != 2 { - t.Fatalf("resolved = %+v, want 2 distinct rules", resolved) - } + require.NoError(t, err) + require.Len(t, resolved, 2) } func TestResolve_NoMatch(t *testing.T) { defs := rulerDefs(t) _, _, err := Resolve(defs, []string{"Does Not Exist"}, "") - if err == nil { - t.Fatal("Resolve: want error for unknown name") - } - if !strings.Contains(err.Error(), "no rule matched") || !strings.Contains(err.Error(), "list") { - t.Errorf("Resolve: error = %q, want it to name 'no rule matched' and point at 'list'", err) - } + require.Error(t, err) + require.Contains(t, err.Error(), "no rule matched") + require.Contains(t, err.Error(), "list") } func TestResolve_NoMatchSubstringSuggestion(t *testing.T) { defs := rulerDefs(t) _, _, err := Resolve(defs, []string{"paused rule"}, "") - if err == nil { - t.Fatal("Resolve: want error for unknown name") - } - if !strings.Contains(err.Error(), "did you mean") || !strings.Contains(err.Error(), "Example Paused Rule") { - t.Errorf("Resolve: error = %q, want a case-insensitive substring suggestion", err) - } + require.Error(t, err) + require.Contains(t, err.Error(), "did you mean") + require.Contains(t, err.Error(), "Example Paused Rule") } func TestResolve_RefusesDatasourceManaged(t *testing.T) { defs, err := ParseDefinitions(readFixture(t, "ruler_datasource_managed.json")) - if err != nil { - t.Fatalf("ParseDefinitions: unexpected error: %v", err) - } + require.NoError(t, err) _, _, err = Resolve(defs, []string{"ExampleTargetDown"}, "") - if err == nil { - t.Fatal("Resolve: want refusal for a datasource-managed rule") - } - if !strings.Contains(err.Error(), "datasource-managed") { - t.Errorf("Resolve: error = %q, want it to name the datasource-managed kind", err) - } + require.Error(t, err) + require.Contains(t, err.Error(), "datasource-managed") } func TestResolve_RefusesRecording(t *testing.T) { defs, err := ParseDefinitions(readFixture(t, "ruler_recording.json")) - if err != nil { - t.Fatalf("ParseDefinitions: unexpected error: %v", err) - } + require.NoError(t, err) _, _, err = Resolve(defs, []string{"uid:rule0000011"}, "") - if err == nil { - t.Fatal("Resolve: want refusal for a recording rule") - } - if !strings.Contains(err.Error(), "recording rule") { - t.Errorf("Resolve: error = %q, want it to name the recording kind", err) - } + require.Error(t, err) + require.Contains(t, err.Error(), "recording rule") } func TestResolve_RejectsEmptySegments(t *testing.T) { @@ -132,12 +93,8 @@ func TestResolve_RejectsEmptySegments(t *testing.T) { for _, name := range cases { t.Run(name, func(t *testing.T) { _, _, err := Resolve(defs, []string{name}, "") - if err == nil { - t.Fatalf("Resolve(%q): want error for an empty /-separated segment", name) - } - if !strings.Contains(err.Error(), "empty") { - t.Errorf("Resolve(%q): error = %q, want it to name the empty segment", name, err) - } + require.Error(t, err) + require.Contains(t, err.Error(), "empty") }) } } @@ -148,51 +105,29 @@ func TestResolve_UIDEmptySuffix(t *testing.T) { // — that would report the misleading "datasource-managed rule, not // supported" for what is really a typo'd/empty uid. defs, err := ParseDefinitions(readFixture(t, "ruler_datasource_managed.json")) - if err != nil { - t.Fatalf("ParseDefinitions: unexpected error: %v", err) - } + require.NoError(t, err) _, _, err = Resolve(defs, []string{"uid:"}, "") - if err == nil { - t.Fatal("Resolve: want error for an empty uid: suffix") - } - if !strings.Contains(err.Error(), "no rule has this uid") { - t.Errorf("Resolve: error = %q, want it to say no rule has this uid", err) - } - if strings.Contains(err.Error(), "datasource-managed") { - t.Errorf("Resolve: error = %q, must not misreport this as a datasource-managed refusal", err) - } + require.Error(t, err) + require.Contains(t, err.Error(), "no rule has this uid") + require.NotContains(t, err.Error(), "datasource-managed") } func TestResolve_UnsupportedKindsExcludedFromNoMatchSurfaces(t *testing.T) { dsDefs, err := ParseDefinitions(readFixture(t, "ruler_datasource_managed.json")) - if err != nil { - t.Fatalf("ParseDefinitions(datasource_managed): unexpected error: %v", err) - } + require.NoError(t, err) recDefs, err := ParseDefinitions(readFixture(t, "ruler_recording.json")) - if err != nil { - t.Fatalf("ParseDefinitions(recording): unexpected error: %v", err) - } + require.NoError(t, err) supported := rulerDefs(t) combined := append(append(append([]Definition{}, supported...), dsDefs...), recDefs...) _, _, err = Resolve(combined, []string{"Example"}, "") - if err == nil { - t.Fatal("Resolve: want a no-match error for a name matching no title exactly") - } + require.Error(t, err, "want a no-match error for a name matching no title exactly") wantCount := fmt.Sprintf("(%d rules available", len(supported)) - if !strings.Contains(err.Error(), wantCount) { - t.Errorf("Resolve: error = %q, want the available count scoped to the %d supported rules, not the %d combined", err, len(supported), len(combined)) - } - if strings.Contains(err.Error(), "ExampleTargetDown") { - t.Errorf("Resolve: error = %q, must not suggest the datasource-managed rule", err) - } - if strings.Contains(err.Error(), "example:recorded_metric:rate5m") { - t.Errorf("Resolve: error = %q, must not suggest the recording rule", err) - } - if !strings.Contains(err.Error(), "Example Paused Rule") { - t.Errorf("Resolve: error = %q, want it to still suggest a matching supported rule", err) - } + require.Contains(t, err.Error(), wantCount) + require.NotContains(t, err.Error(), "ExampleTargetDown") + require.NotContains(t, err.Error(), "example:recorded_metric:rate5m") + require.Contains(t, err.Error(), "Example Paused Rule") } func TestResolve_UnsupportedHomonymResolvesSupportedSilently(t *testing.T) { @@ -205,12 +140,9 @@ func TestResolve_UnsupportedHomonymResolvesSupportedSilently(t *testing.T) { {UID: "", Folder: "F", Group: "G", Title: "Shared Title", Kind: KindDatasourceManaged}, } resolved, _, err := Resolve(defs, []string{"F/G/Shared Title"}, "") - if err != nil { - t.Fatalf("Resolve: unexpected error: %v", err) - } - if len(resolved) != 1 || resolved[0].UID != "supported-1" { - t.Fatalf("resolved = %+v, want the supported rule alone, no ambiguity", resolved) - } + require.NoError(t, err) + require.Len(t, resolved, 1) + require.Equal(t, "supported-1", resolved[0].UID) } func TestResolve_CollapseByUIDGivesNoteNotError(t *testing.T) { @@ -221,15 +153,10 @@ func TestResolve_CollapseByUIDGivesNoteNotError(t *testing.T) { "example_workflow_paused_rule", "ExampleObservability/Example Auth Production/example_workflow_paused_rule", }, "") - if err != nil { - t.Fatalf("Resolve: unexpected error: %v", err) - } - if len(resolved) != 1 || resolved[0].UID != "rule0000007" { - t.Fatalf("resolved = %+v, want exactly one rule0000007", resolved) - } - if len(notes) != 1 { - t.Fatalf("notes = %v, want exactly one collapse note", notes) - } + require.NoError(t, err) + require.Len(t, resolved, 1) + require.Equal(t, "rule0000007", resolved[0].UID) + require.Len(t, notes, 1) } // The same rule named twice with the identical string must collapse to one @@ -240,15 +167,10 @@ func TestResolve_IdenticalDuplicateNameCollapsesWithNote(t *testing.T) { "example_workflow_paused_rule", "example_workflow_paused_rule", }, "") - if err != nil { - t.Fatalf("Resolve: unexpected error: %v", err) - } - if len(resolved) != 1 || resolved[0].UID != "rule0000007" { - t.Fatalf("resolved = %+v, want exactly one rule0000007", resolved) - } - if len(notes) != 1 { - t.Fatalf("notes = %v, want exactly one collapse note", notes) - } + require.NoError(t, err) + require.Len(t, resolved, 1) + require.Equal(t, "rule0000007", resolved[0].UID) + require.Len(t, notes, 1) } func TestResolve_MinObservedCountIsPostCollapse(t *testing.T) { @@ -259,44 +181,30 @@ func TestResolve_MinObservedCountIsPostCollapse(t *testing.T) { "Example Paused Rule", } resolved, notes, err := Resolve(defs, names, "") - if err != nil { - t.Fatalf("Resolve: unexpected error: %v", err) - } + require.NoError(t, err) // The default MinObserved must come from len(resolved) (2 distinct // rules) — never len(names) (3 input lines), which would be unsatisfiable. - if len(resolved) != 2 { - t.Fatalf("resolved = %+v, want 2 distinct rules after collapse", resolved) - } - if len(notes) != 1 { - t.Fatalf("notes = %v, want exactly one collapse note", notes) - } + require.Len(t, resolved, 2) + require.Len(t, notes, 1) } func TestResolve_EmptyAndBlankLinesDiscarded(t *testing.T) { defs := rulerDefs(t) resolved, _, err := Resolve(defs, []string{"", " ", "example_workflow_paused_rule", " \t "}, "") - if err != nil { - t.Fatalf("Resolve: unexpected error: %v", err) - } - if len(resolved) != 1 || resolved[0].UID != "rule0000007" { - t.Fatalf("resolved = %+v, want [rule0000007]", resolved) - } + require.NoError(t, err) + require.Len(t, resolved, 1) + require.Equal(t, "rule0000007", resolved[0].UID) } func TestResolve_FolderScopesBareTitle(t *testing.T) { defs := rulerDefs(t) // Bare title, scoped to the wrong folder — must not match. _, _, err := Resolve(defs, []string{"example_workflow_paused_rule"}, "Example-Zone-A") - if err == nil { - t.Fatal("Resolve: want no-match when folder scope excludes the only candidate") - } + require.Error(t, err, "want no-match when folder scope excludes the only candidate") // Scoped to the right folder — must match. resolved, _, err := Resolve(defs, []string{"example_workflow_paused_rule"}, "ExampleObservability") - if err != nil { - t.Fatalf("Resolve: unexpected error: %v", err) - } - if len(resolved) != 1 || resolved[0].UID != "rule0000007" { - t.Fatalf("resolved = %+v, want [rule0000007]", resolved) - } + require.NoError(t, err) + require.Len(t, resolved, 1) + require.Equal(t, "rule0000007", resolved[0].UID) } diff --git a/grafana-alertcheck/internal/gate/schedule_test.go b/grafana-alertcheck/internal/gate/schedule_test.go index 70cc8817c..96203e6b3 100644 --- a/grafana-alertcheck/internal/gate/schedule_test.go +++ b/grafana-alertcheck/internal/gate/schedule_test.go @@ -4,6 +4,8 @@ import ( "strings" "testing" "time" + + "github.com/stretchr/testify/require" ) func TestDeriveTimings_Default(t *testing.T) { @@ -11,22 +13,12 @@ func TestDeriveTimings_Default(t *testing.T) { {UID: "r1", Title: "R1", IntervalSeconds: 60}, } rules, _, notes := DeriveTimings(defs, 0) - if len(notes) != 0 { - t.Fatalf("notes = %v, want none", notes) - } + require.Empty(t, notes) rt := rules["r1"] - if rt.pollEvery != 30*time.Second { - t.Errorf("pollEvery = %s, want 30s", rt.pollEvery) - } - if rt.maxGap != 60*time.Second { - t.Errorf("maxGap = %s, want 60s", rt.maxGap) - } - if rt.healthGrace != 60*time.Second { - t.Errorf("healthGrace = %s, want 60s", rt.healthGrace) - } - if rt.evalStaleAfter != 120*time.Second { - t.Errorf("evalStaleAfter = %s, want 120s", rt.evalStaleAfter) - } + require.Equal(t, 30*time.Second, rt.pollEvery) + require.Equal(t, 60*time.Second, rt.maxGap) + require.Equal(t, 60*time.Second, rt.healthGrace) + require.Equal(t, 120*time.Second, rt.evalStaleAfter) } func TestDeriveTimings_OverrideVerbatimNoClamp(t *testing.T) { @@ -35,23 +27,16 @@ func TestDeriveTimings_OverrideVerbatimNoClamp(t *testing.T) { } rules, _, notes := DeriveTimings(defs, 20*time.Second) rt := rules["r1"] - if rt.pollEvery != 20*time.Second { - t.Fatalf("pollEvery = %s, want the override verbatim (20s), never clamped down to the 5s default", rt.pollEvery) - } - if rt.maxGap != 40*time.Second { - t.Errorf("maxGap = %s, want 2x the override (40s)", rt.maxGap) - } - if len(notes) != 1 || !strings.Contains(notes[0], "R1") { - t.Fatalf("notes = %v, want one note naming R1's exceeded default", notes) - } + require.Equal(t, 20*time.Second, rt.pollEvery, "the override verbatim (20s), never clamped down to the 5s default") + require.Equal(t, 40*time.Second, rt.maxGap) + require.Len(t, notes, 1) + require.Contains(t, notes[0], "R1") } func TestDeriveTimings_OverrideBelowDefaultNoNote(t *testing.T) { defs := []Definition{{UID: "r1", Title: "R1", IntervalSeconds: 60}} // default pollEvery = 30s _, _, notes := DeriveTimings(defs, 5*time.Second) - if len(notes) != 0 { - t.Fatalf("notes = %v, want none when the override tightens rather than exceeds the default", notes) - } + require.Empty(t, notes) } func TestDeriveTimings_TransitionGraceExcludesSkippedRule(t *testing.T) { @@ -61,12 +46,8 @@ func TestDeriveTimings_TransitionGraceExcludesSkippedRule(t *testing.T) { } _, global, _ := DeriveTimings(defs, 0) want := time.Minute + 60*time.Second // r1's for+interval; r2 (skipped) must not win despite its huge `for` - if global.transitionGrace != want { - t.Fatalf("transitionGrace = %s, want %s (paused rule r2 must be excluded from the max)", global.transitionGrace, want) - } - if !strings.Contains(global.graceSource, "Tight") { - t.Errorf("graceSource = %q, want it to name the contributing rule Tight", global.graceSource) - } + require.Equal(t, want, global.transitionGrace) + require.Contains(t, global.graceSource, "Tight") } // `for: 1d` and `for: 1w` parse correctly (parse_ruler_test.go), @@ -79,25 +60,17 @@ func TestDeriveTimings_TransitionGraceExcludesSkippedRule(t *testing.T) { func TestDeriveTimings_RealForOneWeekRuleSetsTransitionGrace(t *testing.T) { defs := rulerDefs(t) _, global, notes := DeriveTimings(defs, 0) - if len(notes) != 0 { - t.Fatalf("notes = %v, want none: no --poll-interval override is given, so no override note should fire", notes) - } + require.Empty(t, notes, "no --poll-interval override is given, so no override note should fire") want := 7*24*time.Hour + 60*time.Second // rule0000010: for=1w, intervalSeconds=60 - if global.transitionGrace != want { - t.Fatalf("transitionGrace = %s, want %s (rule0000010's for:1w plus its interval)", global.transitionGrace, want) - } - if !strings.Contains(global.graceSource, "Example Failure Ratio Above 10 Percent Weekly") { - t.Errorf("graceSource = %q, want it to name rule0000010", global.graceSource) - } + require.Equal(t, want, global.transitionGrace) + require.Contains(t, global.graceSource, "Example Failure Ratio Above 10 Percent Weekly") } func TestDeriveTimings_TransitionGraceZeroWhenAllSkipped(t *testing.T) { defs := []Definition{{UID: "r1", Title: "R1", IntervalSeconds: 60, For: time.Hour, IsPaused: true}} _, global, _ := DeriveTimings(defs, 0) - if global.transitionGrace != 0 { - t.Fatalf("transitionGrace = %s, want 0 when every rule is skipped", global.transitionGrace) - } + require.Zero(t, global.transitionGrace) } // TestDeriveTimingsFromLog_TransitionGraceFollowsTheHeaderNotTheDefinition @@ -123,28 +96,18 @@ func TestDeriveTimingsFromLog_TransitionGraceFollowsTheHeaderNotTheDefinition(t t.Run("header says active: the rule stays in the max", func(t *testing.T) { h := Header{Rules: []LoggedRule{loggedRule("r1", false)}} _, global, err := DeriveTimingsFromLog(h, defs) - if err != nil { - t.Fatalf("DeriveTimingsFromLog: %v", err) - } - if global.transitionGrace != want { - t.Fatalf("transitionGrace = %s, want %s: the rule was active when the recording opened, "+ - "so a pause applied afterwards must not shrink the window", global.transitionGrace, want) - } - if !strings.Contains(global.graceSource, "R1") { - t.Errorf("graceSource = %q, want it to name R1", global.graceSource) - } + require.NoError(t, err) + require.Equal(t, want, global.transitionGrace, + "the rule was active when the recording opened, so a pause applied afterwards must not shrink the window") + require.Contains(t, global.graceSource, "R1") }) t.Run("header says paused: the rule stays out", func(t *testing.T) { h := Header{Rules: []LoggedRule{loggedRule("r1", true)}} _, global, err := DeriveTimingsFromLog(h, defs) - if err != nil { - t.Fatalf("DeriveTimingsFromLog: %v", err) - } - if global.transitionGrace != 0 { - t.Fatalf("transitionGrace = %s, want 0: a rule paused before the window opened can never fire during it", - global.transitionGrace) - } + require.NoError(t, err) + require.Zero(t, global.transitionGrace, + "a rule paused before the window opened can never fire during it") }) t.Run("drainTimeout counts every rule either way", func(t *testing.T) { @@ -153,13 +116,8 @@ func TestDeriveTimingsFromLog_TransitionGraceFollowsTheHeaderNotTheDefinition(t for _, pausedAtStart := range []bool{false, true} { h := Header{Rules: []LoggedRule{loggedRule("r1", pausedAtStart)}} _, global, err := DeriveTimingsFromLog(h, defs) - if err != nil { - t.Fatalf("DeriveTimingsFromLog: %v", err) - } - if global.drainTimeout != minDrainTimeout { - t.Fatalf("drainTimeout = %s with pausedAtStart=%v, want the %s floor", - global.drainTimeout, pausedAtStart, minDrainTimeout) - } + require.NoError(t, err) + require.Equalf(t, minDrainTimeout, global.drainTimeout, "pausedAtStart=%v", pausedAtStart) } }) } @@ -167,9 +125,7 @@ func TestDeriveTimingsFromLog_TransitionGraceFollowsTheHeaderNotTheDefinition(t func TestDeriveTimings_DrainTimeoutIncludesPaused(t *testing.T) { defs := []Definition{{UID: "r1", Title: "R1", IntervalSeconds: 10}} _, global, _ := DeriveTimings(defs, 0) - if global.drainTimeout != minDrainTimeout { - t.Fatalf("drainTimeout = %s, want the %s floor", global.drainTimeout, minDrainTimeout) - } + require.Equal(t, minDrainTimeout, global.drainTimeout) } func TestDeriveTimings_DrainTimeoutFloor(t *testing.T) { @@ -179,17 +135,13 @@ func TestDeriveTimings_DrainTimeoutFloor(t *testing.T) { } _, global, _ := DeriveTimings(defs, 0) // double the longest interval (2 * 180s) should be the drain timeout - if global.drainTimeout != 2*180*time.Second { - t.Fatalf("drainTimeout = %s, want %s", global.drainTimeout, 2*180*time.Second) - } + require.Equal(t, 2*180*time.Second, global.drainTimeout) } func TestDeriveTimings_DrainTimeoutAboveFloor(t *testing.T) { defs := []Definition{{UID: "r1", Title: "R1", IntervalSeconds: 300}} // 2x300s = 600s > 2m floor _, global, _ := DeriveTimings(defs, 0) - if global.drainTimeout != 600*time.Second { - t.Fatalf("drainTimeout = %s, want 600s", global.drainTimeout) - } + require.Equal(t, 600*time.Second, global.drainTimeout) } // The ordering invariant the burst bound depends on: when several rules become @@ -210,9 +162,8 @@ func TestScheduler_DueOrderingTiesBreakByTightestCadence(t *testing.T) { }, } due := s.Due(now) - if len(due) != 4 || due[0] != "tight" { - t.Fatalf("Due = %v, want the tightest-cadence rule (tight) first when all are simultaneously due", due) - } + require.Len(t, due, 4) + require.Equal(t, "tight", due[0], "the tightest-cadence rule must be first when all are simultaneously due") } func TestScheduler_DueExcludesNotYetDue(t *testing.T) { @@ -222,9 +173,8 @@ func TestScheduler_DueExcludesNotYetDue(t *testing.T) { every: map[string]time.Duration{"soon": 10 * time.Second, "later": 10 * time.Second}, } due := s.Due(now) - if len(due) != 1 || due[0] != "soon" { - t.Fatalf("Due = %v, want only [soon]", due) - } + require.Len(t, due, 1) + require.Equal(t, "soon", due[0]) } func TestScheduler_MarkAdvancesNextDue(t *testing.T) { @@ -233,15 +183,9 @@ func TestScheduler_MarkAdvancesNextDue(t *testing.T) { next: map[string]time.Time{"r1": now}, every: map[string]time.Duration{"r1": 30 * time.Second}, } - if err := s.Mark("r1", now); err != nil { - t.Fatalf("Mark: unexpected error: %v", err) - } - if got := s.Due(now); len(got) != 0 { - t.Fatalf("Due right after Mark = %v, want none (next due is 30s out)", got) - } - if got := s.Due(now.Add(30 * time.Second)); len(got) != 1 { - t.Fatalf("Due at next-due time = %v, want [r1]", got) - } + require.NoError(t, s.Mark("r1", now), "Mark must succeed for a known rule") + require.Empty(t, s.Due(now), "next due is 30s out") + require.Len(t, s.Due(now.Add(30*time.Second)), 1) } func TestScheduler_MarkUnknownUIDFails(t *testing.T) { @@ -288,15 +232,12 @@ func TestScheduler_PerRuleCadenceOverTime(t *testing.T) { // 900s of runtime: "tight" (10s cadence) polls ~90 times, "slack" (300s // cadence) ~3 times. Assert the ratio holds rather than an exact count, // since the staggered initial offset shifts each by up to one cadence. - if counts["tight"] < 85 || counts["tight"] > 91 { - t.Errorf("tight polled %d times over 900s, want ~90 (its own 10s cadence)", counts["tight"]) - } - if counts["slack"] < 2 || counts["slack"] > 4 { - t.Errorf("slack polled %d times over 900s, want ~3 (its own 300s cadence, not tight's)", counts["slack"]) - } - if counts["slack"] >= counts["tight"] { - t.Fatalf("slack polled as often as tight (%d vs %d) — schedules must be per rule, not a shared global cycle", counts["slack"], counts["tight"]) - } + require.GreaterOrEqual(t, counts["tight"], 85) + require.LessOrEqual(t, counts["tight"], 91) + require.GreaterOrEqual(t, counts["slack"], 2) + require.LessOrEqual(t, counts["slack"], 4) + require.Less(t, counts["slack"], counts["tight"], + "schedules must be per rule, not a shared global cycle") } func TestNewScheduler_StaggersWithinPollEvery(t *testing.T) { @@ -304,9 +245,8 @@ func TestNewScheduler_StaggersWithinPollEvery(t *testing.T) { rules := map[string]time.Duration{"r1": 100 * time.Second} s := NewScheduler(rules, now) offset := s.next["r1"].Sub(now) - if offset < 0 || offset >= 100*time.Second { - t.Fatalf("initial offset = %s, want within [0, 100s)", offset) - } + require.GreaterOrEqual(t, offset, time.Duration(0)) + require.Less(t, offset, 100*time.Second) } func TestScheduler_EarliestDueEmpty(t *testing.T) { @@ -359,9 +299,7 @@ func TestCheckBudget_MixedIntervalRegression(t *testing.T) { timings[uid] = ruleTimings{pollEvery: 150 * time.Second} measured[uid] = 1800 * time.Millisecond } - if err := CheckBudget(timings, measured, 1); err != nil { - t.Fatalf("CheckBudget = %v, want nil (utilization 0.6, burst bound 1.8s <= 5s)", err) - } + require.NoError(t, CheckBudget(timings, measured, 1)) } func TestCheckBudget_UtilizationExceeded(t *testing.T) { @@ -371,9 +309,7 @@ func TestCheckBudget_UtilizationExceeded(t *testing.T) { } measured := map[string]time.Duration{"a": 9 * time.Second, "b": 9 * time.Second} err := CheckBudget(timings, measured, 1) - if err == nil { - t.Fatal("CheckBudget = nil, want an error: utilization 1.8 > concurrency 1") - } + require.Error(t, err) assertBudgetMessage(t, err.Error()) } @@ -381,9 +317,7 @@ func TestCheckBudget_SingleRuleExceedsOwnCadence(t *testing.T) { timings := map[string]ruleTimings{"slow": {pollEvery: 5 * time.Second}} measured := map[string]time.Duration{"slow": 6 * time.Second} err := CheckBudget(timings, measured, 10) - if err == nil { - t.Fatal("CheckBudget = nil, want an error: measured 6s exceeds its own 5s poll-interval") - } + require.Error(t, err, "measured 6s exceeds its own 5s poll-interval") assertBudgetMessage(t, err.Error()) } @@ -397,12 +331,8 @@ func TestCheckBudget_BurstBoundViolation(t *testing.T) { } measured := map[string]time.Duration{"tight": 100 * time.Millisecond, "slow": 3 * time.Second} err := CheckBudget(timings, measured, 10) - if err == nil { - t.Fatal("CheckBudget = nil, want a burst-bound error: slow's 3s measured exceeds tight's 2s cadence") - } - if !strings.Contains(err.Error(), "burst bound") { - t.Errorf("error = %q, want it to name the burst bound", err.Error()) - } + require.Error(t, err, "slow's 3s measured exceeds tight's 2s cadence") + require.Contains(t, err.Error(), "burst bound") assertBudgetMessage(t, err.Error()) } @@ -412,17 +342,13 @@ func TestCheckBudget_BurstBoundOKWhenNotExceeded(t *testing.T) { "slow": {pollEvery: 100 * time.Second}, } measured := map[string]time.Duration{"tight": 100 * time.Millisecond, "slow": 1800 * time.Millisecond} - if err := CheckBudget(timings, measured, 10); err != nil { - t.Fatalf("CheckBudget = %v, want nil (1.8s <= 5s tightest cadence)", err) - } + require.NoError(t, CheckBudget(timings, measured, 10)) } func TestCheckBudget_MissingMeasurementIsAnError(t *testing.T) { timings := map[string]ruleTimings{"r1": {pollEvery: 30 * time.Second}} err := CheckBudget(timings, map[string]time.Duration{}, 10) - if err == nil { - t.Fatal("CheckBudget = nil, want an error: r1 was never measured (fail closed, not a silent zero)") - } + require.Error(t, err, "r1 was never measured (fail closed, not a silent zero)") } func TestCheckBudget_MissingMixedMeasurementIsAnError(t *testing.T) { @@ -431,15 +357,12 @@ func TestCheckBudget_MissingMixedMeasurementIsAnError(t *testing.T) { "slow": {pollEvery: 100 * time.Second}, } measured := map[string]time.Duration{"tight": 100 * time.Millisecond} - if err := CheckBudget(timings, measured, 10); err == nil { - t.Fatal("CheckBudget = nil, want an error: slow was never measured (fail closed, not a silent zero)") - } + require.Error(t, CheckBudget(timings, measured, 10), + "slow was never measured (fail closed, not a silent zero)") } func TestCheckBudget_EmptyScheduleIsFine(t *testing.T) { - if err := CheckBudget(nil, nil, 1); err != nil { - t.Fatalf("CheckBudget = %v, want nil for an empty schedule", err) - } + require.NoError(t, CheckBudget(nil, nil, 1)) } func TestCheckBudget_NonPositivePollIntervalIsAnError(t *testing.T) { @@ -462,9 +385,7 @@ func TestCheckBudget_NonPositivePollIntervalIsAnError(t *testing.T) { func assertBudgetMessage(t *testing.T, msg string) { t.Helper() for _, want := range []string{"measured", "concurrency", "poll-interval", "fewer"} { - if !strings.Contains(msg, want) { - t.Errorf("message %q missing %q", msg, want) - } + require.Contains(t, msg, want) } } @@ -477,15 +398,9 @@ func TestStartupSummary_WarningWhenGraceTooLarge(t *testing.T) { to := from.Add(10 * time.Minute) global := globalTimings{transitionGrace: 5 * time.Minute, graceSource: "R (for=4m30s, interval=30s)", drainTimeout: time.Minute} summary, warning := StartupSummary(from, to, global) - if !strings.Contains(summary, "planned run time") { - t.Errorf("summary = %q, want it to name the planned run time", summary) - } - if warning == "" { - t.Fatal("warning = \"\", want one: transitionGrace (5m) > 1/4 of the 10m window") - } - if !strings.Contains(warning, "R (for=4m30s, interval=30s)") { - t.Errorf("warning = %q, want it to name the grace source", warning) - } + require.Contains(t, summary, "planned run time") + require.NotEmpty(t, warning, "transitionGrace (5m) > 1/4 of the 10m window") + require.Contains(t, warning, "R (for=4m30s, interval=30s)") } // The test above pins the warning formula with a hand-built globalTimings. @@ -495,22 +410,14 @@ func TestStartupSummary_WarningWhenGraceTooLarge(t *testing.T) { func TestStartupSummary_RealForOneWeekRuleTriggersWarning(t *testing.T) { defs := rulerDefs(t) _, global, notes := DeriveTimings(defs, 0) - if len(notes) != 0 { - t.Fatalf("notes = %v, want none: no --poll-interval override is given, so no override note should fire", notes) - } + require.Empty(t, notes) from := time.Date(2026, 1, 1, 0, 0, 0, 0, time.UTC) to := from.Add(10 * time.Minute) // transitionGrace (>1w) dwarfs 1/4 of this window summary, warning := StartupSummary(from, to, global) - if !strings.Contains(summary, "planned run time") { - t.Errorf("summary = %q, want it to name the planned run time", summary) - } - if warning == "" { - t.Fatal("warning = \"\", want one: a real for:1w rule's transitionGrace vastly exceeds 1/4 of a 10m window") - } - if !strings.Contains(warning, "Example Failure Ratio Above 10 Percent Weekly") { - t.Errorf("warning = %q, want it to name rule0000010", warning) - } + require.Contains(t, summary, "planned run time") + require.NotEmpty(t, warning) + require.Contains(t, warning, "Example Failure Ratio Above 10 Percent Weekly") } func TestStartupSummary_NoWarningWhenGraceSmall(t *testing.T) { @@ -518,16 +425,12 @@ func TestStartupSummary_NoWarningWhenGraceSmall(t *testing.T) { to := from.Add(time.Hour) global := globalTimings{transitionGrace: time.Minute, graceSource: "R (for=30s, interval=30s)", drainTimeout: time.Minute} _, warning := StartupSummary(from, to, global) - if warning != "" { - t.Fatalf("warning = %q, want none: 1m grace is well under 1/4 of a 1h window", warning) - } + require.Empty(t, warning) } func TestStartupSummary_NoGraceSourceReadsNone(t *testing.T) { from := time.Date(2026, 1, 1, 0, 0, 0, 0, time.UTC) to := from.Add(time.Hour) summary, _ := StartupSummary(from, to, globalTimings{}) - if !strings.Contains(summary, "none") { - t.Fatalf("summary = %q, want it to read \"none\" when no rule set the grace", summary) - } + require.Contains(t, summary, "none") } diff --git a/grafana-alertcheck/internal/gate/source_test.go b/grafana-alertcheck/internal/gate/source_test.go index 39181cfca..bdf6e1634 100644 --- a/grafana-alertcheck/internal/gate/source_test.go +++ b/grafana-alertcheck/internal/gate/source_test.go @@ -7,11 +7,12 @@ import ( "net/http" "net/http/httptest" "net/url" - "strings" "sync" "sync/atomic" "testing" "time" + + "github.com/stretchr/testify/require" ) func healthBody(version string) string { @@ -79,20 +80,18 @@ func TestCheckGrafanaVersion(t *testing.T) { {"", true, nil}, } for _, c := range cases { - err := CheckGrafanaVersion(c.version) - if c.wantErr && err == nil { - t.Errorf("CheckGrafanaVersion(%q): want error, got nil", c.version) - continue - } - if !c.wantErr && err != nil { - t.Errorf("CheckGrafanaVersion(%q): unexpected error: %v", c.version, err) - continue - } - for _, want := range c.wantContains { - if !strings.Contains(err.Error(), want) { - t.Errorf("CheckGrafanaVersion(%q): error %q does not mention %q — it must name both what was found and what is supported", c.version, err.Error(), want) + t.Run(c.version, func(t *testing.T) { + err := CheckGrafanaVersion(c.version) + if c.wantErr { + require.Errorf(t, err, "CheckGrafanaVersion(%q)", c.version) + } else { + require.NoErrorf(t, err, "CheckGrafanaVersion(%q)", c.version) } - } + for _, want := range c.wantContains { + require.Contains(t, err.Error(), want, + "must name both what was found and what is supported") + } + }) } } @@ -102,20 +101,14 @@ func TestBackoffDelay(t *testing.T) { maxWithJitter := maxDelay + maxDelay/5 + time.Millisecond for n := 1; n <= 10; n++ { d := backoffDelay(base, maxDelay, n) - if d <= 0 { - t.Fatalf("backoffDelay(_, _, %d) = %v, want > 0", n, d) - } - if d > maxWithJitter { - t.Fatalf("backoffDelay(_, _, %d) = %v, want <= ~%v", n, d, maxWithJitter) - } + require.Positive(t, d) + require.LessOrEqual(t, d, maxWithJitter) } } func TestHTTPSource_Version_HappyPath(t *testing.T) { srv := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) { - if r.URL.Path != "/api/health" { - t.Errorf("path = %q, want /api/health", r.URL.Path) - } + require.Equal(t, "/api/health", r.URL.Path) w.Header().Set("Content-Type", "application/json") _, _ = w.Write([]byte(healthBody("13.1.0"))) })) @@ -124,20 +117,14 @@ func TestHTTPSource_Version_HappyPath(t *testing.T) { clock := newFakeClock(time.Now()) src := NewHTTPSource(srv.URL, "", clock) v, err := src.Version(context.Background()) - if err != nil { - t.Fatalf("Version(): unexpected error: %v", err) - } - if v != "13.1.0" { - t.Fatalf("Version() = %q, want 13.1.0", v) - } + require.NoError(t, err) + require.Equal(t, "13.1.0", v) } func TestHTTPSource_Version_NeverLogsToken(t *testing.T) { const secret = "super-secret-token" srv := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) { - if got := r.Header.Get("Authorization"); got != "Bearer "+secret { - t.Errorf("Authorization = %q, want Bearer %s", got, secret) - } + require.Equal(t, "Bearer "+secret, r.Header.Get("Authorization")) w.WriteHeader(http.StatusInternalServerError) })) defer srv.Close() @@ -145,12 +132,8 @@ func TestHTTPSource_Version_NeverLogsToken(t *testing.T) { clock := newFakeClock(time.Now()) src := NewHTTPSource(srv.URL, secret, clock) _, err := src.Version(context.Background()) - if err == nil { - t.Fatalf("Version(): want error, got nil") - } - if strings.Contains(err.Error(), secret) { - t.Fatalf("error %q leaks the token", err.Error()) - } + require.Error(t, err) + require.NotContains(t, err.Error(), secret) } func TestHTTPSource_RuleState_EmptyIsNotAnError(t *testing.T) { @@ -163,24 +146,16 @@ func TestHTTPSource_RuleState_EmptyIsNotAnError(t *testing.T) { clock := newFakeClock(time.Now()) src := NewHTTPSource(srv.URL, "", clock) obs, err := src.RuleState(context.Background(), "Anything") - if err != nil { - t.Fatalf("RuleState(): unexpected error: %v", err) - } - if len(obs.Rules) != 0 { - t.Fatalf("Rules = %+v, want empty (an authoritative 2xx is not a transport error)", obs.Rules) - } - if obs.GrafanaNow.IsZero() { - t.Fatalf("GrafanaNow is zero, want the response's Date header value") - } + require.NoError(t, err) + require.Empty(t, obs.Rules, "an authoritative 2xx is not a transport error") + require.False(t, obs.GrafanaNow.IsZero(), "want the response's Date header value") } func TestHTTPSource_RuleState_EscapesRuleName(t *testing.T) { var gotQuery string srv := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) { gotQuery = r.URL.RawQuery - if r.URL.Path != "/api/prometheus/grafana/api/v1/rules" { - t.Errorf("path = %q, want /api/prometheus/grafana/api/v1/rules", r.URL.Path) - } + require.Equal(t, "/api/prometheus/grafana/api/v1/rules", r.URL.Path) w.Header().Set("Content-Type", "application/json") _, _ = w.Write([]byte(emptyStateBody())) })) @@ -189,24 +164,16 @@ func TestHTTPSource_RuleState_EscapesRuleName(t *testing.T) { clock := newFakeClock(time.Now()) src := NewHTTPSource(srv.URL, "", clock) title := "[JD] No Job Proposals & More" - if _, err := src.RuleState(context.Background(), title); err != nil { - t.Fatalf("RuleState(): unexpected error: %v", err) - } - want := "rule_name=" + url.QueryEscape(title) - if gotQuery != want { - t.Fatalf("query = %q, want %q", gotQuery, want) - } + _, err := src.RuleState(context.Background(), title) + require.NoError(t, err) + require.Equal(t, "rule_name="+url.QueryEscape(title), gotQuery) } func TestHTTPSource_Definitions_HappyPath(t *testing.T) { body := readFixture(t, "ruler_rules.json") srv := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) { - if r.URL.Path != "/api/ruler/grafana/api/v1/rules" { - t.Errorf("path = %q, want /api/ruler/grafana/api/v1/rules", r.URL.Path) - } - if r.URL.RawQuery != "" { - t.Errorf("query = %q, want none — Definitions reads the ruler API unfiltered", r.URL.RawQuery) - } + require.Equal(t, "/api/ruler/grafana/api/v1/rules", r.URL.Path) + require.Empty(t, r.URL.RawQuery, "Definitions reads the ruler API unfiltered") w.Header().Set("Content-Type", "application/json") _, _ = w.Write(body) })) @@ -215,12 +182,8 @@ func TestHTTPSource_Definitions_HappyPath(t *testing.T) { clock := newFakeClock(time.Now()) src := NewHTTPSource(srv.URL, "", clock) defs, err := src.Definitions(context.Background()) - if err != nil { - t.Fatalf("Definitions(): unexpected error: %v", err) - } - if len(defs) == 0 { - t.Fatalf("Definitions(): got 0 definitions from a fixture known to have some") - } + require.NoError(t, err) + require.NotEmpty(t, defs) } func TestHTTPSource_Skew(t *testing.T) { @@ -249,14 +212,13 @@ func TestHTTPSource_Skew(t *testing.T) { }) src := NewHTTPSource(srv.URL, "", clock) _, err := src.Version(context.Background()) - if c.wantErr && err == nil { - t.Fatalf("Version(): want error, got nil") + if c.wantErr { + require.Error(t, err) + } else { + require.NoError(t, err) } - if !c.wantErr && err != nil { - t.Fatalf("Version(): unexpected error: %v", err) - } - if c.wantErr && calls.Load() != 1 { - t.Fatalf("calls = %d, want 1 — a skew hard error must never be retried", calls.Load()) + if c.wantErr { + require.Equal(t, int32(1), calls.Load(), "a skew hard error must never be retried") } }) } @@ -273,12 +235,8 @@ func TestHTTPSource_MissingDateHeader(t *testing.T) { clock := newFakeClock(time.Now()) src := NewHTTPSource(srv.URL, "", clock) _, err := src.Version(context.Background()) - if err == nil { - t.Fatalf("Version(): want error, got nil: a missing Date header is a hard error") - } - if calls.Load() != 1 { - t.Fatalf("calls = %d, want 1 — a missing Date header must never be retried", calls.Load()) - } + require.Error(t, err, "a missing Date header is a hard error") + require.Equal(t, int32(1), calls.Load(), "a missing Date header must never be retried") } func TestHTTPSource_UnparseableDateHeader(t *testing.T) { @@ -293,12 +251,8 @@ func TestHTTPSource_UnparseableDateHeader(t *testing.T) { clock := newFakeClock(time.Now()) src := NewHTTPSource(srv.URL, "", clock) _, err := src.Version(context.Background()) - if err == nil { - t.Fatalf("Version(): want error, got nil: an unparseable Date header is a hard error") - } - if calls.Load() != 1 { - t.Fatalf("calls = %d, want 1 — an unparseable Date header must never be retried", calls.Load()) - } + require.Error(t, err, "an unparseable Date header is a hard error") + require.Equal(t, int32(1), calls.Load(), "an unparseable Date header must never be retried") } // TestHTTPSource_ObservationTiming pins the arithmetic behind Observation's @@ -333,18 +287,10 @@ func TestHTTPSource_ObservationTiming(t *testing.T) { }) src := NewHTTPSource(srv.URL, "", clock) obs, err := src.RuleState(context.Background(), "Anything") - if err != nil { - t.Fatalf("RuleState(): unexpected error: %v", err) - } - if obs.Skew != c.drift { - t.Errorf("Skew = %v, want %v", obs.Skew, c.drift) - } - if obs.SkewBound != time.Second { - t.Errorf("SkewBound = %v, want 1s (RTT/2 with a 2s round trip to headers)", obs.SkewBound) - } - if obs.Latency != 4*time.Second { - t.Errorf("Latency = %v, want 4s (send through full body read) — not just the 2s header round trip", obs.Latency) - } + require.NoError(t, err) + require.Equal(t, c.drift, obs.Skew) + require.Equal(t, time.Second, obs.SkewBound, "RTT/2 with a 2s round trip to headers") + require.Equal(t, 4*time.Second, obs.Latency, "send through full body read — not just the 2s header round trip") }) } } @@ -378,15 +324,10 @@ func TestHTTPSourceStalenessNeverFalsePositiveUnderSkew(t *testing.T) { src := NewHTTPSource(srv.URL, "", clock) obs, err := src.RuleState(context.Background(), def.Title) - if err != nil { - t.Fatalf("RuleState(): %v", err) - } - if !obs.GrafanaNow.Equal(serverDate) { - t.Fatalf("GrafanaNow = %s, want the Date header %s, never the runner's clock %s", obs.GrafanaNow, serverDate, runnerNow) - } - if len(obs.Rules) != 1 { - t.Fatalf("Rules = %+v, want exactly one", obs.Rules) - } + require.NoError(t, err) + require.True(t, obs.GrafanaNow.Equal(serverDate), + "want the Date header, never the runner's clock") + require.Len(t, obs.Rules, 1) rt := newRuleTimings(30*time.Second, 60) // evalStaleAfter = 120s from := serverDate.Add(-10 * time.Minute) @@ -396,10 +337,8 @@ func TestHTTPSourceStalenessNeverFalsePositiveUnderSkew(t *testing.T) { sentinel := to res := proveCoverage(Header{StartedAt: from.Add(-time.Hour)}, polls, &sentinel, rt, def, from, to, 0) - if res.Unobservable { - t.Fatalf("Coverage = %+v, want no violation: 100s behind Grafana's TRUE now is under the 120s limit — "+ - "only a runner-clock leak (skewed +30s here) would push this over", res) - } + require.False(t, res.Unobservable, + "100s behind Grafana's TRUE now is under the 120s limit — only a runner-clock leak (skewed +30s here) would push this over") } func TestHTTPSource_Retry_TransientRecovers(t *testing.T) { @@ -422,18 +361,12 @@ func TestHTTPSource_Retry_TransientRecovers(t *testing.T) { clock := newFakeClock(time.Now()) src := NewHTTPSource(srv.URL, "", clock) v, err := src.Version(context.Background()) - if err != nil { - t.Fatalf("Version(): unexpected error after a transient failure: %v", err) - } - if v != "13.1.0" { - t.Fatalf("Version() = %q, want 13.1.0", v) - } + require.NoError(t, err) + require.Equal(t, "13.1.0", v) mu.Lock() n := calls mu.Unlock() - if n != 3 { - t.Fatalf("calls = %d, want 3 (2 failures + 1 success)", n) - } + require.Equal(t, 3, n) } func TestHTTPSource_Retry_ExceedsLimit(t *testing.T) { @@ -447,12 +380,9 @@ func TestHTTPSource_Retry_ExceedsLimit(t *testing.T) { clock := newFakeClock(time.Now()) src := NewHTTPSource(srv.URL, "", clock) _, err := src.Version(context.Background()) - if err == nil { - t.Fatalf("Version(): want error, got nil") - } - if n := calls.Load(); n != 6 { - t.Fatalf("calls = %d, want 6 (maxSequentialFailures=5 tolerates 5, gives up on the 6th)", n) - } + require.Error(t, err) + require.Equal(t, int32(6), calls.Load(), + "maxSequentialFailures=5 tolerates 5, gives up on the 6th") assertRetryExhausted(t, err, 6) } @@ -479,18 +409,12 @@ func TestHTTPSource_RuleState_GarbageBodyRetries(t *testing.T) { clock := newFakeClock(time.Now()) src := NewHTTPSource(srv.URL, "", clock) obs, err := src.RuleState(context.Background(), "Anything") - if err != nil { - t.Fatalf("RuleState(): unexpected error after a transient garbage body: %v", err) - } - if len(obs.Rules) != 0 { - t.Fatalf("Rules = %+v, want empty", obs.Rules) - } + require.NoError(t, err) + require.Empty(t, obs.Rules) mu.Lock() n := calls mu.Unlock() - if n != 3 { - t.Fatalf("calls = %d, want 3 (2 unparseable bodies + 1 valid one)", n) - } + require.Equal(t, 3, n) } func TestHTTPSource_Definitions_GarbageBodyGivesUp(t *testing.T) { @@ -505,12 +429,9 @@ func TestHTTPSource_Definitions_GarbageBodyGivesUp(t *testing.T) { clock := newFakeClock(time.Now()) src := NewHTTPSource(srv.URL, "", clock) _, err := src.Definitions(context.Background()) - if err == nil { - t.Fatalf("Definitions(): want error, got nil") - } - if n := calls.Load(); n != 6 { - t.Fatalf("calls = %d, want 6 — a persistently unparseable 2xx body retries like any other transport failure", n) - } + require.Error(t, err) + require.Equal(t, int32(6), calls.Load(), + "a persistently unparseable 2xx body retries like any other transport failure") assertRetryExhausted(t, err, 6) } @@ -554,9 +475,7 @@ func TestHTTPSource_NetworkFailureRetries(t *testing.T) { clock := newFakeClock(time.Now()) src := NewHTTPSource("http://127.0.0.1:1", "", clock) _, err := src.Version(context.Background()) - if err == nil { - t.Fatalf("Version(): want error, got nil") - } + require.Error(t, err) assertRetryExhausted(t, err, 6) } @@ -568,39 +487,26 @@ func TestHTTPSource_NetworkFailureRetries(t *testing.T) { func assertRetryExhausted(t *testing.T, err error, wantFailures int) { t.Helper() var reErr *RetryExhaustedError - if !errors.As(err, &reErr) { - t.Fatalf("error %v (%T): want a *RetryExhaustedError", err, err) - } - if reErr.Failures != wantFailures { - t.Errorf("RetryExhaustedError.Failures = %d, want %d", reErr.Failures, wantFailures) - } - if !strings.Contains(err.Error(), fmt.Sprintf("gave up after %d", wantFailures)) { - t.Errorf("error %q does not name the failure count", err.Error()) - } - if _, ok := errors.AsType[*TransportError](err); ok { - t.Fatalf("error %v (%T) is classified as *TransportError — an exhausted retry must be a terminal, non-retryable error", err, err) - } + require.ErrorAs(t, err, &reErr) + require.Equal(t, wantFailures, reErr.Failures) + require.Contains(t, err.Error(), fmt.Sprintf("gave up after %d", wantFailures)) + _, ok := errors.AsType[*TransportError](err) + require.False(t, ok, "an exhausted retry must be a terminal, non-retryable error") } func TestFakeClock(t *testing.T) { start := time.Date(2026, 1, 1, 0, 0, 0, 0, time.UTC) c := newFakeClock(start) - if !c.Now().Equal(start) { - t.Fatalf("Now() = %v, want %v", c.Now(), start) - } + require.True(t, c.Now().Equal(start)) c.Advance(5 * time.Minute) want := start.Add(5 * time.Minute) - if !c.Now().Equal(want) { - t.Fatalf("Now() after Advance = %v, want %v", c.Now(), want) - } + require.True(t, c.Now().Equal(want)) select { case fired := <-c.After(time.Hour): - if !fired.Equal(want.Add(time.Hour)) { - t.Fatalf("After fired with %v, want %v", fired, want.Add(time.Hour)) - } + require.True(t, fired.Equal(want.Add(time.Hour))) default: - t.Fatalf("After(1h) did not fire immediately") + require.Fail(t, "After(1h) did not fire immediately") } } @@ -610,37 +516,29 @@ func TestFakeSource(t *testing.T) { f.defs = []Definition{{UID: "u1", Title: "Rule One"}} ctx := context.Background() - if v, err := f.Version(ctx); err != nil || v != "13.1.0" { - t.Fatalf("Version() = (%q, %v), want (13.1.0, nil)", v, err) - } - if defs, err := f.Definitions(ctx); err != nil || len(defs) != 1 { - t.Fatalf("Definitions() = (%v, %v), want one definition", defs, err) - } + v, err := f.Version(ctx) + require.NoError(t, err) + require.Equal(t, "13.1.0", v) + defs, err := f.Definitions(ctx) + require.NoError(t, err) + require.Len(t, defs, 1) f.script("Rule One", Observation{Rules: []StateRule{{UID: "u1"}}}, nil) f.script("Rule One", Observation{}, fmt.Errorf("boom")) f.script("Rule One", Observation{Rules: nil}, nil) obs, err := f.RuleState(ctx, "Rule One") - if err != nil || len(obs.Rules) != 1 { - t.Fatalf("RuleState() call 1 = (%v, %v), want one rule, no error", obs, err) - } - if _, err := f.RuleState(ctx, "Rule One"); err == nil { - t.Fatalf("RuleState() call 2: want the scripted error, got nil") - } + require.NoError(t, err) + require.Len(t, obs.Rules, 1) + _, err = f.RuleState(ctx, "Rule One") + require.Error(t, err, "RuleState() call 2: want the scripted error, got nil") obs, err = f.RuleState(ctx, "Rule One") - if err != nil { - t.Fatalf("RuleState() call 3: unexpected error: %v", err) - } - if obs.Rules != nil { - t.Fatalf("RuleState() call 3: Rules = %v, want nil (last script entry, then repeats)", obs.Rules) - } + require.NoError(t, err) + require.Nil(t, obs.Rules, "last script entry, then repeats") obs, err = f.RuleState(ctx, "Rule One") - if err != nil || obs.Rules != nil { - t.Fatalf("RuleState() call 4: want the last scripted entry to repeat, got (%v, %v)", obs, err) - } + require.NoError(t, err) + require.Nil(t, obs.Rules) - if _, err := f.RuleState(ctx, "Unscripted Rule"); err == nil { - t.Fatalf("RuleState() for an unscripted title: want an error, got nil") - } + _, err = f.RuleState(ctx, "Unscripted Rule") + require.Error(t, err) } diff --git a/grafana-alertcheck/internal/gate/watch_daemon_test.go b/grafana-alertcheck/internal/gate/watch_daemon_test.go index c3a91cd12..73a6f3d7a 100644 --- a/grafana-alertcheck/internal/gate/watch_daemon_test.go +++ b/grafana-alertcheck/internal/gate/watch_daemon_test.go @@ -15,6 +15,8 @@ import ( "syscall" "testing" "time" + + "github.com/stretchr/testify/require" ) // TestMain doubles this test binary as the detached recorder. Watch spawns @@ -163,37 +165,25 @@ func grafanaTestServer(t *testing.T) *httptest.Server { func patchedStateBody(t *testing.T) []byte { t.Helper() var body map[string]any - if err := json.Unmarshal(readFixture(t, "state_one_instance.json"), &body); err != nil { - t.Fatalf("unmarshal state fixture: %v", err) - } + require.NoError(t, json.Unmarshal(readFixture(t, "state_one_instance.json"), &body)) data, ok := body["data"].(map[string]any) - if !ok { - t.Fatal("state fixture: no data object") - } + require.True(t, ok, "state fixture: no data object") groups, ok := data["groups"].([]any) - if !ok || len(groups) == 0 { - t.Fatal("state fixture: no groups") - } + require.True(t, ok, "state fixture: no groups") + require.NotEmpty(t, groups, "state fixture: no groups") group, ok := groups[0].(map[string]any) - if !ok { - t.Fatal("state fixture: group 0 is not an object") - } + require.True(t, ok, "state fixture: group 0 is not an object") rules, ok := group["rules"].([]any) - if !ok || len(rules) == 0 { - t.Fatal("state fixture: group 0 has no rules") - } + require.True(t, ok, "state fixture: group 0 has no rules") + require.NotEmpty(t, rules, "state fixture: group 0 has no rules") rule, ok := rules[0].(map[string]any) - if !ok { - t.Fatal("state fixture: rule 0 is not an object") - } + require.True(t, ok, "state fixture: rule 0 is not an object") rule["uid"] = watchActiveUID rule["name"] = watchActiveTitle rule["lastEvaluation"] = time.Now().UTC().Format(time.RFC3339Nano) b, err := json.Marshal(body) - if err != nil { - t.Fatalf("marshal patched state fixture: %v", err) - } + require.NoError(t, err) return b } @@ -209,7 +199,7 @@ func waitFor(t *testing.T, what string, timeout time.Duration, cond func() bool) } time.Sleep(20 * time.Millisecond) } - t.Fatalf("timed out after %s waiting for %s", timeout, what) + require.Fail(t, fmt.Sprintf("timed out after %s waiting for %s", timeout, what)) } // The one watch integration test: everything from the version gate to the @@ -237,9 +227,7 @@ func TestWatchSpawnsADetachedRecorder(t *testing.T) { Notes: ¬es, } - if err := Watch(context.Background(), cfg); err != nil { - t.Fatalf("Watch: %v\nnotes:\n%s", err, notes.String()) - } + require.NoError(t, Watch(context.Background(), cfg)) t.Cleanup(func() { if t.Failed() { t.Logf("notes:\n%s", notes.String()) @@ -248,20 +236,14 @@ func TestWatchSpawnsADetachedRecorder(t *testing.T) { }) pid, err := ReadPidFile(out + ".pid") - if err != nil { - t.Fatalf("ReadPidFile: %v", err) - } - if err := syscall.Kill(pid, 0); err != nil { - t.Fatalf("recorder pid %d is not running right after Watch returned: %v", pid, err) - } + require.NoError(t, err) + require.NoError(t, syscall.Kill(pid, 0), "recorder pid %d is not running right after Watch returned", pid) // Setsid, not a bare `&`: a session leader's process group id is its own // pid. Without this the child would still share the parent's process group // and die with the step that started it. - if pgid, err := syscall.Getpgid(pid); err != nil { - t.Errorf("Getpgid(%d): %v", pid, err) - } else if pgid != pid { - t.Errorf("recorder pgid = %d, want %d: it did not get its own session", pgid, pid) - } + pgid, err := syscall.Getpgid(pid) + require.NoError(t, err) + require.Equal(t, pid, pgid, "it did not get its own session") // The parent already wrote the first heartbeat before it returned; // these later ones prove the detached child is the one appending now. @@ -271,35 +253,24 @@ func TestWatchSpawnsADetachedRecorder(t *testing.T) { }) // Stop it exactly the way check does. - if err := syscall.Kill(pid, syscall.SIGTERM); err != nil { - t.Fatalf("SIGTERM %d: %v", pid, err) - } + require.NoError(t, syscall.Kill(pid, syscall.SIGTERM)) waitFor(t, "the stopped sentinel", 10*time.Second, func() bool { _, _, sentinel, err := ReadLog(out) return err == nil && sentinel != nil }) header, polls, sentinel, err := ReadLog(out) - if err != nil { - t.Fatalf("ReadLog: %v", err) - } - if header.URL != srv.URL || header.GrafanaVersion != "13.1.0" { - t.Errorf("header identity = %q/%q, want %q/13.1.0", header.URL, header.GrafanaVersion, srv.URL) - } - if len(header.Rules) != 1 || header.Rules[0].PollEverySeconds != 0.2 { - t.Errorf("header rules = %+v, want one rule recorded at 0.2s", header.Rules) - } + require.NoError(t, err) + require.Equal(t, srv.URL, header.URL) + require.Equal(t, "13.1.0", header.GrafanaVersion) + require.Len(t, header.Rules, 1) + require.Equal(t, float64(0.2), header.Rules[0].PollEverySeconds) for i, p := range polls { - if p.RuleUID != watchActiveUID || !p.Found { - t.Fatalf("poll %d = %+v, want a found observation of %s", i, p, watchActiveUID) - } - if p.GrafanaNow.IsZero() { - t.Fatalf("poll %d has no grafana_now; every poll needs the Date header of its own response", i) - } - } - if sentinel.Before(header.StartedAt) { - t.Errorf("sentinel at %s precedes the record start %s", sentinel, header.StartedAt) + require.Equalf(t, watchActiveUID, p.RuleUID, "poll %d", i) + require.Truef(t, p.Found, "poll %d", i) + require.Falsef(t, p.GrafanaNow.IsZero(), "poll %d has no grafana_now; every poll needs the Date header of its own response", i) } + require.False(t, sentinel.Before(header.StartedAt), "sentinel precedes the record start") waitFor(t, "the recorder to exit", 10*time.Second, func() bool { return syscall.Kill(pid, 0) != nil @@ -317,27 +288,17 @@ func TestDaemonChildRejectsAnAlreadyFinishedLog(t *testing.T) { clock := newFakeClock(testNow) w, err := NewWriter(path, clock) - if err != nil { - t.Fatalf("NewWriter: %v", err) - } - if err := w.WriteHeader(testHeader()); err != nil { - t.Fatalf("WriteHeader: %v", err) - } - if err := w.Stop(); err != nil { - t.Fatalf("Stop: %v", err) - } + require.NoError(t, err) + require.NoError(t, w.WriteHeader(testHeader())) + require.NoError(t, w.Stop()) err = RunDaemonChild(context.Background(), DaemonChildConfig{ URL: testHeader().URL, Out: path, Clock: clock, }) - if err == nil { - t.Fatal("RunDaemonChild: no error against a log that already carries a stopped sentinel") - } - if !strings.Contains(err.Error(), "sentinel") { - t.Errorf("error = %v, want it to name the stopped sentinel", err) - } + require.Error(t, err, "no error against a log that already carries a stopped sentinel") + require.Contains(t, err.Error(), "sentinel") } // TestWatchFailsWhenTheChildCannotStartRecording is the other half of the @@ -365,13 +326,9 @@ func TestWatchFailsWhenTheChildCannotStartRecording(t *testing.T) { Concurrency: 2, Notes: ¬es, }) - if err == nil { - t.Fatal("Watch: no error, but the child could never have started recording") - } - if !strings.Contains(err.Error(), "records url") { - t.Errorf("error does not quote the child's own reason:\n%v", err) - } - if _, statErr := os.Stat(out + ".pid"); !os.IsNotExist(statErr) { - t.Errorf("a pidfile survived a failed detach (%v); pids are reused, so the next step would signal a stranger", statErr) - } + require.Error(t, err, "the child could never have started recording") + require.Contains(t, err.Error(), "records url") + _, statErr := os.Stat(out + ".pid") + require.True(t, os.IsNotExist(statErr), + "a pidfile survived a failed detach; pids are reused, so the next step would signal a stranger") } diff --git a/grafana-alertcheck/internal/gate/watch_test.go b/grafana-alertcheck/internal/gate/watch_test.go index 33c455af7..91b52ca4b 100644 --- a/grafana-alertcheck/internal/gate/watch_test.go +++ b/grafana-alertcheck/internal/gate/watch_test.go @@ -4,14 +4,14 @@ import ( "context" "errors" "fmt" - "maps" "os" "path/filepath" - "slices" "strings" "sync" "testing" "time" + + "github.com/stretchr/testify/require" ) // The two fixture rules every prepareWatch test below uses: one live, one @@ -73,12 +73,8 @@ func testStateRule(uid, title string, interval time.Duration, grafanaNow time.Ti func newLoopWriter(t *testing.T, path string, clock Clock) *Writer { t.Helper() w, err := NewWriter(path, clock) - if err != nil { - t.Fatalf("NewWriter: %v", err) - } - if err := w.WriteHeader(testHeader()); err != nil { - t.Fatalf("WriteHeader: %v", err) - } + require.NoError(t, err) + require.NoError(t, w.WriteHeader(testHeader())) return w } @@ -123,29 +119,21 @@ func TestWatchLoopPollsEachRuleAtItsOwnCadence(t *testing.T) { Concurrency: 2, Clock: clock, }) - if err != nil { - t.Fatalf("watchLoop: %v", err) - } + require.NoError(t, err) _, polls, sentinel, readErr := ReadLog(path) - if readErr != nil { - t.Fatalf("ReadLog: %v", readErr) - } - if sentinel == nil { - t.Fatal("no stopped sentinel after a clean stop") - } - if sentinel.Before(testNow.Add(300 * time.Second)) { - t.Errorf("sentinel at %s, want >= the stop time %s", sentinel, testNow.Add(300*time.Second)) - } + require.NoError(t, readErr) + require.NotNil(t, sentinel, "no stopped sentinel after a clean stop") + require.False(t, sentinel.Before(testNow.Add(300*time.Second))) // 300s of window at 5s and 150s, minus the initial stagger offset of up to // one cadence: 59-60 and 1-2. The assertion is the ratio, not the exact // count — a single global cycle would give both rules the same number. - if got := countPolls(polls, tightUID); got < 59 || got > 61 { - t.Errorf("tight rule polled %d times, want ~60 (300s at 5s)", got) - } - if got := countPolls(polls, slackUID); got < 1 || got > 3 { - t.Errorf("slack rule polled %d times, want ~2 (300s at 150s)", got) - } + got := countPolls(polls, tightUID) + require.GreaterOrEqual(t, got, 59) + require.LessOrEqual(t, got, 61) + got = countPolls(polls, slackUID) + require.GreaterOrEqual(t, got, 1) + require.LessOrEqual(t, got, 3) } // Fail-closed from the recorder's side: a recorder that dies must look exactly @@ -174,20 +162,12 @@ func TestWatchLoopHardErrorLeavesNoSentinel(t *testing.T) { Concurrency: 1, Clock: clock, }) - if !errors.Is(err, boom) { - t.Fatalf("watchLoop error = %v, want %v", err, boom) - } + require.ErrorIs(t, err, boom) _, polls, sentinel, readErr := ReadLog(path) - if readErr != nil { - t.Fatalf("ReadLog: %v", readErr) - } - if sentinel != nil { - t.Errorf("sentinel at %s after a failed recording; check would read that as a finished window", sentinel) - } - if len(polls) != 1 { - t.Errorf("kept %d polls, want the 1 that succeeded before the failure", len(polls)) - } + require.NoError(t, readErr) + require.Nil(t, sentinel, "check would read that as a finished window") + require.Len(t, polls, 1, "want the 1 that succeeded before the failure") } // SIGTERM arriving while a poll is in flight is a clean stop, so the aborted @@ -211,7 +191,7 @@ func TestWatchLoopSignalDuringPollIsACleanStop(t *testing.T) { return observation(now, testStateRule("r1", title, time.Minute, now)), nil }) - if err := watchLoop(ctx, watchLoopConfig{ + require.NoError(t, watchLoop(ctx, watchLoopConfig{ Src: src, Writer: w, Reducer: NewReducer(), @@ -219,15 +199,11 @@ func TestWatchLoopSignalDuringPollIsACleanStop(t *testing.T) { Cadence: map[string]time.Duration{"r1": 30 * time.Second}, Concurrency: 1, Clock: clock, - }); err != nil { - t.Fatalf("watchLoop: %v", err) - } + })) - if _, _, sentinel, err := ReadLog(path); err != nil { - t.Fatalf("ReadLog: %v", err) - } else if sentinel == nil { - t.Error("no sentinel after a signalled stop; check would call a fully observed window unobservable") - } + _, _, sentinel, err := ReadLog(path) + require.NoError(t, err) + require.NotNil(t, sentinel, "no sentinel after a signalled stop; check would call a fully observed window unobservable") } // TestWatchLoopWithNothingToPollStillFinishesTheLog covers the every-rule-is- @@ -242,7 +218,7 @@ func TestWatchLoopWithNothingToPollStillFinishesTheLog(t *testing.T) { return Observation{}, fmt.Errorf("nothing should be polled, got %q", title) }) - if err := watchLoop(context.Background(), watchLoopConfig{ + require.NoError(t, watchLoop(context.Background(), watchLoopConfig{ Src: src, Writer: w, Reducer: NewReducer(), @@ -251,20 +227,12 @@ func TestWatchLoopWithNothingToPollStillFinishesTheLog(t *testing.T) { Until: testNow.Add(time.Minute), Concurrency: 1, Clock: clock, - }); err != nil { - t.Fatalf("watchLoop: %v", err) - } + })) _, polls, sentinel, err := ReadLog(path) - if err != nil { - t.Fatalf("ReadLog: %v", err) - } - if len(polls) != 0 { - t.Errorf("wrote %d polls with nothing to poll", len(polls)) - } - if sentinel == nil { - t.Error("no sentinel: check cannot tell this recording from one that died") - } + require.NoError(t, err) + require.Empty(t, polls) + require.NotNil(t, sentinel, "no sentinel: check cannot tell this recording from one that died") } // TestWatchLoopPollBatchKeepsTheHeartbeatsItGot: one rule's failure must not @@ -293,20 +261,13 @@ func TestWatchLoopPollBatchKeepsTheHeartbeatsItGot(t *testing.T) { Concurrency: 2, Clock: clock, } - if err := cfg.pollBatch(context.Background(), []string{"ok", "bad"}); !errors.Is(err, boom) { - t.Fatalf("pollBatch error = %v, want %v", err, boom) - } - if err := w.Close(); err != nil { - t.Fatalf("Close: %v", err) - } + require.ErrorIs(t, cfg.pollBatch(context.Background(), []string{"ok", "bad"}), boom) + require.NoError(t, w.Close()) _, polls, _, err := ReadLog(path) - if err != nil { - t.Fatalf("ReadLog: %v", err) - } - if len(polls) != 1 || polls[0].RuleUID != "ok" { - t.Errorf("polls = %+v, want the one heartbeat that was actually observed", polls) - } + require.NoError(t, err) + require.Len(t, polls, 1) + require.Equal(t, "ok", polls[0].RuleUID) } // The vanish-versus-clear distinction at the one seam the parent/child handoff @@ -326,27 +287,20 @@ func TestReducerSeedFromKeepsMarkersAcrossTheHandoff(t *testing.T) { r := NewReducer() r.seedFrom([]Poll{parentPoll}) p := r.Reduce("r1", childObs) - if !slices.Contains(p.Vanished, key) { - t.Errorf("vanished = %v, want it to contain %q", p.Vanished, key) - } - if len(p.Cleared) != 0 { - t.Errorf("cleared = %v, want none: a vanish is not a recovery", p.Cleared) - } + require.Contains(t, p.Vanished, key) + require.Empty(t, p.Cleared, "a vanish is not a recovery") }) t.Run("unseeded loses the transition", func(t *testing.T) { p := NewReducer().Reduce("r1", childObs) - if len(p.Vanished) != 0 { - t.Fatalf("vanished = %v; this subtest exists to show the seed is what produces the marker", p.Vanished) - } + require.Empty(t, p.Vanished, "this subtest exists to show the seed is what produces the marker") }) t.Run("a not-found poll does not clear the seed", func(t *testing.T) { r := NewReducer() r.seedFrom([]Poll{parentPoll, {RuleUID: "r1", Found: false}}) - if p := r.Reduce("r1", childObs); !slices.Contains(p.Vanished, key) { - t.Errorf("vanished = %v, want it to contain %q: an absent rule leaves the abnormal set untouched", p.Vanished, key) - } + p := r.Reduce("r1", childObs) + require.Contains(t, p.Vanished, key, "an absent rule leaves the abnormal set untouched") }) } @@ -392,50 +346,34 @@ func TestPrepareWatchDoesNotWaitForPausedRules(t *testing.T) { src := watchTestSource(t, liveObservation(testNow)) prep, err := prepareWatch(context.Background(), cfg, src) - if err != nil { - t.Fatalf("prepareWatch: %v", err) - } - if err := prep.writer.Close(); err != nil { - t.Fatalf("Close: %v", err) - } + require.NoError(t, err) + require.NoError(t, prep.writer.Close()) header, polls, sentinel, err := ReadLog(cfg.Out) - if err != nil { - t.Fatalf("ReadLog: %v", err) - } - if sentinel != nil { - t.Error("the parent wrote a sentinel; that would tell check the recording ended before the child started") - } + require.NoError(t, err) + require.Nil(t, sentinel, "the parent wrote a sentinel; that would tell check the recording ended before the child started") - if len(header.Rules) != 2 { - t.Fatalf("header names %d rules, want both the live and the paused one", len(header.Rules)) - } + require.Len(t, header.Rules, 2) for _, lr := range header.Rules { - if lr.PollEverySeconds <= 0 { - t.Errorf("header rule %s records poll_every_seconds=%v; check needs a positive cadence to derive maxGap from", lr.UID, lr.PollEverySeconds) - } - if lr.UID == watchPausedUID && !lr.IsPaused { - t.Errorf("header rule %s: is_paused = false, want the resolve-time snapshot to say true", lr.UID) + require.Positive(t, lr.PollEverySeconds, + "check needs a positive cadence to derive maxGap from") + if lr.UID == watchPausedUID { + require.True(t, lr.IsPaused, "want the resolve-time snapshot to say true") } } // One poll, for the live rule only — and it is already in the log before // prepareWatch returned, which is the whole point of the record step. - if len(polls) != 1 || polls[0].RuleUID != watchActiveUID { - t.Fatalf("polls = %+v, want exactly one first observation of %s", polls, watchActiveUID) - } - if !polls[0].Found || !polls[0].GrafanaNow.Equal(testNow) { - t.Errorf("first poll = %+v, want a found observation at %s", polls[0], testNow) - } + require.Len(t, polls, 1) + require.Equal(t, watchActiveUID, polls[0].RuleUID) + require.True(t, polls[0].Found) + require.True(t, polls[0].GrafanaNow.Equal(testNow)) // The poll record holds the state histogram, asserted through a real // prepareWatch()/Reducer call rather than log_test.go's hand-built // Writer/ReadLog round trip. - if want := map[string]int{"normal": 1}; !maps.Equal(polls[0].Histogram, want) { - t.Errorf("Histogram = %v, want %v: watch must record the state histogram on every poll it writes", polls[0].Histogram, want) - } - if !strings.Contains(notes.String(), watchPausedTitle) || !strings.Contains(notes.String(), "paused") { - t.Errorf("notes do not mention the paused rule:\n%s", notes.String()) - } + require.Equal(t, map[string]int{"normal": 1}, polls[0].Histogram) + require.Contains(t, notes.String(), watchPausedTitle) + require.Contains(t, notes.String(), "paused") } // One authority for the cadence, from the writing side: whatever @@ -448,20 +386,13 @@ func TestPrepareWatchHeaderRecordsTheOverriddenCadence(t *testing.T) { src := watchTestSource(t, liveObservation(testNow)) prep, err := prepareWatch(context.Background(), cfg, src) - if err != nil { - t.Fatalf("prepareWatch: %v", err) - } + require.NoError(t, err) defer prep.writer.Close() - if got := prep.header.Rules[0].PollEverySeconds; got != 120 { - t.Errorf("header poll_every_seconds = %v, want 120 (the override, used verbatim and never clamped)", got) - } - if got := prep.timings[watchActiveUID].maxGap; got != 240*time.Second { - t.Errorf("maxGap = %s, want 240s (2 x the recorded cadence)", got) - } - if !strings.Contains(notes.String(), "--poll-interval") { - t.Errorf("notes do not report that the override exceeds half the evaluation interval:\n%s", notes.String()) - } + require.Equal(t, float64(120), prep.header.Rules[0].PollEverySeconds, + "the override, used verbatim and never clamped") + require.Equal(t, 240*time.Second, prep.timings[watchActiveUID].maxGap) + require.Contains(t, notes.String(), "--poll-interval") } // The budget check runs on the latencies the parent just measured, before the @@ -475,9 +406,7 @@ func TestPrepareWatchFailsWhenTheScheduleDoesNotFit(t *testing.T) { src := watchTestSource(t, obs) _, err := prepareWatch(context.Background(), cfg, src) - if err == nil { - t.Fatal("prepareWatch: no error on a schedule that cannot hold its own cadence") - } + require.Error(t, err, "a schedule cannot hold its own cadence") assertBudgetMessage(t, err.Error()) } @@ -493,20 +422,14 @@ func TestPrepareWatchVerifiesNormalInstancesAreVisible(t *testing.T) { src := watchTestSource(t, observation(testNow, rule)) _, err := prepareWatch(context.Background(), cfg, src) - if err == nil { - t.Fatal("prepareWatch: no error when totals claim normal instances the response omitted") - } - if !strings.Contains(err.Error(), "no longer returns normal instances") { - t.Errorf("error does not say the endpoint stopped returning normal instances: %v", err) - } + require.Error(t, err, "totals claim normal instances the response omitted") + require.Contains(t, err.Error(), "no longer returns normal instances") // The failure happens before any poll is appended, so the log holds a // header and nothing else. - if _, polls, _, readErr := ReadLog(cfg.Out); readErr != nil { - t.Fatalf("ReadLog: %v", readErr) - } else if len(polls) != 0 { - t.Errorf("wrote %d polls from an observation it refused to trust", len(polls)) - } + _, polls, _, readErr := ReadLog(cfg.Out) + require.NoError(t, readErr) + require.Empty(t, polls) } func TestPrepareWatchRejectsAnUnsupportedGrafana(t *testing.T) { @@ -515,11 +438,10 @@ func TestPrepareWatchRejectsAnUnsupportedGrafana(t *testing.T) { src := watchTestSource(t, liveObservation(testNow)) src.version = "12.4.0" - if _, err := prepareWatch(context.Background(), cfg, src); err == nil { - t.Fatal("prepareWatch: no error on an unsupported grafana version") - } else if !strings.Contains(err.Error(), "12.4.0") || !strings.Contains(err.Error(), "13.0.0") { - t.Errorf("error names neither what was found nor what is supported: %v", err) - } + _, err := prepareWatch(context.Background(), cfg, src) + require.Error(t, err) + require.Contains(t, err.Error(), "12.4.0") + require.Contains(t, err.Error(), "13.0.0") } // A rule that resolved in the ruler API but is absent from the state endpoint @@ -531,23 +453,14 @@ func TestPrepareWatchNotesAnAbsentRule(t *testing.T) { src := watchTestSource(t, observation(testNow)) // an authoritative, empty 2xx prep, err := prepareWatch(context.Background(), cfg, src) - if err != nil { - t.Fatalf("prepareWatch: %v", err) - } - if err := prep.writer.Close(); err != nil { - t.Fatalf("Close: %v", err) - } + require.NoError(t, err) + require.NoError(t, prep.writer.Close()) _, polls, _, err := ReadLog(cfg.Out) - if err != nil { - t.Fatalf("ReadLog: %v", err) - } - if len(polls) != 1 || polls[0].Found { - t.Fatalf("polls = %+v, want one poll recorded as not found", polls) - } - if !strings.Contains(notes.String(), "absent from the state endpoint") { - t.Errorf("notes do not warn about the absent rule:\n%s", notes.String()) - } + require.NoError(t, err) + require.Len(t, polls, 1) + require.False(t, polls[0].Found, "want one poll recorded as not found") + require.Contains(t, notes.String(), "absent from the state endpoint") } func TestWatchConfigValidation(t *testing.T) { @@ -575,26 +488,16 @@ func TestWatchConfigValidation(t *testing.T) { cfg := base() tc.mutate(&cfg) err := cfg.withDefaults().validate() - if err == nil { - t.Fatalf("validate: no error, want one naming %q", tc.want) - } - if !strings.Contains(err.Error(), tc.want) { - t.Errorf("validate error = %v, want it to name %q", err, tc.want) - } + require.Errorf(t, err, "validate: no error, want one naming %q", tc.want) + require.Containsf(t, err.Error(), tc.want, "validate error") }) } t.Run("defaults derive the pidfile and daemon log from the log path", func(t *testing.T) { cfg := base().withDefaults() - if cfg.PidFile != cfg.Out+".pid" { - t.Errorf("PidFile = %q, want %q — check finds the recorder by this convention", cfg.PidFile, cfg.Out+".pid") - } - if cfg.DaemonLog == "" { - t.Error("DaemonLog is empty: a detached child would have nowhere to explain a failure") - } - if err := cfg.validate(); err != nil { - t.Errorf("validate: %v", err) - } + require.Equal(t, cfg.Out+".pid", cfg.PidFile) + require.NotEmpty(t, cfg.DaemonLog, "a detached child would have nowhere to explain a failure") + require.NoError(t, cfg.validate()) }) } @@ -609,15 +512,10 @@ func TestChildScheduleUsesTheRecordedCadence(t *testing.T) { }} titles, cadence, err := childSchedule(h) - if err != nil { - t.Fatalf("childSchedule: %v", err) - } - if _, ok := titles["paused"]; ok { - t.Error("the child scheduled a rule that was paused when the window opened") - } - if got := cadence["fast"]; got != 5*time.Second { - t.Errorf("pollEvery = %s, want 5s from the header, not %s from the interval", got, defaultPollEvery(300)) - } + require.NoError(t, err) + _, ok := titles["paused"] + require.False(t, ok, "the child scheduled a rule that was paused when the window opened") + require.Equal(t, 5*time.Second, cadence["fast"]) } func TestChildScheduleRejectsAnUnusableHeader(t *testing.T) { @@ -641,11 +539,9 @@ func TestChildScheduleRejectsAnUnusableHeader(t *testing.T) { }, } { t.Run(tc.name, func(t *testing.T) { - if _, _, err := childSchedule(tc.h); err == nil { - t.Fatalf("childSchedule: no error, want one naming %q", tc.want) - } else if !strings.Contains(err.Error(), tc.want) { - t.Errorf("error = %v, want it to name %q", err, tc.want) - } + _, _, err := childSchedule(tc.h) + require.Errorf(t, err, "childSchedule: no error, want one naming %q", tc.want) + require.Contains(t, err.Error(), tc.want) }) } } @@ -670,38 +566,25 @@ func TestChildArgsCarryNoSecretsAndNoRuleSet(t *testing.T) { joined := strings.Join(args, " ") for _, want := range []string{DaemonChildFlag, "--out /tmp/log.jsonl", "--concurrency 3", "--until ", ReadyFDFlag + " 3"} { - if !strings.Contains(joined, want) { - t.Errorf("child args %q do not contain %q", joined, want) - } + require.Contains(t, joined, want) } for _, forbidden := range []string{"secret-token", "Example", "--folder", "--poll-interval", "--pidfile"} { - if strings.Contains(joined, forbidden) { - t.Errorf("child args %q contain %q, which must not reach argv", joined, forbidden) - } + require.NotContains(t, joined, forbidden) } } func TestPidFileRoundTrip(t *testing.T) { path := filepath.Join(t.TempDir(), "log.jsonl.pid") - if err := writePidFile(path, 4242); err != nil { - t.Fatalf("writePidFile: %v", err) - } + require.NoError(t, writePidFile(path, 4242)) pid, err := ReadPidFile(path) - if err != nil { - t.Fatalf("ReadPidFile: %v", err) - } - if pid != 4242 { - t.Errorf("pid = %d, want 4242", pid) - } + require.NoError(t, err) + require.Equal(t, 4242, pid) t.Run("garbage is an error, never a pid", func(t *testing.T) { bad := filepath.Join(t.TempDir(), "bad.pid") - if err := os.WriteFile(bad, []byte("not-a-pid\n"), 0o644); err != nil { - t.Fatalf("write: %v", err) - } - if _, err := ReadPidFile(bad); err == nil { - t.Error("ReadPidFile: no error on an unparseable pidfile") - } + require.NoError(t, os.WriteFile(bad, []byte("not-a-pid\n"), 0o644)) + _, err := ReadPidFile(bad) + require.Error(t, err, "no error on an unparseable pidfile") }) } From 1e4c4859e5d419be76f14321ea260d08075352de Mon Sep 17 00:00:00 2001 From: Bartek Tofel Date: Mon, 7 Sep 2026 16:57:31 +0200 Subject: [PATCH 36/43] chore: move remaining assumptions to testify --- .../cmd/grafana-alertcheck/version_test.go | 19 +++----- .../internal/gate/coverage_test.go | 6 +-- .../internal/gate/flock_test.go | 18 +++----- .../internal/gate/parse_ruler_test.go | 5 +-- .../internal/gate/schedule_test.go | 44 ++++++------------- .../internal/gate/source_test.go | 12 ++--- .../internal/gate/watch_test.go | 33 ++++---------- 7 files changed, 40 insertions(+), 97 deletions(-) diff --git a/grafana-alertcheck/cmd/grafana-alertcheck/version_test.go b/grafana-alertcheck/cmd/grafana-alertcheck/version_test.go index 4122c2f13..9463e990c 100644 --- a/grafana-alertcheck/cmd/grafana-alertcheck/version_test.go +++ b/grafana-alertcheck/cmd/grafana-alertcheck/version_test.go @@ -2,31 +2,24 @@ package main import ( "bytes" - "strings" "testing" + + "github.com/stretchr/testify/require" ) func TestRunVersion(t *testing.T) { var stdout, stderr bytes.Buffer code := run([]string{"version"}, &stdout, &stderr) - if code != 0 { - t.Fatalf("code = %d, want 0", code) - } + require.Equal(t, 0, code) out := stdout.String() for _, want := range []string{"version:", "commit:", "date:", "builtBy:"} { - if !strings.Contains(out, want) { - t.Errorf("stdout = %q, want it to contain %q", out, want) - } - } - if stderr.String() != "" { - t.Errorf("stderr = %q, want empty", stderr.String()) + require.Contains(t, out, want) } + require.Empty(t, stderr.String()) } func TestRunVersion_RejectsArgs(t *testing.T) { var stdout, stderr bytes.Buffer code := run([]string{"version", "extra"}, &stdout, &stderr) - if code != 2 { - t.Fatalf("code = %d, want 2", code) - } + require.Equal(t, 2, code) } diff --git a/grafana-alertcheck/internal/gate/coverage_test.go b/grafana-alertcheck/internal/gate/coverage_test.go index 053cc43d2..a35d7c6fb 100644 --- a/grafana-alertcheck/internal/gate/coverage_test.go +++ b/grafana-alertcheck/internal/gate/coverage_test.go @@ -566,10 +566,8 @@ func TestProveCoverage_FutureLastEvaluationIsUnobservable(t *testing.T) { sentinel := to res := proveCoverage(Header{StartedAt: from.Add(-time.Hour)}, polls, &sentinel, rt, def, from, to, 0) - if res.Reason != ReasonFutureEvaluation { - t.Fatalf("Reason = %q, want future_evaluation: a lastEvaluation in the future of grafana_now must fail "+ - "closed rather than read its negative staleness as fresh", res.Reason) - } + require.Equal(t, ReasonFutureEvaluation, res.Reason, + "a lastEvaluation in the future of grafana_now must fail closed rather than read its negative staleness as fresh") } // --- Check 3, tightened: the boundary segments must widen by the skew bound --- diff --git a/grafana-alertcheck/internal/gate/flock_test.go b/grafana-alertcheck/internal/gate/flock_test.go index 3e9ef268c..3dfdd8b73 100644 --- a/grafana-alertcheck/internal/gate/flock_test.go +++ b/grafana-alertcheck/internal/gate/flock_test.go @@ -5,18 +5,16 @@ import ( "fmt" "syscall" "testing" + + "github.com/stretchr/testify/require" ) func TestIsLockContention(t *testing.T) { contended := []error{syscall.EWOULDBLOCK, syscall.EAGAIN} for _, e := range contended { - if !isLockContention(e) { - t.Errorf("isLockContention(%v) = false, want true", e) - } + require.Truef(t, isLockContention(e), "isLockContention(%v)", e) // lockExclusive wraps the raw error via fmt.Errorf("flock: %w", ...). - if !isLockContention(fmt.Errorf("flock: %w", e)) { - t.Errorf("isLockContention(wrapped %v) = false, want true", e) - } + require.Truef(t, isLockContention(fmt.Errorf("flock: %w", e)), "isLockContention(wrapped %v)", e) } notContended := []error{ @@ -28,11 +26,7 @@ func TestIsLockContention(t *testing.T) { errors.New("something else"), } for _, e := range notContended { - if isLockContention(e) { - t.Errorf("isLockContention(%v) = true, want false (not a contender)", e) - } - if isLockContention(fmt.Errorf("flock: %w", e)) { - t.Errorf("isLockContention(wrapped %v) = true, want false", e) - } + require.Falsef(t, isLockContention(e), "isLockContention(%v)", e) + require.Falsef(t, isLockContention(fmt.Errorf("flock: %w", e)), "isLockContention(wrapped %v)", e) } } diff --git a/grafana-alertcheck/internal/gate/parse_ruler_test.go b/grafana-alertcheck/internal/gate/parse_ruler_test.go index b60dffa68..280dc9910 100644 --- a/grafana-alertcheck/internal/gate/parse_ruler_test.go +++ b/grafana-alertcheck/internal/gate/parse_ruler_test.go @@ -85,7 +85,6 @@ func TestParseDefinitions_Recording(t *testing.T) { // A datasource-managed rule with no alert/record name must fail parsing. func TestParseDefinitions_DatasourceManagedNoName(t *testing.T) { body := []byte(`{"ExampleMetrics":[{"name":"g","rules":[{"expr":"up == 0","for":"5m"}]}]}`) - if _, err := ParseDefinitions(body); err == nil { - t.Fatalf("ParseDefinitions: expected error for datasource-managed rule with no alert/record, got nil") - } + _, err := ParseDefinitions(body) + require.Error(t, err, "a datasource-managed rule with no alert/record must fail") } diff --git a/grafana-alertcheck/internal/gate/schedule_test.go b/grafana-alertcheck/internal/gate/schedule_test.go index 96203e6b3..ace598650 100644 --- a/grafana-alertcheck/internal/gate/schedule_test.go +++ b/grafana-alertcheck/internal/gate/schedule_test.go @@ -1,7 +1,6 @@ package gate import ( - "strings" "testing" "time" @@ -194,13 +193,11 @@ func TestScheduler_MarkUnknownUIDFails(t *testing.T) { next: map[string]time.Time{"r1": now}, every: map[string]time.Duration{"r1": 30 * time.Second}, } - if err := s.Mark("not-a-rule", now); err == nil { - t.Fatalf("Mark of an unknown uid: want error, got nil (a missing cadence must not read as zero and loop)") - } + err := s.Mark("not-a-rule", now) + require.Error(t, err, "a missing cadence must not read as zero and loop") // The failed Mark must not have inserted a bogus next-due entry. - if _, ok := s.next["not-a-rule"]; ok { - t.Errorf("Mark of an unknown uid inserted a next-due entry") - } + _, ok := s.next["not-a-rule"] + require.False(t, ok, "a failed Mark must not insert a next-due entry") } // TestScheduler_PerRuleCadenceOverTime simulates a run and counts how often @@ -223,9 +220,7 @@ func TestScheduler_PerRuleCadenceOverTime(t *testing.T) { now := start.Add(elapsed) for _, uid := range s.Due(now) { counts[uid]++ - if err := s.Mark(uid, now); err != nil { - t.Fatalf("Mark(%q): unexpected error: %v", uid, err) - } + require.NoErrorf(t, s.Mark(uid, now), "Mark(%q)", uid) } } @@ -251,9 +246,8 @@ func TestNewScheduler_StaggersWithinPollEvery(t *testing.T) { func TestScheduler_EarliestDueEmpty(t *testing.T) { s := &Scheduler{next: map[string]time.Time{}, every: map[string]time.Duration{}} - if _, ok := s.earliestDue(); ok { - t.Fatalf("earliestDue on an empty scheduler = ok=true, want false") - } + _, ok := s.earliestDue() + require.False(t, ok) } // A zero next-due time is real, not an empty scheduler. @@ -263,12 +257,8 @@ func TestScheduler_EarliestDueZeroTime(t *testing.T) { every: map[string]time.Duration{"r1": time.Second}, } earliest, ok := s.earliestDue() - if !ok { - t.Fatalf("earliestDue = ok=false, want true (the zero time is a real next-due, not an empty scheduler)") - } - if !earliest.IsZero() { - t.Errorf("earliestDue = %v, want the zero time", earliest) - } + require.True(t, ok, "the zero time is a real next-due, not an empty scheduler") + require.Truef(t, earliest.IsZero(), "earliestDue = %v, want the zero time", earliest) } func TestScheduler_EarliestDuePicksMinimum(t *testing.T) { @@ -281,12 +271,8 @@ func TestScheduler_EarliestDuePicksMinimum(t *testing.T) { every: map[string]time.Duration{"later": time.Minute, "soon": time.Minute}, } earliest, ok := s.earliestDue() - if !ok { - t.Fatalf("earliestDue = ok=false, want true") - } - if !earliest.Equal(now.Add(time.Minute)) { - t.Errorf("earliestDue = %v, want the earliest next-due time", earliest) - } + require.True(t, ok) + require.Truef(t, earliest.Equal(now.Add(time.Minute)), "earliestDue = %v, want the earliest next-due time", earliest) } // One rule at 10s beside twenty at 300s, all measured ~1.8s, must not error at @@ -370,12 +356,8 @@ func TestCheckBudget_NonPositivePollIntervalIsAnError(t *testing.T) { timings := map[string]ruleTimings{"r1": {pollEvery: pe}} measured := map[string]time.Duration{"r1": time.Second} err := CheckBudget(timings, measured, 1) - if err == nil { - t.Fatalf("CheckBudget(pollEvery=%s) = nil, want error (non-positive poll-interval would divide by zero)", pe) - } - if !strings.Contains(err.Error(), "non-positive") { - t.Errorf("error %q does not name the non-positive poll-interval", err.Error()) - } + require.Errorf(t, err, "pollEvery=%s would divide by zero", pe) + require.Contains(t, err.Error(), "non-positive", "the error must name the non-positive poll-interval") } } diff --git a/grafana-alertcheck/internal/gate/source_test.go b/grafana-alertcheck/internal/gate/source_test.go index bdf6e1634..e5f24f46f 100644 --- a/grafana-alertcheck/internal/gate/source_test.go +++ b/grafana-alertcheck/internal/gate/source_test.go @@ -455,17 +455,11 @@ func TestHTTPSource_ResponseBodyTooLarge(t *testing.T) { clock := newFakeClock(time.Now()) src := NewHTTPSource(srv.URL, "", clock) _, err := src.Version(context.Background()) - if err == nil { - t.Fatalf("Version(): want error, got nil (an oversized body must fail loudly)") - } - if !strings.Contains(err.Error(), "exceeded") { - t.Fatalf("error %q does not name the size limit", err.Error()) - } + require.Error(t, err, "an oversized body must fail loudly") + require.Contains(t, err.Error(), "exceeded", "the error must name the size limit") // An oversized body is a stable condition, not a transient one: it must // fail hard on the first attempt, never burning retries re-reading it. - if n := calls.Load(); n != 1 { - t.Fatalf("calls = %d, want 1 — an oversized body must never be retried", n) - } + require.Equal(t, int32(1), calls.Load(), "an oversized body must never be retried") } func TestHTTPSource_NetworkFailureRetries(t *testing.T) { diff --git a/grafana-alertcheck/internal/gate/watch_test.go b/grafana-alertcheck/internal/gate/watch_test.go index 91b52ca4b..41ee3b09b 100644 --- a/grafana-alertcheck/internal/gate/watch_test.go +++ b/grafana-alertcheck/internal/gate/watch_test.go @@ -591,47 +591,30 @@ func TestPidFileRoundTrip(t *testing.T) { func TestDaemonLogTail(t *testing.T) { t.Run("missing file is unreadable", func(t *testing.T) { out := daemonLogTail(filepath.Join(t.TempDir(), "nope.daemon.log"), 0) - if !strings.Contains(out, "unreadable") { - t.Errorf("daemonLogTail = %q, want it to name the file as unreadable", out) - } + require.Contains(t, out, "unreadable") }) t.Run("small file returns its content", func(t *testing.T) { path := filepath.Join(t.TempDir(), "small.daemon.log") - if err := os.WriteFile(path, []byte("line one\nline two\n"), 0o644); err != nil { - t.Fatalf("write: %v", err) - } - if out := daemonLogTail(path, 0); out != "line one\nline two" { - t.Errorf("daemonLogTail = %q, want the full trimmed content", out) - } + require.NoError(t, os.WriteFile(path, []byte("line one\nline two\n"), 0o644)) + require.Equal(t, "line one\nline two", daemonLogTail(path, 0)) }) t.Run("large file keeps only the tail", func(t *testing.T) { path := filepath.Join(t.TempDir(), "large.daemon.log") prefix := strings.Repeat("P", 1000) suffix := strings.Repeat("S", daemonLogTailBytes) - if err := os.WriteFile(path, []byte(prefix+suffix), 0o644); err != nil { - t.Fatalf("write: %v", err) - } - out := daemonLogTail(path, 0) - if out != suffix { - t.Errorf("daemonLogTail = %q, want exactly the trailing %d bytes (the %d leading bytes dropped)", out, daemonLogTailBytes, len(prefix)) - } + require.NoError(t, os.WriteFile(path, []byte(prefix+suffix), 0o644)) + require.Equal(t, suffix, daemonLogTail(path, 0)) }) t.Run("offset skips a previous run's content", func(t *testing.T) { path := filepath.Join(t.TempDir(), "shared.daemon.log") prior := strings.Repeat("p", 2000) - if err := os.WriteFile(path, []byte(prior), 0o644); err != nil { - t.Fatalf("write: %v", err) - } + require.NoError(t, os.WriteFile(path, []byte(prior), 0o644)) from := int64(len(prior)) thisRun := "this run's output\n" - if err := os.WriteFile(path, []byte(prior+thisRun), 0o644); err != nil { - t.Fatalf("write: %v", err) - } - if out := daemonLogTail(path, from); out != "this run's output" { - t.Errorf("daemonLogTail = %q, want only this run's bytes after offset %d", out, from) - } + require.NoError(t, os.WriteFile(path, []byte(prior+thisRun), 0o644)) + require.Equal(t, "this run's output", daemonLogTail(path, from)) }) } From ca3ccc656e3e2c03e204a7e84b334bb57d678db9 Mon Sep 17 00:00:00 2001 From: Bartek Tofel Date: Mon, 7 Sep 2026 16:59:17 +0200 Subject: [PATCH 37/43] chore: address code review comments --- grafana-alertcheck/internal/gate/check.go | 28 ++++++++++++----------- 1 file changed, 15 insertions(+), 13 deletions(-) diff --git a/grafana-alertcheck/internal/gate/check.go b/grafana-alertcheck/internal/gate/check.go index 2e8cd98c9..4c4d6cd2f 100644 --- a/grafana-alertcheck/internal/gate/check.go +++ b/grafana-alertcheck/internal/gate/check.go @@ -236,20 +236,22 @@ func check(ctx context.Context, cfg Config, src Source) (Result, error) { return Result{}, fmt.Errorf("log identity: %w", err) } logHasHdr = true - // Fail fast on a statically-knowable bound violation. `from < - // StartedAt` makes coverage unprovable no matter how healthy the polls - // that DO exist look, and StartedAt is immutable (line 1, written - // first), so this cannot disagree with the authoritative header read - // later. proveCoverage's check 2 remains the backstop against the - // authoritative header, so a bad advisory read can only ever fail - // closed, never produce a false pass. This is recorder mode only: the - // single-step branch has no header, and its own `from < startedAt` is - // a warning-and-pass (see below), not an error. - if from.Before(earlyHdr.StartedAt) { - return Result{}, fmt.Errorf("check: `from` %s is before recording started at %s", - from.Format(time.RFC3339), earlyHdr.StartedAt.Format(time.RFC3339)) - } resolved, notes, err = resolveFromLog(allDefs, earlyHdr, cfg) + if err == nil { + // Fail fast on a statically-knowable bound violation. `from < + // StartedAt` makes coverage unprovable no matter how healthy the polls + // that DO exist look, and StartedAt is immutable (line 1, written + // first), so this cannot disagree with the authoritative header read + // later. proveCoverage's check 2 remains the backstop against the + // authoritative header, so a bad advisory read can only ever fail + // closed, never produce a false pass. This is recorder mode only: the + // single-step branch has no header, and its own `from < startedAt` is + // a warning-and-pass (see below), not an error. + if from.Before(earlyHdr.StartedAt) { + return Result{}, fmt.Errorf("check: `from` %s is before recording started at %s", + from.Format(time.RFC3339), earlyHdr.StartedAt.Format(time.RFC3339)) + } + } } else { resolved, notes, err = Resolve(allDefs, cfg.namedAlerts(), cfg.Folder) } From cb27bc8ba8d7bfb68b6a1e4cbdaab82bc382cc6f Mon Sep 17 00:00:00 2001 From: Bartek Tofel Date: Thu, 3 Sep 2026 13:20:34 +0200 Subject: [PATCH 38/43] chore: fix logging and std out printing --- .../cmd/grafana-alertcheck/check.go | 2 +- .../cmd/grafana-alertcheck/style.go | 125 ++++++++++++++++++ .../cmd/grafana-alertcheck/table.go | 77 ++++++++--- .../cmd/grafana-alertcheck/table_test.go | 12 +- .../cmd/grafana-alertcheck/watch.go | 2 +- grafana-alertcheck/internal/gate/check.go | 21 ++- grafana-alertcheck/internal/gate/schedule.go | 6 +- 7 files changed, 205 insertions(+), 40 deletions(-) create mode 100644 grafana-alertcheck/cmd/grafana-alertcheck/style.go diff --git a/grafana-alertcheck/cmd/grafana-alertcheck/check.go b/grafana-alertcheck/cmd/grafana-alertcheck/check.go index 3f2d295d6..5493ed747 100644 --- a/grafana-alertcheck/cmd/grafana-alertcheck/check.go +++ b/grafana-alertcheck/cmd/grafana-alertcheck/check.go @@ -82,7 +82,7 @@ func runCheck(args []string, stdin io.Reader, stdout, stderr io.Writer) int { PidFile: *pidfile, Concurrency: *common.concurrency, Clock: gate.SystemClock{}, - Notes: stderr, + Notes: newNoteStyler(stderr), } if *to == "" { fmt.Fprintln(stderr, "check: --to is required") diff --git a/grafana-alertcheck/cmd/grafana-alertcheck/style.go b/grafana-alertcheck/cmd/grafana-alertcheck/style.go new file mode 100644 index 000000000..bb46fc2ec --- /dev/null +++ b/grafana-alertcheck/cmd/grafana-alertcheck/style.go @@ -0,0 +1,125 @@ +package main + +import ( + "bytes" + "io" + "os" + "strings" +) + +// ANSI SGR codes for the human-facing notes and table footer. The colours are +// applied only when the destination is a terminal (see colorEnabled); a pipe, +// file or CI log gets plain text, so stdout stays reserved for --output json +// and no machine reader ever sees escape sequences. +const ( + ansiReset = "\x1b[0m" + ansiRed = "\x1b[31m" + ansiGreen = "\x1b[32m" + ansiYellow = "\x1b[33m" + ansiCyan = "\x1b[36m" + // Orange has no entry in the base-16 palette; 256-colour 208 is a legible + // orange used for warnings, distinct from the yellow used for notes. + ansiOrange = "\x1b[38;5;208m" +) + +// colorEnabled reports whether ANSI colour should be written to w. Colour is +// written only when three things hold: NO_COLOR is unset, w is a real *os.File +// (so text/tabwriter buffers, strings.Builder and bytes.Buffer tests all stay +// plain), and that file is a character device (a terminal, not a redirect). +func colorEnabled(w io.Writer) bool { + if os.Getenv("NO_COLOR") != "" { + return false + } + f, ok := w.(*os.File) + if !ok { + return false + } + fi, err := f.Stat() + if err != nil { + return false + } + return fi.Mode()&os.ModeCharDevice != 0 +} + +// styleLine applies the note vocabulary's colour to one line when enabled. The +// colour wraps the text only; the terminating newline is written uncoloured so +// the terminal's line discipline is never inside the escape sequence. +func styleLine(line string, enabled bool) string { + if !enabled { + return line + } + content := strings.TrimRight(line, "\n") + var color string + switch { + case strings.HasPrefix(content, "warning:"): + color = ansiOrange + case strings.HasPrefix(content, "note:"): + color = ansiYellow + case strings.HasPrefix(content, "drain wait:"): + color = ansiCyan + } + if color == "" { + return line + } + return color + content + ansiReset + "\n" +} + +// noteStyler wraps the gate package's Notes stream — a presentation seam that +// keeps colour out of the library. It colourises each line by its known prefix +// and separates the collection countdown from the setup phase with a single +// blank line before the first "collecting:" line. The gate keeps emitting plain +// prose; only the CLI lays it out. +type noteStyler struct { + w io.Writer + enabled bool + pending []byte + sawCollecting bool +} + +func newNoteStyler(w io.Writer) *noteStyler { + return ¬eStyler{w: w, enabled: colorEnabled(w)} +} + +// startsSection reports whether a line opens a new phase of the stream and so +// deserves a blank line above it. "collecting:" opens the countdown (once — +// later countdown lines follow on from the first), and "drain wait:" opens the +// drain phase. The setup lines (planned run time, warning, min-observed, notes) +// are one contiguous block and are not separated from each other. +func (s *noteStyler) startsSection(line string) bool { + switch { + case strings.HasPrefix(line, "warning:"): + return true + case strings.HasPrefix(line, "drain wait:"): + return true + case strings.HasPrefix(line, "collecting:"): + if s.sawCollecting { + return false + } + s.sawCollecting = true + return true + } + return false +} + +func (s *noteStyler) Write(p []byte) (int, error) { + n := len(p) + s.pending = append(s.pending, p...) + for { + i := bytes.IndexByte(s.pending, '\n') + if i < 0 { + break + } + line := string(s.pending[:i+1]) + s.pending = s.pending[i+1:] + + if s.startsSection(line) { + if _, err := io.WriteString(s.w, "\n"); err != nil { + return n, err + } + } + if _, err := io.WriteString(s.w, styleLine(line, s.enabled)); err != nil { + return n, err + } + } + return n, nil +} diff --git a/grafana-alertcheck/cmd/grafana-alertcheck/table.go b/grafana-alertcheck/cmd/grafana-alertcheck/table.go index 11b72a074..39ec75cbc 100644 --- a/grafana-alertcheck/cmd/grafana-alertcheck/table.go +++ b/grafana-alertcheck/cmd/grafana-alertcheck/table.go @@ -14,29 +14,39 @@ import ( // which the caller (runCheck) always points at stderr — stdout is reserved for // the machine-readable --output json. // -// Three sections, in order: +// Three titled tables, in order (the name column is RULE in all of them — one +// row is one resolved alert rule, never a firing instance): // -// 1. one line per rule: outcome, BadFor, pollEvery, proved-or-not with the -// largest gap; -// 2. one line per Violation: a rule's worst-of outcome does not carry the -// State/Health of the instance that actually caused it — Violation does — -// so this is also where those two columns appear, sorted after the rule -// table rather than folded into it, and it is the only place an operator -// running WITHOUT --output json sees the --allow-paused hint that -// Violation.Note already carries (classify.go); -// 3. a footer with the numbers that answer "why" on exit 2: each non-skipped -// rule's maxGap/healthGrace/evalStaleAfter, the global transitionGrace and -// drainTimeout, and the largest measured clock skew alongside its own -// error bound (RTT/2) — SkewHardLimit is a separate, fixed input threshold -// and is reported next to it, never as if it were that bound. +// 1. RESULTS, one line per rule: outcome, BadFor, pollEvery, proved-or-not +// with the largest gap; +// 2. VIOLATIONS, one line per Violation (only when any): a rule's worst-of +// outcome does not carry the State/Health of the instance that actually +// caused it — Violation does — so this is also where those two columns +// appear, sorted after the result table rather than folded into it, and it +// is the only place an operator running WITHOUT --output json sees the +// --allow-paused hint that Violation.Note already carries (classify.go); +// 3. THRESHOLDS, the numbers that answer "why" on exit 2: each non-skipped +// rule's maxGap/healthGrace/evalStaleAfter, followed by the global +// transitionGrace and drainTimeout, and the largest measured clock skew +// alongside its own error bound (RTT/2) — SkewHardLimit is a separate, +// fixed input threshold and is reported next to it, never as if it were +// that bound. func renderTable(w io.Writer, res gate.Result) error { alertOf := make(map[string]string, len(res.Verdicts)) for _, v := range res.Verdicts { alertOf[v.RuleUID] = v.Alert } + enabled := colorEnabled(w) + // A blank line separates the result table from the notes the gate streamed + // before it (planned run time, warning, min-observed, collecting, drain + // wait), so the verdict reads as its own section rather than the tail of a + // wall of progress text. + fmt.Fprintln(w) + + fmt.Fprintln(w, "RESULTS") tw := tabwriter.NewWriter(w, 0, 4, 2, ' ', 0) - fmt.Fprintln(tw, "ALERT\tOUTCOME\tBADFOR\tPOLLEVERY\tPROVED\tNOTE") + fmt.Fprintln(tw, "RULE\tOUTCOME\tBADFOR\tPOLLEVERY\tPROVED\tNOTE") for _, v := range sortedVerdicts(res.Verdicts) { fmt.Fprintf(tw, "%s\t%s\t%s\t%s\t%s\t%s\n", v.Alert, v.Outcome, v.BadFor.Round(time.Second), v.PollEvery.Round(time.Second), @@ -49,7 +59,7 @@ func renderTable(w io.Writer, res gate.Result) error { if len(res.Violations) > 0 { fmt.Fprintln(w, "\nVIOLATIONS") vtw := tabwriter.NewWriter(w, 0, 4, 2, ' ', 0) - fmt.Fprintln(vtw, "ALERT\tOUTCOME\tSTATE\tHEALTH\tNOTE") + fmt.Fprintln(vtw, "RULE\tOUTCOME\tSTATE\tHEALTH\tNOTE") for _, v := range sortedViolations(res.Violations) { fmt.Fprintf(vtw, "%s\t%s\t%s\t%s\t%s\n", alertLabel(v, alertOf), v.Outcome, v.State, v.Health, v.Note) } @@ -58,20 +68,49 @@ func renderTable(w io.Writer, res gate.Result) error { } } + // The per-rule thresholds answer "why" on exit 2: a table, not the prose + // "rule NAME: maxGap=... healthGrace=... evalStaleAfter=..." that repeated + // the rule name a fourth time. It is separated from the result above by a + // blank line. fmt.Fprintln(w) + fmt.Fprintln(w, "THRESHOLDS") + ttw := tabwriter.NewWriter(w, 0, 4, 2, ' ', 0) + fmt.Fprintln(ttw, "RULE\tMAXGAP\tHEALTHGRACE\tEVALSTALEAFTER") for _, uid := range sortedThresholdUIDs(res.Thresholds, alertOf) { t := res.Thresholds[uid] - fmt.Fprintf(w, "rule %s: maxGap=%s healthGrace=%s evalStaleAfter=%s\n", + fmt.Fprintf(ttw, "%s\t%s\t%s\t%s\n", alertOr(uid, alertOf), t.MaxGap, t.HealthGrace, t.EvalStaleAfter) } + if err := ttw.Flush(); err != nil { + return fmt.Errorf("render table: %w", err) + } + + fmt.Fprintln(w) fmt.Fprintf(w, "global: transitionGrace=%s (source: %s) drainTimeout=%s\n", res.Global.TransitionGrace, res.Global.GraceSource, res.Global.DrainTimeout) - fmt.Fprintf(w, "violations: %d, largest measured clock skew: %s (bound ±%s, hard limit %s), grafana %s\n", - len(res.Violations), res.ClockSkew.Round(time.Millisecond), res.ClockSkewBound.Round(time.Millisecond), + fmt.Fprintf(w, "largest measured clock skew: %s (bound ±%s, hard limit %s), grafana %s\n", + res.ClockSkew.Round(time.Millisecond), res.ClockSkewBound.Round(time.Millisecond), gate.SkewHardLimit, res.GrafanaVersion) + // The verdict — the single number a terminal operator reads last — sits on + // its own line at the very bottom, separated from the diagnostics above and + // from the shell prompt below. + fmt.Fprintf(w, "\n%s\n\n", violationsLabel(len(res.Violations), enabled)) return nil } +// violationsLabel colours the "violations: N" prefix of the footer: green for a +// clean run, red otherwise. The rest of the line is written uncoloured. +func violationsLabel(n int, enabled bool) string { + s := fmt.Sprintf("violations: %d", n) + if !enabled { + return s + } + if n == 0 { + return ansiGreen + s + ansiReset + } + return ansiRed + s + ansiReset +} + // provedLabel is the table's PROVED column: "yes" for a clean coverage // proof, "no" with the reason and largest gap for an unobservable rule, and // "-" for a rule decide never asked proveCoverage about at all (skipped — diff --git a/grafana-alertcheck/cmd/grafana-alertcheck/table_test.go b/grafana-alertcheck/cmd/grafana-alertcheck/table_test.go index 9da9ea3ba..86c80a19b 100644 --- a/grafana-alertcheck/cmd/grafana-alertcheck/table_test.go +++ b/grafana-alertcheck/cmd/grafana-alertcheck/table_test.go @@ -67,11 +67,13 @@ func TestRenderTable(t *testing.T) { require.Contains(t, out, string(gate.StateFiring)) require.Contains(t, out, "error") - // The footer: per-rule thresholds, global thresholds, and skew with its own - // bound rather than the fixed hard limit. - require.Contains(t, out, "Ape Alert: maxGap=1m0s healthGrace=2m0s evalStaleAfter=1m0s") - require.Contains(t, out, "Zebra Alert: maxGap=1m0s healthGrace=1m0s evalStaleAfter=1m0s") - require.NotContains(t, out, "Paused Alert: maxGap") + // The footer: per-rule thresholds are a table (RULE/MAXGAP/HEALTHGRACE/ + // EVALSTALEAFTER) rather than prose, followed by the global thresholds and + // the violations count with the skew and its own bound rather than the + // fixed hard limit. + require.Contains(t, out, "MAXGAP") + require.Contains(t, out, "HEALTHGRACE") + require.Contains(t, out, "EVALSTALEAFTER") require.Contains(t, out, "global: transitionGrace=5m0s (source: Ape Alert (for=5m)) drainTimeout=2m0s") require.Contains(t, out, "largest measured clock skew: 1.5s (bound ±250ms, hard limit 1m0s)") require.Contains(t, out, "violations: 2") diff --git a/grafana-alertcheck/cmd/grafana-alertcheck/watch.go b/grafana-alertcheck/cmd/grafana-alertcheck/watch.go index df2c50029..d6db86672 100644 --- a/grafana-alertcheck/cmd/grafana-alertcheck/watch.go +++ b/grafana-alertcheck/cmd/grafana-alertcheck/watch.go @@ -78,7 +78,7 @@ func runWatch(args []string, stdin io.Reader, stdout, stderr io.Writer) int { DaemonLog: *daemonLog, Concurrency: *common.concurrency, Clock: gate.SystemClock{}, - Notes: stderr, + Notes: newNoteStyler(stderr), } if *until != "" { t, err := time.Parse(time.RFC3339, *until) diff --git a/grafana-alertcheck/internal/gate/check.go b/grafana-alertcheck/internal/gate/check.go index 4c4d6cd2f..0ca528e16 100644 --- a/grafana-alertcheck/internal/gate/check.go +++ b/grafana-alertcheck/internal/gate/check.go @@ -286,6 +286,16 @@ func check(ctx context.Context, cfg Config, src Source) (Result, error) { } summary, warning := StartupSummary(from, cfg.To, gt) fmt.Fprintln(cfg.Notes, summary) + // MinObserved is printed with the plan, beside "planned run time", rather + // than after it: it is a fact about the run, not a diagnostic. Its default + // is the resolved rule count AFTER duplicate names collapse, which is + // len(resolved) by construction; decide defaults it identically, and it is + // resolved here rather than inferred from the verdict afterwards. + minObserved := cfg.MinObserved + if minObserved == 0 { + minObserved = len(resolved) + } + fmt.Fprintf(cfg.Notes, "min-observed: %d of %d resolved rule(s)\n", minObserved, len(resolved)) if warning != "" { fmt.Fprintf(cfg.Notes, "warning: %s\n", warning) } @@ -336,17 +346,6 @@ func check(ctx context.Context, cfg Config, src Source) (Result, error) { } } - // ---- Apply MinObserved. ----------------------------------------------- - // Its default is the resolved rule count AFTER duplicate names collapse, - // which is len(resolved) by construction. decide defaults it identically; - // it is resolved here as well so the value the run will judge against is - // printed before the wait rather than inferred from the verdict afterwards. - minObserved := cfg.MinObserved - if minObserved == 0 { - minObserved = len(resolved) - } - fmt.Fprintf(cfg.Notes, "min-observed: %d of %d resolved rule(s)\n", minObserved, len(resolved)) - // ---- Collect the evidence. -------------------------------------------- // Collect ONLY. No classification happens here and there is no early exit, // even once a violation is certain: the loop always runs to diff --git a/grafana-alertcheck/internal/gate/schedule.go b/grafana-alertcheck/internal/gate/schedule.go index f2b9bd759..ae366baf4 100644 --- a/grafana-alertcheck/internal/gate/schedule.go +++ b/grafana-alertcheck/internal/gate/schedule.go @@ -352,12 +352,12 @@ func StartupSummary(from, to time.Time, global globalTimings) (summary, warning source = "none" } summary = fmt.Sprintf( - "planned run time: %s (window %s + transitionGrace %s [source: %s] + drainTimeout %s)", - total, window, global.transitionGrace, source, global.drainTimeout) + "planned run time: %s\n window %s + transitionGrace %s + drainTimeout %s\n transitionGrace source: %s", + total, window, global.transitionGrace, global.drainTimeout, source) if window > 0 && float64(global.transitionGrace) > float64(window)*graceWarnFraction { warning = fmt.Sprintf( - "transitionGrace %s is more than %.0f%% of the window %s (source: %s) — the window may be too short for this alert's `for`", + "transitionGrace %s is more than %.0f%% of the window %s — the window may be too short for this alert's `for`\n source: %s", global.transitionGrace, graceWarnFraction*100, window, source) } return summary, warning From f25698380ecc6eb21065d375ed829e64977c1a35 Mon Sep 17 00:00:00 2001 From: Bartek Tofel Date: Fri, 4 Sep 2026 14:13:31 +0200 Subject: [PATCH 39/43] chore: truncate to seconds when comparing from time --- grafana-alertcheck/internal/gate/check.go | 13 ++--- .../internal/gate/check_test.go | 26 +++++++++ grafana-alertcheck/internal/gate/coverage.go | 14 +++-- .../internal/gate/coverage_test.go | 57 +++++++++++++++++++ 4 files changed, 96 insertions(+), 14 deletions(-) diff --git a/grafana-alertcheck/internal/gate/check.go b/grafana-alertcheck/internal/gate/check.go index 0ca528e16..ef6543b71 100644 --- a/grafana-alertcheck/internal/gate/check.go +++ b/grafana-alertcheck/internal/gate/check.go @@ -238,15 +238,10 @@ func check(ctx context.Context, cfg Config, src Source) (Result, error) { logHasHdr = true resolved, notes, err = resolveFromLog(allDefs, earlyHdr, cfg) if err == nil { - // Fail fast on a statically-knowable bound violation. `from < - // StartedAt` makes coverage unprovable no matter how healthy the polls - // that DO exist look, and StartedAt is immutable (line 1, written - // first), so this cannot disagree with the authoritative header read - // later. proveCoverage's check 2 remains the backstop against the - // authoritative header, so a bad advisory read can only ever fail - // closed, never produce a false pass. This is recorder mode only: the - // single-step branch has no header, and its own `from < startedAt` is - // a warning-and-pass (see below), not an error. + // Fail fast on a bound violation that can't change: StartedAt is + // immutable (line 1), so check 2's backstop still catches any bad + // advisory read — fail closed, never false-pass. Recorder mode only; + // single-step warns-and-passes (see below). if from.Before(earlyHdr.StartedAt) { return Result{}, fmt.Errorf("check: `from` %s is before recording started at %s", from.Format(time.RFC3339), earlyHdr.StartedAt.Format(time.RFC3339)) diff --git a/grafana-alertcheck/internal/gate/check_test.go b/grafana-alertcheck/internal/gate/check_test.go index 6029d5bb7..780184bd3 100644 --- a/grafana-alertcheck/internal/gate/check_test.go +++ b/grafana-alertcheck/internal/gate/check_test.go @@ -657,6 +657,32 @@ func TestCheckFailFastWhenFromPrecedesRecordStart(t *testing.T) { require.True(t, clock.Now().Equal(testNow), "it must fail before the wait") } +// A whole-second `from` in the same second as the recording's sub-second +// StartedAt is not a blind interval: the whole-second comparison lets the run +// proceed to a clean pass instead of the fail-fast above. +func TestCheckRecorderModeFromSameSecondAsStartedAtPasses(t *testing.T) { + dir := t.TempDir() + windowEnd := testNow.Add(5*time.Minute + checkGrace) + // StartedAt is 500ms after `from` (testNow via recorderConfig) — the same + // whole second. Polls still cover the whole window. + logPath := recordedLog(t, dir, "https://grafana.example.com", + testNow.Add(500*time.Millisecond), testNow.Add(-time.Minute), windowEnd.Add(30*time.Second), windowEnd.Add(30*time.Second), 0) + writePid(t, logPath+".pid", fmt.Sprintf("%d\n", deadPid(t))) + + clock := newVirtualClock(testNow.Add(time.Minute)) + cfg := recorderConfig(t, clock, logPath) + src := newCheckSource(func(title string, _ int) (Observation, error) { + require.Fail(t, fmt.Sprintf("the drain wait polled %q although the log already proves the evaluations", title)) + return Observation{}, errors.New("unexpected poll") + }) + + res, err := check(context.Background(), cfg, src) + require.NoError(t, err) + require.Empty(t, res.Violations) + require.Len(t, res.Verdicts, 1) + require.Equal(t, OutcomeClean, res.Verdicts[0].Outcome) +} + // The coverage proof failed: a hole in the middle of the recording is not // saved by healthy data at both ends. func TestCheckFailClosedOnCoverageGap(t *testing.T) { diff --git a/grafana-alertcheck/internal/gate/coverage.go b/grafana-alertcheck/internal/gate/coverage.go index 527720bfe..535af07d2 100644 --- a/grafana-alertcheck/internal/gate/coverage.go +++ b/grafana-alertcheck/internal/gate/coverage.go @@ -88,12 +88,16 @@ func proveCoverage(h Header, polls []Poll, sentinel *time.Time, t ruleTimings, d // Check 2 — from bounds: from < StartedAt makes coverage unprovable, no // matter how healthy the polls that DO exist look. Both are runner-domain // clock reads (the recorder's own Clock.Now()), so no cross-domain - // translation applies here. The other half of the bound — from too far - // ahead of the runner's clock — is Check's input validation, once per run - // rather than per rule. - if from.Before(h.StartedAt) { + // translation applies here. The comparison is at whole-second granularity: + // `from` is supplied at second precision (--from RFC3339) while StartedAt + // carries the recorder's sub-second clock stamp, so an operator naming the + // exact second the recording opened must not be judged early for the + // sub-second sliver inside that same second. The other half of the bound — + // from too far ahead of the runner's clock — is Check's input validation, + // once per run rather than per rule. + if from.Truncate(time.Second).Before(h.StartedAt.Truncate(time.Second)) { fail(ReasonFromBeforeRecord, fmt.Sprintf( - "requested from %s is before recording started at %s", from.Format(time.RFC3339), h.StartedAt.Format(time.RFC3339))) + "requested from %s is before recording started at %s", from.Format(time.RFC3339Nano), h.StartedAt.Format(time.RFC3339Nano))) } // Filtered once and threaded through every remaining check. diff --git a/grafana-alertcheck/internal/gate/coverage_test.go b/grafana-alertcheck/internal/gate/coverage_test.go index a35d7c6fb..3507ee4fc 100644 --- a/grafana-alertcheck/internal/gate/coverage_test.go +++ b/grafana-alertcheck/internal/gate/coverage_test.go @@ -126,6 +126,63 @@ func TestProveCoverage_FromBeforeRecordIsUnobservable(t *testing.T) { require.Equal(t, OutcomeUnobservable, dres.Verdicts[0].Outcome) } +// The from-bounds check compares at whole-second granularity: a whole-second +// `from` may precede the recorder's sub-second StartedAt INSIDE the same second +// without being judged early. That one sliver is the --from truncation, not a +// blind interval, so the window is still proved. +func TestProveCoverage_FromSameSecondAsStartedAtIsProved(t *testing.T) { + from := time.Date(2026, 1, 1, 0, 0, 0, 0, time.UTC) + to := from.Add(10 * time.Minute) + started := from.Add(500 * time.Millisecond) + rt := newRuleTimings(30*time.Second, 60) + def := Definition{UID: "r1", Title: "R1"} + + var polls []Poll + for ts := from; !ts.After(to); ts = ts.Add(30 * time.Second) { + polls = append(polls, Poll{RuleUID: "r1", GrafanaNow: ts, Found: true, Health: "ok", State: "inactive", LastEvaluation: ts}) + } + sentinel := to + + res := proveCoverage(Header{StartedAt: started}, polls, &sentinel, rt, def, from, to, 0) + require.True(t, res.Proved) + require.False(t, res.Unobservable) + require.Empty(t, res.Reason) +} + +// Exactly one whole second later is a different second: even at the boundary, +// the whole-second comparison reads it as before, however healthy the polls. +func TestProveCoverage_FromExactlyOneSecondBeforeStartedAtIsUnobservable(t *testing.T) { + from := time.Date(2026, 1, 1, 0, 0, 0, 0, time.UTC) + to := from.Add(10 * time.Minute) + started := from.Add(time.Second) + rt := newRuleTimings(30*time.Second, 60) + def := Definition{UID: "r1", Title: "R1"} + + sentinel := to + res := proveCoverage(Header{StartedAt: started}, nil, &sentinel, rt, def, from, to, 0) + require.Equal(t, ReasonFromBeforeRecord, res.Reason) + require.True(t, res.Unobservable) + require.False(t, res.Proved) +} + +// A sub-second sliver that straddles the second boundary is still "before": +// 900ms into one second vs 100ms into the next are distinct seconds, so the +// 200ms gap is a from_before_record, not rounding noise. +func TestProveCoverage_FromSubSecondEarlierAcrossSecondBoundaryIsUnobservable(t *testing.T) { + base := time.Date(2026, 1, 1, 0, 0, 0, 0, time.UTC) + from := base.Add(900 * time.Millisecond) + started := base.Add(time.Second + 100*time.Millisecond) + to := from.Add(10 * time.Minute) + rt := newRuleTimings(30*time.Second, 60) + def := Definition{UID: "r1", Title: "R1"} + + sentinel := to + res := proveCoverage(Header{StartedAt: started}, nil, &sentinel, rt, def, from, to, 0) + require.Equal(t, ReasonFromBeforeRecord, res.Reason) + require.True(t, res.Unobservable) + require.False(t, res.Proved) +} + // --- Check 3: heartbeat continuity --- // The core heartbeat regression: data at both ends with a hole between is not From 570dcba526fec09fb3e66b495129e005ef7140c2 Mon Sep 17 00:00:00 2001 From: Bartek Tofel Date: Fri, 4 Sep 2026 17:44:06 +0200 Subject: [PATCH 40/43] chore: add centralized docs --- grafana-alertcheck/docs/_category_.yaml | 8 ++ grafana-alertcheck/docs/advanced.md | 41 +++++++++ .../docs/how-alerts-are-evaluated.md | 85 +++++++++++++++++ grafana-alertcheck/docs/index.md | 83 +++++++++++++++++ .../docs/reference/_category_.yaml | 8 ++ grafana-alertcheck/docs/reference/cli.md | 91 +++++++++++++++++++ 6 files changed, 316 insertions(+) create mode 100644 grafana-alertcheck/docs/_category_.yaml create mode 100644 grafana-alertcheck/docs/advanced.md create mode 100644 grafana-alertcheck/docs/how-alerts-are-evaluated.md create mode 100644 grafana-alertcheck/docs/index.md create mode 100644 grafana-alertcheck/docs/reference/_category_.yaml create mode 100644 grafana-alertcheck/docs/reference/cli.md diff --git a/grafana-alertcheck/docs/_category_.yaml b/grafana-alertcheck/docs/_category_.yaml new file mode 100644 index 000000000..3cbc431c4 --- /dev/null +++ b/grafana-alertcheck/docs/_category_.yaml @@ -0,0 +1,8 @@ +position: 1 +label: 'Grafana Alertcheck' +collapsible: true +collapsed: false +link: + type: generated-index + slug: /platform-services/devex/cicd/grafana-alertcheck/index + description: 'CD quality gate for Grafana alerts: watch, classify, and gate releases.' diff --git a/grafana-alertcheck/docs/advanced.md b/grafana-alertcheck/docs/advanced.md new file mode 100644 index 000000000..073292cd6 --- /dev/null +++ b/grafana-alertcheck/docs/advanced.md @@ -0,0 +1,41 @@ +--- +id: grafana-alertcheck-advanced +title: Check budget and scheduling +sidebar_label: Budget and scheduling +sidebar_position: 2 +description: Why grafana-alertcheck schedules per rule, how the request budget works, and why it never queries state history. +--- + +# Check budget and scheduling + +## Per-rule schedules, never a global cycle + +Each rule polls at its **own** cadence, `--poll-interval` (default: half the rule's own evaluation interval). There is deliberately no single global minimum-interval cycle. + +One rule at `intervalSeconds=10` beside twenty at `300` keeps a 5 s cadence for itself and 150 s for the other twenty — not a 5 s cycle for all of them, which would be a 60× request bloat at ~1.8 s per request and would fail to start on a reasonable fleet. + +The scheduler staggers each rule's initial next-due time across its cadence, and serves due rules **earliest-due-first**, so a tight rule never queues behind slack ones. + +## The check budget + +The gate records one observation of every rule up front and checks the schedule against those **measured** latencies (payload sizes vary ~230× across rules, so a fixed estimate is meaningless). It errors at start — before waiting — if any of three conditions hold: + +- **Utilization** — total request rate exceeds `--concurrency`. +- **Per-rule** — one rule's request can't fit its own cadence. +- **Burst bound** — the slowest request exceeds the fleet's tightest cadence, which can open a mid-run gap. + +The error names the three levers only: raise `--concurrency`, raise `--poll-interval`, or watch fewer alerts. It never prescribes a single interval. + +## Why the gate never queries state history + +Querying Grafana's alert state history after the fact fails closed *in the wrong direction* — it returns "pass" when the truth is unknown: + +- History stores **transitions**, not states. An alert firing through the whole window has its only record *before* the window. +- The annotations API **does not serve Loki-backed history** at all. +- An empty result is indistinguishable from a healthy one: no alert fired, the backend differs, retention removed data, the token lacked permission — all look identical. +- There is **no coverage signal** — nothing proves the history is complete to time T. +- Artifact transitions (`Paused`, `RuleDeleted`, `Updated`, `MissingSeries`) look like recoveries. + +Instead, `watch` records its own evidence live and the log becomes the source of truth. The trade-off: the gate can miss an episode shorter than a rule's poll interval, though `activeAt` still surfaces sub-interval onsets for instances still active at a poll. + +A corollary of recording fresh: there is no replay. Re-running a failed job is a new deploy with a new `from` and a new recording — never a re-classification of old evidence. diff --git a/grafana-alertcheck/docs/how-alerts-are-evaluated.md b/grafana-alertcheck/docs/how-alerts-are-evaluated.md new file mode 100644 index 000000000..7c9cb3b6a --- /dev/null +++ b/grafana-alertcheck/docs/how-alerts-are-evaluated.md @@ -0,0 +1,85 @@ +--- +id: grafana-alertcheck-evaluation +title: How alerts are evaluated +sidebar_label: How alerts are evaluated +sidebar_position: 1 +description: The verdict model, instance timelines, and coverage proof behind grafana-alertcheck. +--- + +# How alerts are evaluated + +Both `watch`+`check` (recorder mode) and `check` alone (single-step mode) converge on the same input: a flat list of polls. Everything below runs over that list; the mode only changes where the polls came from. + +## Instance states + +Grafana reports instance states in two vocabularies (`Alerting`/`Normal` at instance level, `firing`/`inactive` at rule level). The gate normalizes every instance to one canonical set: + +| Canonical | Meaning | +| --------- | ------- | +| `normal` | Healthy | +| `firing` | The condition is true and `for` has elapsed | +| `pending` | The condition is true, `for` has not elapsed | +| `nodata` | The query returned no series (synthetic instance) | +| `error` | The query failed (synthetic instance) | + +A rule's **rule-level** `state` and `health` are kept verbatim and only reported — they are never classified. The **instance** state is what the classifier reasons about. + +A "bad" instance is one whose canonical state is in `--states` (default `firing`). `pending` and `nodata` are excluded by default. + +## Verdict model + +For each instance the gate builds a timeline of bad spans over `[from, to]`, then takes the worst outcome across a rule's instances as the rule's outcome. + +| Outcome | Shape | Exit | +| ------- | ----- | ---- | +| `clean` | Good throughout, observed throughout | → 0 | +| `newly_bad` | Entered a bad state **inside** the window | → 1 | +| `persistently_bad` | Bad at `from`, still bad at `to` | → 1 | +| `recovered` | Bad at `from`, cleared before `to`, stayed clear | → 0 | +| `flapping` | Cleared, then became bad again | → 1 | +| `skipped` | Paused **before** the window opened | reported, not observable | +| `unobservable` | Coverage gap / sustained `health=error` / stale / absent | → 2 | + +`recovered` has **no deadline** — an alert that clears at minute 58 of a 60-minute window still passes. The total bad time is reported as `BadFor`; the removed deadline is replaced by that measured value rather than a derived limit. + +### Preexisting policy + +For an instance already bad when `from` opened, `--preexisting` decides: + +- `fail-unless-recovered` (default) — clears and stays clear → pass; never clears → fail. +- `fail` — any preexisting instance fails, recovered or not. +- `ignore` — preexisting instances are disregarded; only new episodes fail. + +## Cleared vs vanished + +When an instance leaves the bad set, the gate looks it up **in the same response**: + +- Present as `normal` → `cleared` (a real recovery). +- Absent, or present as `normal (MissingSeries)` → `vanished` (a discontinuity, **not** a recovery). + +A vanished instance that was bad stays `persistently_bad`. A metric that stops being emitted is not evidence of health — this is deliberate and can surprise users whose fix is to remove a metric rather than drive it to a good value. + +## Coverage proof + +Before classifying, `check` must **prove** continuous coverage of `[from, to]` for each alert. Nine checks run; any failure makes the rule `unobservable`: + +1. **Sentinel** — a clean recorder stop, timestamped at or after `to + transitionGrace`. A recorder that died mid-window looks exactly like a coverage gap and is one. +2. **`from` bounds** — `from` earlier than the recording start is unprovable. +3. **Heartbeat gap** — any gap larger than `maxGap` (= 2 × poll cadence) inside the window. Data at both ends with a hole between is not enough. +4. **`health=error`** — a contiguous run longer than `healthGrace` consumes coverage; a short blip is a note. +5. **`health=nodata`** — a note, never fatal (unless `--nodata-is-unobservable`). +6. **Liveness** — `grafana_now − lastEvaluation` must not exceed `evalStaleAfter`. This is an **absolute** check, never a "did it increase since the last poll" delta. +7. **In-window pause** — a poll reporting `isPaused` mid-window is `unobservable` (the primary pause detector). +8. **Rule absent** — an authoritative `2xx` with no matching rule. +9. **`KeepLast`** — a note naming a stale-state blind spot. + +## Health: `error` vs `nodata` + +- `health=error` means the query **failed** — a malfunction. Sustained past `healthGrace`, it makes the rule `unobservable`. +- `health=nodata` means the query **ran and returned no series** — indistinguishable from a quiet system. It is not fatal by default; most of a fleet runs `no_data_state: OK`. + +## The drain wait and `transitionGrace` + +A condition that arises just before `to` becomes `firing` only at the first evaluation after its `for` elapses. `transitionGrace` (derived from the watched rules' `for` values) extends the classification bound past `to` so such a surfacing condition is caught. After collection, a **drain wait** polls until each rule has evaluated through `to + transitionGrace` (bounded by `drainTimeout`); a rule that never does is `unobservable`. + +Run time = `(to − from) + transitionGrace + drainTimeout`. This is printed at start, and the grace is warned about when it exceeds a quarter of the window — the window may be too short for the alert's `for`. diff --git a/grafana-alertcheck/docs/index.md b/grafana-alertcheck/docs/index.md new file mode 100644 index 000000000..3e5bee11f --- /dev/null +++ b/grafana-alertcheck/docs/index.md @@ -0,0 +1,83 @@ +--- +id: grafana-alertcheck-index +title: Grafana Alertcheck +sidebar_label: Overview +sidebar_position: 0 +description: A CD quality gate that bookends a release with alert-state observation and answers whether any watched Grafana alert was bad during the release window. +--- + +# Grafana Alertcheck + +`grafana-alertcheck` is a CD quality gate for Grafana alerts. It bookends a release with two commands — `watch` (record) and `check` (classify) — and answers one question: + +> A release finished at time T. Was any of these Grafana alerts in a bad state during the next N minutes? + +The contract is `watch → your work → check`. Between the two you run whatever you want (deploy, tests, migration); the gate only observes, then classifies. + +It **fails closed**: if it cannot get an answer, it stops the release. It never passes an unproven window. + +## How it works, in one paragraph + +`watch` starts a background recorder that polls each named alert and appends snapshots to a JSONL log. Your work then emits two RFC3339 timestamps — `from` (when the change landed) and `to` (when the work ended). `check` proves continuous coverage of `[from, to]`, builds a state timeline per alert, classifies it, and exits `0`, `1`, or `2`. + +## Install + +```bash +go install github.com/smartcontractkit/chainlink-testing-framework/grafana-alertcheck/cmd/grafana-alertcheck@latest +``` + +Connection details come from the environment — the token is env-only, never a flag: + +```bash +export GRAFANA_URL=https://grafana.example.com +export GRAFANA_TOKEN=… +``` + +Requires Grafana >= 13.0.0 and < 14.0.0. Outside that range the gate exits `2`. + +## Quickstart — recorder mode + +```bash +grafana-alertcheck watch --out /tmp/run.jsonl --alerts alerts.txt +./deploy.sh # emits deployed_at= when the rollout is stable +./verify.sh # emits finished_at= when the work is done +grafana-alertcheck check --in /tmp/run.jsonl --from "$deployed_at" --to "$finished_at" +``` + +`alerts.txt` holds one alert name per line. See [Naming alerts](./reference/cli#naming-alerts). + +`watch` returns only after the recorder has observed every named, non-paused alert once and reported ready — so auth, name-resolution, and parse failures surface **before** your deploy runs. + +## Quickstart — single-step mode + +Skip the recorder and observe the window inline, from inside `check` itself: + +```bash +grafana-alertcheck check --alerts alerts.txt --to "$finished_at" +``` + +In single-step mode the window starts at `check`'s first observation; if you give no `--from`, the interval before that first observation is declared as a blind spot with a warning (not an error). + +## Exit codes + +| Code | Meaning | +| ---- | ------- | +| `0` | Pass — no violations | +| `1` | Violations (including a paused-rule-only `--min-observed` shortfall) | +| `2` | The gate could not check — config, auth, resolution, a coverage gap, health/staleness, the drain limit, transport, … | + +An error is never a pass: `2` wins over any violation found alongside it. + +## Common surprises + +- A **paused rule fails by default** — even one someone else paused. Use `--allow-paused`. +- A fix that **stops emitting a metric is not a recovery** — the instance vanishes, which is a discontinuity, not health. +- The gate checks alert **state and health**, not notification delivery — a silenced alert that still fires fails. +- `recovered` has **no deadline** — a bad-at-`from` alert that clears by `to` passes; set `--preexisting fail` to forbid it. +- A **retry is a new deploy**, not a replay — re-running the job re-records against a new `from`. + +## More + +- [How alerts are evaluated](./how-alerts-are-evaluated) — the verdict model and coverage proof +- [Check budget and scheduling](./advanced) — why the schedule and budget look the way they do, and why history isn't queried +- [CLI reference](./reference/cli) — every subcommand and flag diff --git a/grafana-alertcheck/docs/reference/_category_.yaml b/grafana-alertcheck/docs/reference/_category_.yaml new file mode 100644 index 000000000..2e5d50946 --- /dev/null +++ b/grafana-alertcheck/docs/reference/_category_.yaml @@ -0,0 +1,8 @@ +position: 3 +label: Reference +collapsible: true +collapsed: false +link: + type: generated-index + slug: /platform-services/devex/cicd/grafana-alertcheck/reference + description: 'CLI reference for grafana-alertcheck.' diff --git a/grafana-alertcheck/docs/reference/cli.md b/grafana-alertcheck/docs/reference/cli.md new file mode 100644 index 000000000..1530c508d --- /dev/null +++ b/grafana-alertcheck/docs/reference/cli.md @@ -0,0 +1,91 @@ +--- +id: grafana-alertcheck-cli +title: CLI reference +sidebar_label: CLI reference +sidebar_position: 0 +description: Full reference for the grafana-alertcheck CLI: watch, check, list, environment, naming, and output. +--- + +# CLI reference + +``` +grafana-alertcheck +``` + +Connection details are always from the environment: `GRAFANA_URL` and `GRAFANA_TOKEN`. The token is never a flag and never logged. + +## `list` + +Lists every rule from the ruler endpoint — kind, folder, group, title, uid. Useful to check auth and to find `uid:` names. + +```bash +grafana-alertcheck list +``` + +## `watch` — record + +```bash +grafana-alertcheck watch --out [--pidfile F] [--daemon-log F] \ + --alerts [--folder F] [--poll-interval D] [--concurrency N] [--until RFC3339] +``` + +| Flag | Default | Meaning | +| ---- | ------- | ------- | +| `--out` | — | JSONL log path (required) | +| `--pidfile` | `.pid` | Where the recorder's pid is written | +| `--daemon-log` | `.daemon.log` | stdout/stderr sink for the detached recorder | +| `--alerts` | — | File of alert names, one per line, or `-` for stdin (required) | +| `--folder` | — | Default folder to scope unqualified names | +| `--poll-interval` | half the rule's interval | Override every rule's cadence (never clamped) | +| `--concurrency` | `1` | Max concurrent requests to Grafana | +| `--until` | run until signalled | Optional hard stop | + +`watch` writes the header, observes every non-paused rule once, checks the budget, then detaches a background recorder and returns. Recording is **unfiltered** — there is no `--states` here, so the same log can be re-classified later under different `--states` without re-recording. + +## `check` — classify + +```bash +grafana-alertcheck check [--in ] [--pidfile F] --from RFC3339 --to RFC3339 \ + [--alerts ...] [--folder F] [--states ...] [--preexisting ...] [--min-observed N] \ + [--allow-paused] [--nodata-is-unobservable] [--concurrency N] [--output json] +``` + +| Flag | Default | Meaning | +| ---- | ------- | ------- | +| `--in` | — | Log recorded by `watch`; empty selects single-step mode | +| `--pidfile` | `.pid` | Recorder to stop before reading `--in` | +| `--from` | see below | Moment the deploy finished | +| `--to` | — | End of the window (required) | +| `--alerts` | — | Required **without** `--in`; refused **with** `--in` | +| `--states` | `firing` | Comma-separated bad states: `firing,pending,nodata,error` | +| `--preexisting` | `fail-unless-recovered` | `fail-unless-recovered` \| `fail` \| `ignore` | +| `--min-observed` | every resolved rule | Minimum rules that must be observed | +| `--allow-paused` | `false` | Don't count pre-window-paused rules against `--min-observed` | +| `--nodata-is-unobservable` | `false` | Treat sustained `health=nodata` as unobservable | +| `--concurrency` | `1` | Max concurrent requests | +| `--output` | `table` | `json` also writes the machine-readable result to stdout | + +`--from` and `--to` are RFC3339 with an explicit offset and must come from your work — `from` from the deploy step, `to` from the step that finishes. In recorder mode an absent `--from` is a hard error; in single-step mode it falls back (with a warning) to the start of the step. + +## Naming alerts + +Alert names take one of four forms: + +| Form | Meaning | +| ---- | ------- | +| `HighErrorRate` | Title only, scoped by `--folder` | +| `Platform/HighErrorRate` | Folder + title | +| `Platform/api/HighErrorRate` | Folder + group + title (always unique) | +| `uid:abc123` | Exact uid (present on both endpoints) | + +Datasource-managed and recording rules are refused with a specific error. A name matching multiple rules errors listing every candidate with the copyable `Folder/Group/Title` and its `uid:` form. A no-match errors with case-insensitive substring suggestions and points at `list`. Duplicate names that resolve to the same uid collapse to one (a note, not an error). + +## Output and exit codes + +The human table goes to **stderr**: `RESULTS` (one row per rule), `VIOLATIONS` (one per violation), and `THRESHOLDS` (each rule's `maxGap`/`healthGrace`/`evalStaleAfter` plus global `transitionGrace`/`drainTimeout` and the largest measured clock skew). `--output json` writes the result to stdout. + +| Code | Meaning | +| ---- | ------- | +| `0` | Pass | +| `1` | Violations | +| `2` | Could not check — every library error, never a pass | From 6e9d55f6568950fa0e6b491e0fdf68f1f80fdb22 Mon Sep 17 00:00:00 2001 From: Bartek Tofel Date: Mon, 7 Sep 2026 11:33:48 +0200 Subject: [PATCH 41/43] chore: further update docs --- grafana-alertcheck/README.md | 46 ++++++++- grafana-alertcheck/docs/architecture.md | 68 +++++++++++++ grafana-alertcheck/docs/index.md | 4 + .../docs/reference/log-format.md | 98 +++++++++++++++++++ 4 files changed, 213 insertions(+), 3 deletions(-) create mode 100644 grafana-alertcheck/docs/architecture.md create mode 100644 grafana-alertcheck/docs/reference/log-format.md diff --git a/grafana-alertcheck/README.md b/grafana-alertcheck/README.md index 43cc03fc4..96aa0e999 100644 --- a/grafana-alertcheck/README.md +++ b/grafana-alertcheck/README.md @@ -1,6 +1,46 @@ # grafana-alertcheck -A CD quality gate for Grafana alerts: bookend a release with `watch` (record) and `check` (classify) to -answer whether any watched alert was in a bad state during the release window. +A CD quality gate for Grafana alerts. It bookends a release with two commands — `watch` (record) and +`check` (classify) — and answers whether any watched alert was in a bad state during the release window. -Under construction. +``` +watch → your work → check +``` + +`watch` starts a background recorder that polls each named alert into a JSONL log. After the work emits a +`from`/`to` pair, `check` proves continuous coverage of that window, classifies each alert's state +timeline, and exits `0`, `1`, or `2`. + +It **fails closed**: if it cannot get an answer, it stops the release — never a pass on an unproven window. + +## Quickstart + +```bash +export GRAFANA_URL=https://grafana.example.com +export GRAFANA_TOKEN=… + +grafana-alertcheck watch --out /tmp/run.jsonl --alerts alerts.txt +./deploy.sh # emits deployed_at= +./verify.sh # emits finished_at= +grafana-alertcheck check --in /tmp/run.jsonl --from "$deployed_at" --to "$finished_at" +``` + +Requires Grafana >= 13.0.0 and < 14.0.0. Connection details come from the environment only — the token is +never a flag. + +## Documentation + +| Doc | Covers | +| --- | ------ | +| [`docs/index.md`](./docs/index.md) | Overview, quickstarts, exit codes, common surprises | +| [`docs/how-alerts-are-evaluated.md`](./docs/how-alerts-are-evaluated.md) | Verdict model, coverage proof, health/liveness | +| [`docs/advanced.md`](./docs/advanced.md) | Check budget, scheduling, why history isn't queried | +| [`docs/architecture.md`](./docs/architecture.md) | Design invariants, the pure-function seam, recorder lifecycle | +| [`docs/reference/cli.md`](./docs/reference/cli.md) | Full CLI reference — subcommands, flags, naming | +| [`docs/reference/log-format.md`](./docs/reference/log-format.md) | The JSONL log schema, for debugging artifacts | + +## Build + +```bash +go build ./... && go test ./... +``` diff --git a/grafana-alertcheck/docs/architecture.md b/grafana-alertcheck/docs/architecture.md new file mode 100644 index 000000000..ca493220e --- /dev/null +++ b/grafana-alertcheck/docs/architecture.md @@ -0,0 +1,68 @@ +--- +id: grafana-alertcheck-architecture +title: Architecture +sidebar_label: Architecture +sidebar_position: 3 +description: The design invariants, pure-function seam, and recorder lifecycle of grafana-alertcheck, for maintainers. +--- + +# Architecture + +This page documents the invariants and seams a maintainer must not break. It exists because most of them are the difference between a gate that fails closed and one that silently passes broken windows. + +## Fail-closed invariants + +The gate must stop the release if it cannot get an answer. Every rule below is a specific instance of that: + +- **An error is never a pass.** A pass is exactly `len(Violations) == 0 && err == nil`. Every error path leaves `err` non-nil, and the CLI maps that to exit `2` unconditionally. +- **Inability beats violation.** Any `unobservable` rule is exit `2`, even alongside a real violation found first. +- **Absent never means normal.** An instance that leaves the bad set is looked up in the *same* response: present as `normal` → cleared; absent (or `MissingSeries`) → vanished (a discontinuity, not a recovery). +- **Staleness is absolute.** `grafana_now − lastEvaluation` is compared against a threshold, never "did it increase since the last poll" — a delta check reports stale on ~half the polls of a healthy rule. +- **`grafana_now` is the response `Date` header.** Never the runner clock, in any comparison against a Grafana timestamp. +- **No early exit.** `check` collects to `to + transitionGrace` before classifying once. +- **No replay.** No run-id key, no artifact download, no state between attempts. A retry is a new deploy. + +## The pure-function seam + +All correctness lives in two phases written as **pure functions** over a flat list of polls — no HTTP, no files, no clock, no goroutines: + +``` +HTTP ──> Source ──> []StateRule ──> reduce ──> []Poll ──> proveCoverage ──> decide ──> Result + │ + JSONL log ──> ReadLog ──┘ +``` + +- `proveCoverage` (the nine coverage checks) and `decide` (the instance timelines and outcomes) are pure; tests drive them with `[]Poll` literals and a fake `Clock`, with no sleeping or fixture server. +- `Check`/`Watch` are I/O shells: HTTP, signals, the pidfile, file reads, the countdown print. The only test doubles needed are the `Source` and `Clock` interfaces. +- `Policy` is the narrowed view of `Config` that reaches the pure layer — classification knobs and the window, no URL and no token. The token must never cross that line, which is the cheapest guarantee it never lands in an error string or a result. + +## Strict parsing as the version guard + +Both API responses are parsed strictly: a missing or unparseable **required** field (`health`, `state`, `lastEvaluation`, `interval`) is an error, never a zero value. Optional keys (`alerts`, `totals`, `labels`, `keepFiringFor`) are absent-tolerant, and unknown keys are ignored — so Grafana can add fields without breaking the parser, but removing one fails loudly. + +This, plus the declared supported range (Grafana >= 13.0.0, < 14.0.0), is how a deprecation or schema change is caught instead of silently misread. + +## The recorder lifecycle + +`watch` detaches a background recorder so observation survives the step boundary: + +1. Parent resolves names, writes the header, observes every non-paused rule once, checks the budget. +2. Parent re-execs itself as the child (`--daemon-child`) under a new session/process group, stdout/stderr to the daemon log. +3. Child re-reads the header, reopens the log `O_APPEND`, takes the exclusive `flock`, and writes one readiness byte on `--ready-fd`. +4. Parent writes the pidfile **after** the readiness report, then returns. + +Two authorities, only one of which is evidence: + +- The **pidfile** says a recording ever started (written only after ready, removed on failure). It can go stale — a pid gets reused. +- The **flock** says a writer exists *now*. The kernel drops it on exit, so the lock is always authoritative. + +On a clean stop (SIGTERM/SIGINT/`--until`) the child finishes the in-flight write, appends the `stopped` sentinel, fsyncs, and exits. A hard error writes no sentinel — so a recorder that died reads exactly like a coverage gap, because it is one. + +`check` signals via the pidfile, waits for the **lock** to release (never the pid), and only then reads the log once. Reading while a writer can still append can only produce a shorter window than was recorded. + +## The log is the source of truth + +`watch` records raw evidence, so nothing trusts a state that could become unreachable. Two consequences a maintainer must preserve: + +- The **header is authoritative for recording facts** (the cadence actually used, the URL, the alert set); the ruler API is authoritative for **rule facts** (`for`, `intervalSeconds`, kind). `check` always re-resolves definitions fresh and never reconstructs them from the header — the header duplicates `for`/`interval` only so the uploaded artifact is self-describing. +- The **cadence authority** is the header's `poll_every_seconds`, not the definitions. Re-deriving it would compare gaps recorded at an override cadence against default-cadence thresholds — fail-open in the faster-override direction. diff --git a/grafana-alertcheck/docs/index.md b/grafana-alertcheck/docs/index.md index 3e5bee11f..e8289a664 100644 --- a/grafana-alertcheck/docs/index.md +++ b/grafana-alertcheck/docs/index.md @@ -75,9 +75,13 @@ An error is never a pass: `2` wins over any violation found alongside it. - The gate checks alert **state and health**, not notification delivery — a silenced alert that still fires fails. - `recovered` has **no deadline** — a bad-at-`from` alert that clears by `to` passes; set `--preexisting fail` to forbid it. - A **retry is a new deploy**, not a replay — re-running the job re-records against a new `from`. +- `watch` and `check` must run in **one job, one runner, one filesystem** — nothing persists across jobs or attempts. +- The gate **never exits early** — a violation at minute 2 still holds the runner to `to + transitionGrace + drainTimeout`; size the job timeout to the planned run time the gate prints at start. ## More - [How alerts are evaluated](./how-alerts-are-evaluated) — the verdict model and coverage proof - [Check budget and scheduling](./advanced) — why the schedule and budget look the way they do, and why history isn't queried +- [Architecture](./architecture) — design invariants and the recorder lifecycle, for maintainers - [CLI reference](./reference/cli) — every subcommand and flag +- [Log format](./reference/log-format) — the JSONL log schema, for debugging artifacts diff --git a/grafana-alertcheck/docs/reference/log-format.md b/grafana-alertcheck/docs/reference/log-format.md new file mode 100644 index 000000000..66e071770 --- /dev/null +++ b/grafana-alertcheck/docs/reference/log-format.md @@ -0,0 +1,98 @@ +--- +id: grafana-alertcheck-log-format +title: Log format +sidebar_label: Log format +sidebar_position: 1 +description: The JSONL log schema written by watch and read by check, for debugging the forensic artifact. +--- + +# Log format + +`watch` records evidence to a JSONL log — one JSON object per line. A poll record *is* the heartbeat; there is no separate heartbeat type. + +## Record types + +Exactly three: + +| `type` | Meaning | +| ------ | ------- | +| `header` | Line 1 — identity and the alert set | +| `poll` | One reduced observation of one rule | +| `stopped` | The sentinel, written on a clean stop only | + +The header must be line 1, appear once, and carry `schema_version` `1` (any other value is a read error). Any unparseable line — including the last, or one after the sentinel — makes the log unreadable: a truncated log is evidence the recorder was killed, and must not pass. + +## Header + +```json +{ + "type": "header", + "schema_version": 1, + "url": "https://grafana.example.com", + "grafana_version": "13.1.0", + "started_at": "2026-09-07T10:00:00Z", + "rules": [ + { + "uid": "rule0000001", + "title": "HighErrorRate", + "folder": "Platform", + "group": "api", + "for_seconds": 300, + "interval_seconds": 60, + "is_paused": false, + "no_data_state": "OK", + "exec_err_state": "OK", + "poll_every_seconds": 30 + } + ] +} +``` + +- `url` and `rules` are the log's identity — `check` validates them against the current environment and a fresh ruler read. +- `is_paused` records the pause state at record start (the moment `skipped` means). +- `poll_every_seconds` is the cadence the recording **actually used** (after any `--poll-interval` override). `check` derives `maxGap` from it, never from `interval_seconds`. +- `for_seconds`, `interval_seconds`, `no_data_state`, `exec_err_state` are forensic only — `check` re-resolves definitions and never reads them back. + +## Poll + +```json +{ + "type": "poll", + "rule_uid": "rule0000001", + "grafana_now": "2026-09-07T10:00:30Z", + "skew_ms": 20, + "skew_bound_ms": 40, + "latency_ms": 123, + "found": true, + "state": "inactive", + "health": "ok", + "last_evaluation": "2026-09-07T10:00:28Z", + "is_paused": false, + "histogram": { "alerting": 0, "normal": 2004 }, + "reasons": { "NoData": 1091 }, + "abnormal": [ { "labels": { "env": "prod" }, "state": "firing", "active_at": "2026-09-07T09:50:00Z", "value": "1.5" } ], + "cleared": [ "env=prod\u0001..." ], + "vanished": [] +} +``` + +Field notes: + +- `grafana_now` is the response's `Date` header — never the runner clock. +- `skew_ms`/`skew_bound_ms` are the per-poll clock-skew estimate and its uncertainty (RTT/2), in milliseconds for compactness only. +- `found: false` is an authoritative `2xx` in which this rule was absent — a transport failure is retried and never becomes a poll. +- `state`, `health`, `last_error` are raw rule-level strings, reporting-only. +- `histogram` is a verbatim copy of the response `totals`; written, never analysed. +- `reasons` counts non-empty instance reasons (`NoData`, `Error`, `KeepLast`, …); composite states stay visible only here. +- `abnormal` holds only instances whose **canonical** state is not `normal`. +- `cleared`/`vanished` are instance keys that left the bad set, resolved against the same response: `cleared` = a real recovery; `vanished` = a discontinuity, never a recovery. + +Instance keys are a sorted `k=v\n` join of labels, so they correlate across polls without hashing. + +## Stopped + +```json +{ "type": "stopped", "at": "2026-09-07T10:10:30Z" } +``` + +`at` is the recorder's own stop time. `check` compares it against `to + transitionGrace`; absent or earlier is `unobservable` — never a pass. From 95b9b272c3a5a7a26284a4c2e1cd1698dfb2489f Mon Sep 17 00:00:00 2001 From: Bartek Tofel Date: Wed, 9 Sep 2026 10:56:10 +0200 Subject: [PATCH 42/43] chore: address code review comments --- .../cmd/grafana-alertcheck/check.go | 13 +++++-- .../cmd/grafana-alertcheck/watch.go | 8 +++++ grafana-alertcheck/internal/gate/classify.go | 2 +- grafana-alertcheck/internal/gate/coverage.go | 2 +- .../internal/gate/parse_ruler.go | 2 +- .../internal/gate/parse_state_test.go | 4 --- grafana-alertcheck/internal/gate/schedule.go | 15 +++++--- grafana-alertcheck/internal/gate/source.go | 34 ++++++++----------- 8 files changed, 47 insertions(+), 33 deletions(-) diff --git a/grafana-alertcheck/cmd/grafana-alertcheck/check.go b/grafana-alertcheck/cmd/grafana-alertcheck/check.go index 5493ed747..e95fce376 100644 --- a/grafana-alertcheck/cmd/grafana-alertcheck/check.go +++ b/grafana-alertcheck/cmd/grafana-alertcheck/check.go @@ -3,6 +3,7 @@ package main import ( "context" "encoding/json" + "errors" "flag" "fmt" "io" @@ -40,6 +41,13 @@ func runCheck(args []string, stdin io.Reader, stdout, stderr io.Writer) int { output := fs.String("output", "", `"json" writes the machine-readable Result to stdout in addition to the table; default is the table alone`) if err := fs.Parse(args); err != nil { + if errors.Is(err, flag.ErrHelp) { + return 0 + } + return 2 + } + if fs.NArg() != 0 { + fmt.Fprintf(stderr, "check: unexpected arguments %v\n", fs.Args()) return 2 } if *output != "" && *output != "json" { @@ -113,11 +121,10 @@ func runCheck(args []string, stdin io.Reader, stdout, stderr io.Writer) int { result, checkErr := gate.Check(ctx, cfg) - if err := renderTable(stderr, result); err != nil { - fmt.Fprintln(stderr, err) - } if checkErr != nil { fmt.Fprintln(stderr, checkErr) + } else if err := renderTable(stderr, result); err != nil { + fmt.Fprintln(stderr, err) } if *output == "json" { enc := json.NewEncoder(stdout) diff --git a/grafana-alertcheck/cmd/grafana-alertcheck/watch.go b/grafana-alertcheck/cmd/grafana-alertcheck/watch.go index d6db86672..69b585298 100644 --- a/grafana-alertcheck/cmd/grafana-alertcheck/watch.go +++ b/grafana-alertcheck/cmd/grafana-alertcheck/watch.go @@ -2,6 +2,7 @@ package main import ( "context" + "errors" "flag" "fmt" "io" @@ -49,6 +50,13 @@ func runWatch(args []string, stdin io.Reader, stdout, stderr io.Writer) int { readyFD := fs.Int(gate.ReadyFDFlag[2:], 0, "") if err := fs.Parse(args); err != nil { + if errors.Is(err, flag.ErrHelp) { + return 0 + } + return 2 + } + if fs.NArg() != 0 { + fmt.Fprintf(stderr, "watch: unexpected arguments %v\n", fs.Args()) return 2 } diff --git a/grafana-alertcheck/internal/gate/classify.go b/grafana-alertcheck/internal/gate/classify.go index d5d9c8e12..d8d058a34 100644 --- a/grafana-alertcheck/internal/gate/classify.go +++ b/grafana-alertcheck/internal/gate/classify.go @@ -476,7 +476,7 @@ func decide(h Header, polls []Poll, sentinel *time.Time, defs []Definition, Thresholds: make(map[string]RuleThresholds), Global: GlobalThresholds{ TransitionGrace: gt.transitionGrace, - GraceSource: gt.graceSource, + GraceSource: graceSourceOrNone(gt.graceSource), DrainTimeout: gt.drainTimeout, }, } diff --git a/grafana-alertcheck/internal/gate/coverage.go b/grafana-alertcheck/internal/gate/coverage.go index 535af07d2..603a7f7d8 100644 --- a/grafana-alertcheck/internal/gate/coverage.go +++ b/grafana-alertcheck/internal/gate/coverage.go @@ -146,7 +146,7 @@ func proveCoverage(h Header, polls []Poll, sentinel *time.Time, t ruleTimings, d // corrupted or hand-edited data (ReadLog does no field validation); // GrafanaNow-LastEvaluation would go negative and silently read as // fresh — fail-open. Treat it as unobservable instead. - if p.LastEvaluation.After(p.GrafanaNow) { + if p.LastEvaluation.Truncate(time.Second).After(p.GrafanaNow) { fail(ReasonFutureEvaluation, fmt.Sprintf( "lastEvaluation %s is after grafana_now %s (corrupted poll)", p.LastEvaluation.Format(time.RFC3339), p.GrafanaNow.Format(time.RFC3339))) diff --git a/grafana-alertcheck/internal/gate/parse_ruler.go b/grafana-alertcheck/internal/gate/parse_ruler.go index ae878d4c9..5517fcce2 100644 --- a/grafana-alertcheck/internal/gate/parse_ruler.go +++ b/grafana-alertcheck/internal/gate/parse_ruler.go @@ -84,7 +84,7 @@ func ParseDefinitions(body []byte) ([]Definition, error) { func parseDefinition(raw json.RawMessage, folder, group string) (Definition, error) { var m map[string]json.RawMessage if err := json.Unmarshal(raw, &m); err != nil { - return Definition{}, fmt.Errorf("%w", err) + return Definition{}, err } var forStr string diff --git a/grafana-alertcheck/internal/gate/parse_state_test.go b/grafana-alertcheck/internal/gate/parse_state_test.go index 9e2ed899c..f55c10185 100644 --- a/grafana-alertcheck/internal/gate/parse_state_test.go +++ b/grafana-alertcheck/internal/gate/parse_state_test.go @@ -217,10 +217,6 @@ func TestInstanceKey(t *testing.T) { require.Equal(t, "null", instanceKey(nil), "instanceKey(nil) should be \"null\"") } -<<<<<<< HEAD -// Label values may contain "\n" or "="; the JSON encoding must keep them distinct. -======= ->>>>>>> c276546b (chore: use testify's require in tests) func TestInstanceKey_NoCollision(t *testing.T) { require.NotEqual(t, instanceKey(map[string]string{"a": "1\nb=2"}), instanceKey(map[string]string{"a": "1", "b": "2"}), "instanceKey should not collide for sets {a:1\\nb=2} and {a:1,b:2}") diff --git a/grafana-alertcheck/internal/gate/schedule.go b/grafana-alertcheck/internal/gate/schedule.go index ae366baf4..c2fe6e1c5 100644 --- a/grafana-alertcheck/internal/gate/schedule.go +++ b/grafana-alertcheck/internal/gate/schedule.go @@ -339,6 +339,16 @@ func CheckBudget(t map[string]ruleTimings, measured map[string]time.Duration, co return fmt.Errorf("%s", b.String()) } +// graceSourceOrNone is the single "none" default for the grace-source field: +// an empty source means no rule contributed a transitionGrace. Applied here so +// StartupSummary and the human table print the same thing. +func graceSourceOrNone(source string) string { + if source == "" { + return "none" + } + return source +} + // StartupSummary formats the pre-run print an operator sees before the wait: // the total planned run time and the rule (with its `for` value) that set // transitionGrace, plus a warning when the grace eats more than @@ -347,10 +357,7 @@ func CheckBudget(t map[string]ruleTimings, measured map[string]time.Duration, co func StartupSummary(from, to time.Time, global globalTimings) (summary, warning string) { window := to.Sub(from) total := window + global.transitionGrace + global.drainTimeout - source := global.graceSource - if source == "" { - source = "none" - } + source := graceSourceOrNone(global.graceSource) summary = fmt.Sprintf( "planned run time: %s\n window %s + transitionGrace %s + drainTimeout %s\n transitionGrace source: %s", total, window, global.transitionGrace, global.drainTimeout, source) diff --git a/grafana-alertcheck/internal/gate/source.go b/grafana-alertcheck/internal/gate/source.go index bebb1839e..721151e21 100644 --- a/grafana-alertcheck/internal/gate/source.go +++ b/grafana-alertcheck/internal/gate/source.go @@ -44,9 +44,10 @@ type Observation struct { Latency time.Duration // t_send through the full body read — see requestResult.Latency } -// TransportError marks a failure worth retrying: a non-2xx response, a network -// failure, or a body that failed to parse. Not a deleted rule (an authoritative -// 2xx) and not a clock problem (a hard error — see doRequest). +// TransportError marks a failure worth retrying: a 5xx/429 response, a network +// failure, or a body that failed to parse. Not a 4xx (wrong auth, missing +// resource), not a deleted rule (an authoritative 2xx) and not a clock problem +// (a hard error — see doRequest). type TransportError struct { Err error Status int // 0 when the failure never got a status (network/transport failure) @@ -111,17 +112,7 @@ func parseGrafanaVersion(s string) (grafanaVersion, error) { var v grafanaVersion fields := [3]*int{&v.major, &v.minor, &v.patch} for i, field := range fields { - // Trim any trailing non-digit suffix (prerelease/build metadata, e.g. - // "0+security") rather than requiring an exact numeric match. - digits := parts[i] - j := 0 - for j < len(digits) && digits[j] >= '0' && digits[j] <= '9' { - j++ - } - if j == 0 { - return grafanaVersion{}, fmt.Errorf("unparseable version %q", s) - } - n, err := strconv.Atoi(digits[:j]) + n, err := strconv.Atoi(parts[i]) if err != nil { return grafanaVersion{}, fmt.Errorf("unparseable version %q: %w", s, err) } @@ -259,10 +250,11 @@ type requestResult struct { Latency time.Duration } -// doRequest performs one HTTP GET and classifies the outcome: network failure, -// non-2xx, or body-read failure is retryable (*TransportError); a missing or -// unparseable Date header or a skew beyond SkewHardLimit is a hard error — -// retrying can never fix either, so neither enters the backoff loop. +// doRequest performs one HTTP GET and classifies the outcome: a 5xx/429, a +// network failure, or a body-read failure is retryable (*TransportError); a +// 4xx (wrong auth, missing resource — retrying cannot fix it), a missing or +// unparseable Date header, or a skew beyond SkewHardLimit is a hard error, so +// none of those enters the backoff loop. // // The Date/skew check runs on every endpoint (even /api/health): a skew only // noticed once RuleState starts polling has already masked earlier reads, so it @@ -298,7 +290,11 @@ func (s *httpSource) doRequest(ctx context.Context, path string) (requestResult, latency := tBodyRead.Sub(tSend) if resp.StatusCode < 200 || resp.StatusCode >= 300 { - return requestResult{}, &TransportError{Err: fmt.Errorf("unexpected status %d", resp.StatusCode), Status: resp.StatusCode} + err := fmt.Errorf("unexpected status %d", resp.StatusCode) + if resp.StatusCode >= 400 && resp.StatusCode < 500 && resp.StatusCode != http.StatusTooManyRequests { + return requestResult{}, err + } + return requestResult{}, &TransportError{Err: err, Status: resp.StatusCode} } dateHeader := resp.Header.Get("Date") From c8d0cc5dc8c1efdf0dad4d476e49b1af3a22a439 Mon Sep 17 00:00:00 2001 From: Bartek Tofel Date: Wed, 9 Sep 2026 11:09:59 +0200 Subject: [PATCH 43/43] chore: get rid of goreleaser --- .../workflows/grafana-alertcheck-release.yml | 34 ------------------- grafana-alertcheck/.goreleaser.yaml | 33 ------------------ .../cmd/{grafana-alertcheck => }/check.go | 0 .../{grafana-alertcheck => }/check_test.go | 0 .../cmd/{grafana-alertcheck => }/common.go | 0 .../cmd/{grafana-alertcheck => }/env.go | 0 .../cmd/grafana-alertcheck/version.go | 29 ---------------- .../cmd/grafana-alertcheck/version_test.go | 25 -------------- .../cmd/{grafana-alertcheck => }/list.go | 0 .../cmd/{grafana-alertcheck => }/list_test.go | 0 .../cmd/{grafana-alertcheck => }/main.go | 4 +-- .../cmd/{grafana-alertcheck => }/main_test.go | 0 .../cmd/{grafana-alertcheck => }/style.go | 0 .../cmd/{grafana-alertcheck => }/table.go | 0 .../{grafana-alertcheck => }/table_test.go | 0 .../cmd/{grafana-alertcheck => }/watch.go | 0 .../{grafana-alertcheck => }/watch_test.go | 0 grafana-alertcheck/docs/index.md | 2 +- 18 files changed, 2 insertions(+), 125 deletions(-) delete mode 100644 .github/workflows/grafana-alertcheck-release.yml delete mode 100644 grafana-alertcheck/.goreleaser.yaml rename grafana-alertcheck/cmd/{grafana-alertcheck => }/check.go (100%) rename grafana-alertcheck/cmd/{grafana-alertcheck => }/check_test.go (100%) rename grafana-alertcheck/cmd/{grafana-alertcheck => }/common.go (100%) rename grafana-alertcheck/cmd/{grafana-alertcheck => }/env.go (100%) delete mode 100644 grafana-alertcheck/cmd/grafana-alertcheck/version.go delete mode 100644 grafana-alertcheck/cmd/grafana-alertcheck/version_test.go rename grafana-alertcheck/cmd/{grafana-alertcheck => }/list.go (100%) rename grafana-alertcheck/cmd/{grafana-alertcheck => }/list_test.go (100%) rename grafana-alertcheck/cmd/{grafana-alertcheck => }/main.go (91%) rename grafana-alertcheck/cmd/{grafana-alertcheck => }/main_test.go (100%) rename grafana-alertcheck/cmd/{grafana-alertcheck => }/style.go (100%) rename grafana-alertcheck/cmd/{grafana-alertcheck => }/table.go (100%) rename grafana-alertcheck/cmd/{grafana-alertcheck => }/table_test.go (100%) rename grafana-alertcheck/cmd/{grafana-alertcheck => }/watch.go (100%) rename grafana-alertcheck/cmd/{grafana-alertcheck => }/watch_test.go (100%) diff --git a/.github/workflows/grafana-alertcheck-release.yml b/.github/workflows/grafana-alertcheck-release.yml deleted file mode 100644 index 4f946b18f..000000000 --- a/.github/workflows/grafana-alertcheck-release.yml +++ /dev/null @@ -1,34 +0,0 @@ -name: Grafana Alertcheck Release - -on: - push: - tags: - - grafana-alertcheck/v* - -jobs: - release: - name: Build and Release - runs-on: ubuntu-latest - environment: integration - permissions: - id-token: write - contents: write - steps: - - name: Checkout repo - uses: actions/checkout@v7 - with: - fetch-depth: 0 - - name: Set up Go - uses: actions/setup-go@v7 - with: - go-version-file: ./grafana-alertcheck/go.mod - cache-dependency-path: ./grafana-alertcheck/go.mod - - name: Goreleaser Release - uses: goreleaser/goreleaser-action@f06c13b6b1a9625abc9e6e439d9c05a8f2190e94 # v7.2.3 - with: - distribution: goreleaser-pro - version: "~> v2" - args: release --clean -f ./grafana-alertcheck/.goreleaser.yaml - env: - GITHUB_TOKEN: ${{ secrets.GITHUB_TOKEN }} - GORELEASER_KEY: ${{ secrets.GORELEASER_KEY }} diff --git a/grafana-alertcheck/.goreleaser.yaml b/grafana-alertcheck/.goreleaser.yaml deleted file mode 100644 index 0890d9f82..000000000 --- a/grafana-alertcheck/.goreleaser.yaml +++ /dev/null @@ -1,33 +0,0 @@ -# yaml-language-server: $schema=https://goreleaser.com/static/schema-pro.json -version: 2 -project_name: grafana-alertcheck - -dist: grafana-alertcheck/dist - -monorepo: - tag_prefix: grafana-alertcheck/ - dir: grafana-alertcheck - -builds: - - id: grafana-alertcheck - main: ./cmd/grafana-alertcheck/main.go - ldflags: - - -s - - -w - - -X github.com/smartcontractkit/chainlink-testing-framework/grafana-alertcheck/cmd/grafana-alertcheck.version={{.Version}} - - -X github.com/smartcontractkit/chainlink-testing-framework/grafana-alertcheck/cmd/grafana-alertcheck.commit={{.ShortCommit}} - - -X github.com/smartcontractkit/chainlink-testing-framework/grafana-alertcheck/cmd/grafana-alertcheck.date={{.CommitDate}} - - -X github.com/smartcontractkit/chainlink-testing-framework/grafana-alertcheck/cmd/grafana-alertcheck.builtBy=goreleaser - goos: - - linux - - darwin - goarch: - - amd64 - - arm64 - binary: grafana-alertcheck - env: - - CGO_ENABLED=0 - -before: - hooks: - - sh -c "cd grafana-alertcheck && go mod tidy" diff --git a/grafana-alertcheck/cmd/grafana-alertcheck/check.go b/grafana-alertcheck/cmd/check.go similarity index 100% rename from grafana-alertcheck/cmd/grafana-alertcheck/check.go rename to grafana-alertcheck/cmd/check.go diff --git a/grafana-alertcheck/cmd/grafana-alertcheck/check_test.go b/grafana-alertcheck/cmd/check_test.go similarity index 100% rename from grafana-alertcheck/cmd/grafana-alertcheck/check_test.go rename to grafana-alertcheck/cmd/check_test.go diff --git a/grafana-alertcheck/cmd/grafana-alertcheck/common.go b/grafana-alertcheck/cmd/common.go similarity index 100% rename from grafana-alertcheck/cmd/grafana-alertcheck/common.go rename to grafana-alertcheck/cmd/common.go diff --git a/grafana-alertcheck/cmd/grafana-alertcheck/env.go b/grafana-alertcheck/cmd/env.go similarity index 100% rename from grafana-alertcheck/cmd/grafana-alertcheck/env.go rename to grafana-alertcheck/cmd/env.go diff --git a/grafana-alertcheck/cmd/grafana-alertcheck/version.go b/grafana-alertcheck/cmd/grafana-alertcheck/version.go deleted file mode 100644 index 5cf68b6af..000000000 --- a/grafana-alertcheck/cmd/grafana-alertcheck/version.go +++ /dev/null @@ -1,29 +0,0 @@ -package main - -import ( - "fmt" - "io" -) - -// Build metadata. These are package-level variables so goreleaser's ldflags -// (-X) can stamp them at build time; left unstamped they fall back to the -// "dev" defaults below, which is what a plain `go build` produces. -var ( - version = "dev" - commit = "unknown" - date = "unknown" - builtBy = "unknown" -) - -// runVersion prints the build metadata to stdout. Unlike list/watch/check it -// needs no Grafana connection, so it never touches the environment or the -// network; it exists purely so operators can answer "what am I running?" -// against a deployed binary. -func runVersion(args []string, stdout, stderr io.Writer) int { - if len(args) != 0 { - fmt.Fprintf(stderr, "version takes no arguments, got %v\n", args) - return 2 - } - fmt.Fprintf(stdout, "version: %s\ncommit: %s\ndate: %s\nbuiltBy: %s\n", version, commit, date, builtBy) - return 0 -} diff --git a/grafana-alertcheck/cmd/grafana-alertcheck/version_test.go b/grafana-alertcheck/cmd/grafana-alertcheck/version_test.go deleted file mode 100644 index 9463e990c..000000000 --- a/grafana-alertcheck/cmd/grafana-alertcheck/version_test.go +++ /dev/null @@ -1,25 +0,0 @@ -package main - -import ( - "bytes" - "testing" - - "github.com/stretchr/testify/require" -) - -func TestRunVersion(t *testing.T) { - var stdout, stderr bytes.Buffer - code := run([]string{"version"}, &stdout, &stderr) - require.Equal(t, 0, code) - out := stdout.String() - for _, want := range []string{"version:", "commit:", "date:", "builtBy:"} { - require.Contains(t, out, want) - } - require.Empty(t, stderr.String()) -} - -func TestRunVersion_RejectsArgs(t *testing.T) { - var stdout, stderr bytes.Buffer - code := run([]string{"version", "extra"}, &stdout, &stderr) - require.Equal(t, 2, code) -} diff --git a/grafana-alertcheck/cmd/grafana-alertcheck/list.go b/grafana-alertcheck/cmd/list.go similarity index 100% rename from grafana-alertcheck/cmd/grafana-alertcheck/list.go rename to grafana-alertcheck/cmd/list.go diff --git a/grafana-alertcheck/cmd/grafana-alertcheck/list_test.go b/grafana-alertcheck/cmd/list_test.go similarity index 100% rename from grafana-alertcheck/cmd/grafana-alertcheck/list_test.go rename to grafana-alertcheck/cmd/list_test.go diff --git a/grafana-alertcheck/cmd/grafana-alertcheck/main.go b/grafana-alertcheck/cmd/main.go similarity index 91% rename from grafana-alertcheck/cmd/grafana-alertcheck/main.go rename to grafana-alertcheck/cmd/main.go index 2672de1a3..7ab5d6397 100644 --- a/grafana-alertcheck/cmd/grafana-alertcheck/main.go +++ b/grafana-alertcheck/cmd/main.go @@ -12,7 +12,7 @@ func main() { os.Exit(run(os.Args[1:], os.Stdout, os.Stderr)) } -const usage = "usage: grafana-alertcheck " +const usage = "usage: grafana-alertcheck " // run is the whole of main's testable surface: parse the subcommand, dispatch, // return the process exit code. Exit codes below 2 (pass/violations) belong to @@ -37,8 +37,6 @@ func run(args []string, stdout, stderr io.Writer) int { return runWatch(args[1:], os.Stdin, stdout, stderr) case "check": return runCheck(args[1:], os.Stdin, stdout, stderr) - case "version": - return runVersion(args[1:], stdout, stderr) case "-h", "-help", "--help": fmt.Fprintln(stdout, usage) return 0 diff --git a/grafana-alertcheck/cmd/grafana-alertcheck/main_test.go b/grafana-alertcheck/cmd/main_test.go similarity index 100% rename from grafana-alertcheck/cmd/grafana-alertcheck/main_test.go rename to grafana-alertcheck/cmd/main_test.go diff --git a/grafana-alertcheck/cmd/grafana-alertcheck/style.go b/grafana-alertcheck/cmd/style.go similarity index 100% rename from grafana-alertcheck/cmd/grafana-alertcheck/style.go rename to grafana-alertcheck/cmd/style.go diff --git a/grafana-alertcheck/cmd/grafana-alertcheck/table.go b/grafana-alertcheck/cmd/table.go similarity index 100% rename from grafana-alertcheck/cmd/grafana-alertcheck/table.go rename to grafana-alertcheck/cmd/table.go diff --git a/grafana-alertcheck/cmd/grafana-alertcheck/table_test.go b/grafana-alertcheck/cmd/table_test.go similarity index 100% rename from grafana-alertcheck/cmd/grafana-alertcheck/table_test.go rename to grafana-alertcheck/cmd/table_test.go diff --git a/grafana-alertcheck/cmd/grafana-alertcheck/watch.go b/grafana-alertcheck/cmd/watch.go similarity index 100% rename from grafana-alertcheck/cmd/grafana-alertcheck/watch.go rename to grafana-alertcheck/cmd/watch.go diff --git a/grafana-alertcheck/cmd/grafana-alertcheck/watch_test.go b/grafana-alertcheck/cmd/watch_test.go similarity index 100% rename from grafana-alertcheck/cmd/grafana-alertcheck/watch_test.go rename to grafana-alertcheck/cmd/watch_test.go diff --git a/grafana-alertcheck/docs/index.md b/grafana-alertcheck/docs/index.md index e8289a664..95d413772 100644 --- a/grafana-alertcheck/docs/index.md +++ b/grafana-alertcheck/docs/index.md @@ -23,7 +23,7 @@ It **fails closed**: if it cannot get an answer, it stops the release. It never ## Install ```bash -go install github.com/smartcontractkit/chainlink-testing-framework/grafana-alertcheck/cmd/grafana-alertcheck@latest +go install github.com/smartcontractkit/chainlink-testing-framework/grafana-alertcheck/cmd@latest ``` Connection details come from the environment — the token is env-only, never a flag: