diff --git a/.github/workflows/validate.yml b/.github/workflows/validate.yml index 7bf5550..d333dc4 100644 --- a/.github/workflows/validate.yml +++ b/.github/workflows/validate.yml @@ -54,6 +54,21 @@ jobs: -sS -f -o /dev/null http://localhost:9222/healthz docker rm -f smoke-test + alerts: + runs-on: ubuntu-latest + steps: + - uses: actions/checkout@v7.0.1 + with: + persist-credentials: false + + # Syntax check plus unit tests: each rule must fire, and stay quiet, on synthetic series. + - name: promtool check and test + run: | + docker run --rm -v "$PWD/alerts:/alerts" -w /alerts --entrypoint promtool \ + prom/prometheus:v3.15.0 check rules github-actions-runner-exporter.rules.yml + docker run --rm -v "$PWD/alerts:/alerts" -w /alerts --entrypoint promtool \ + prom/prometheus:v3.15.0 test rules github-actions-runner-exporter.test.yml + zizmor: uses: drumandbytes/reusable-actions/.github/workflows/zizmor.yml@v1 @@ -61,7 +76,7 @@ jobs: required-checks-passed: name: Required checks passed runs-on: ubuntu-latest - needs: [lint, smoke-test, zizmor] + needs: [lint, smoke-test, alerts, zizmor] if: always() steps: - if: contains(needs.*.result, 'failure') || contains(needs.*.result, 'cancelled') diff --git a/README.md b/README.md index 1a1f772..c0d093d 100644 --- a/README.md +++ b/README.md @@ -95,6 +95,23 @@ images to GHCR with SLSA provenance attestation on push to `main`/tags. Import it in Grafana (Dashboards → New → Import) and pick your Prometheus data source. +## Alerting + +[`alerts/github-actions-runner-exporter.rules.yml`](alerts/github-actions-runner-exporter.rules.yml) has ready-made Prometheus alerting rules. Add it under `rule_files:`, or paste its group into a `PrometheusRule`'s `spec.groups` on kube-prometheus-stack. + +| Alert | Fires when | Severity | +| --- | --- | --- | +| `GithubRunnerOffline` | a registered runner is offline for 10m | warning | +| `GithubRunnersAllOffline` | every runner is offline for 5m | critical | +| `GithubRunnersSaturated` | every online runner has been busy for 30m (jobs are likely queueing) | warning | +| `GithubRunnerExporterStale` | the runner poll keeps failing (a stale cache is served) for 5m | warning | +| `GithubOrgStatsStale` | the org stats poll keeps failing for 30m | warning | +| `GithubApiRateLimitLow` | under 10% of the API rate limit is left for 10m | warning | +| `GithubWorkflowFailing` | a workflow's latest run failed or timed out, for 1h | info | +| `GithubDependabotCriticalAlerts` | a repo has open critical Dependabot alerts for 1h | warning | + +`GithubWorkflowFailing` looks at each workflow's latest completed run on any branch, so it's `info` rather than paging. The thresholds are starting points. Every rule has unit tests in [`alerts/github-actions-runner-exporter.test.yml`](alerts/github-actions-runner-exporter.test.yml), run in CI with `promtool test rules`. + ## How it was made Built with the help of an AI coding assistant (Claude). I review and test what gets published. diff --git a/alerts/github-actions-runner-exporter.rules.yml b/alerts/github-actions-runner-exporter.rules.yml new file mode 100644 index 0000000..0a84e42 --- /dev/null +++ b/alerts/github-actions-runner-exporter.rules.yml @@ -0,0 +1,79 @@ +# Prometheus alerting rules for github-actions-runner-exporter. +# Load with `rule_files:`, or paste the group into a PrometheusRule's spec.groups. +# Thresholds are starting points; tune `for:` and the ratios to your runner fleet. +groups: + - name: github-actions-runner-exporter + rules: + - alert: GithubRunnerOffline + expr: max by (runner, os) (github_runner_up) == 0 + for: 10m + labels: + severity: warning + annotations: + summary: "Self-hosted runner {{ $labels.runner }} is offline" + description: The runner is registered but not connected to GitHub. Jobs that need it will queue. + + - alert: GithubRunnersAllOffline + expr: count(github_runner_up) > 0 and sum(github_runner_up) == 0 + for: 5m + labels: + severity: critical + annotations: + summary: All self-hosted runners are offline + description: No self-hosted runner can pick up jobs; every workflow that targets them is stuck. + + - alert: GithubRunnersSaturated + # Every online runner busy for a long stretch: jobs are probably queueing. + expr: sum(github_runner_busy) >= sum(github_runner_up) and sum(github_runner_up) > 0 + for: 30m + labels: + severity: warning + annotations: + summary: All online self-hosted runners have been busy for 30 minutes + description: New jobs are likely waiting in the queue. Consider adding runners. + + - alert: GithubRunnerExporterStale + expr: max(github_runners_up) == 0 + for: 5m + labels: + severity: warning + annotations: + summary: Runner status is stale + description: The exporter's last runner poll failed and a cached result is being served. Check the token and GitHub API reachability. + + - alert: GithubOrgStatsStale + expr: max(github_org_up) == 0 + for: 30m + labels: + severity: warning + annotations: + summary: Org and repo stats are stale + description: The org stats poll keeps failing; CI, PR and Dependabot metrics are no longer updating. + + - alert: GithubApiRateLimitLow + expr: max(github_rate_limit_remaining) / max(github_rate_limit_limit) < 0.10 + for: 10m + labels: + severity: warning + annotations: + summary: "GitHub API rate limit at {{ $value | humanizePercentage }} remaining" + description: The exporter (or anything else sharing the token) is close to the limit, and polls will start failing. Raise the cache TTLs or use a separate token. + + - alert: GithubWorkflowFailing + # The latest completed run of the workflow, on whatever branch ran last. + expr: max by (repo, workflow, url, conclusion) (github_repo_ci_last_run_conclusion{conclusion=~"failure|timed_out"}) == 1 + for: 1h + labels: + severity: info + annotations: + summary: "{{ $labels.repo }}: {{ $labels.workflow }} last run {{ $labels.conclusion }}" + description: "Latest run: {{ $labels.url }}" + + - alert: GithubDependabotCriticalAlerts + expr: sum by (repo) (github_repo_dependabot_alerts_open{severity="critical"}) > 0 + for: 1h + labels: + severity: warning + annotations: + summary: "{{ $labels.repo }} has {{ $value }} open critical Dependabot alert(s)" + description: A dependency has a known critical vulnerability with an available advisory. diff --git a/alerts/github-actions-runner-exporter.test.yml b/alerts/github-actions-runner-exporter.test.yml new file mode 100644 index 0000000..46db996 --- /dev/null +++ b/alerts/github-actions-runner-exporter.test.yml @@ -0,0 +1,133 @@ +# promtool test rules alerts/github-actions-runner-exporter.test.yml +rule_files: + - github-actions-runner-exporter.rules.yml +evaluation_interval: 1m + +tests: + # one of two runners offline: that runner alerts, "all offline" doesn't + - interval: 1m + input_series: + - series: 'github_runner_up{runner="gha-x64-1",os="Linux"}' + values: 0x20 + - series: 'github_runner_up{runner="gha-x64-2",os="Linux"}' + values: 1x20 + alert_rule_test: + - eval_time: 11m + alertname: GithubRunnerOffline + exp_alerts: + - exp_labels: {severity: warning, runner: gha-x64-1, os: Linux} + exp_annotations: + summary: Self-hosted runner gha-x64-1 is offline + description: The runner is registered but not connected to GitHub. Jobs that need it will queue. + - eval_time: 11m + alertname: GithubRunnersAllOffline + exp_alerts: [] + + - interval: 1m + input_series: + - series: 'github_runner_up{runner="gha-x64-1",os="Linux"}' + values: 0x10 + - series: 'github_runner_up{runner="gha-x64-2",os="Linux"}' + values: 0x10 + alert_rule_test: + - eval_time: 6m + alertname: GithubRunnersAllOffline + exp_alerts: + - exp_labels: {severity: critical} + exp_annotations: + summary: All self-hosted runners are offline + description: No self-hosted runner can pick up jobs; every workflow that targets them is stuck. + + # saturated only when every online runner is busy + - interval: 1m + input_series: + - series: 'github_runner_up{runner="gha-x64-1",os="Linux"}' + values: 1x40 + - series: 'github_runner_up{runner="gha-x64-2",os="Linux"}' + values: 1x40 + - series: 'github_runner_busy{runner="gha-x64-1",os="Linux"}' + values: 1x40 + - series: 'github_runner_busy{runner="gha-x64-2",os="Linux"}' + values: 1x20 0x20 + alert_rule_test: + - eval_time: 31m + alertname: GithubRunnersSaturated + exp_alerts: [] + - eval_time: 19m + alertname: GithubRunnersSaturated + exp_alerts: [] + + - interval: 1m + input_series: + - series: 'github_runner_up{runner="gha-x64-1",os="Linux"}' + values: 1x40 + - series: 'github_runner_busy{runner="gha-x64-1",os="Linux"}' + values: 1x40 + alert_rule_test: + - eval_time: 31m + alertname: GithubRunnersSaturated + exp_alerts: + - exp_labels: {severity: warning} + exp_annotations: + summary: All online self-hosted runners have been busy for 30 minutes + description: New jobs are likely waiting in the queue. Consider adding runners. + + - interval: 1m + input_series: + - series: github_runners_up + values: 0x40 + - series: github_org_up + values: 0x40 + - series: github_rate_limit_remaining + values: 300x40 + - series: github_rate_limit_limit + values: 5000x40 + alert_rule_test: + - eval_time: 6m + alertname: GithubRunnerExporterStale + exp_alerts: + - exp_labels: {severity: warning} + exp_annotations: + summary: Runner status is stale + description: The exporter's last runner poll failed and a cached result is being served. Check the token and GitHub API reachability. + - eval_time: 31m + alertname: GithubOrgStatsStale + exp_alerts: + - exp_labels: {severity: warning} + exp_annotations: + summary: Org and repo stats are stale + description: The org stats poll keeps failing; CI, PR and Dependabot metrics are no longer updating. + - eval_time: 11m + alertname: GithubApiRateLimitLow + exp_alerts: + - exp_labels: {severity: warning} + exp_annotations: + summary: GitHub API rate limit at 6% remaining + description: The exporter (or anything else sharing the token) is close to the limit, and polls will start failing. Raise the cache TTLs or use a separate token. + + # only failed/timed-out runs and critical Dependabot alerts + - interval: 1m + input_series: + - series: 'github_repo_ci_last_run_conclusion{repo="app",workflow="Validate",url="https://github.com/org/app/actions/runs/1",conclusion="failure"}' + values: 1x70 + - series: 'github_repo_ci_last_run_conclusion{repo="app",workflow="Build",url="https://github.com/org/app/actions/runs/2",conclusion="success"}' + values: 1x70 + - series: 'github_repo_dependabot_alerts_open{repo="app",severity="critical"}' + values: 2x70 + - series: 'github_repo_dependabot_alerts_open{repo="web",severity="high"}' + values: 5x70 + alert_rule_test: + - eval_time: 61m + alertname: GithubWorkflowFailing + exp_alerts: + - exp_labels: {severity: info, repo: app, workflow: Validate, url: "https://github.com/org/app/actions/runs/1", conclusion: failure} + exp_annotations: + summary: "app: Validate last run failure" + description: "Latest run: https://github.com/org/app/actions/runs/1" + - eval_time: 61m + alertname: GithubDependabotCriticalAlerts + exp_alerts: + - exp_labels: {severity: warning, repo: app} + exp_annotations: + summary: app has 2 open critical Dependabot alert(s) + description: A dependency has a known critical vulnerability with an available advisory.