From e56845515c5fc27dc2cf57f85eeed3fd6f184884 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?M=C4=81ris=20Pop=C4=93ns?= Date: Sun, 4 Oct 2026 20:18:47 +0300 Subject: [PATCH] fix: report the real API rate-limit budget from response headers GET /rate_limit reported used=0 for our tokens, so github_rate_limit_remaining sat at 5000 and the low-budget alert could never fire. Read X-RateLimit-* from every API response instead, as GitHub documents. GitHub counts per region, so one token can be in several core windows at once: track each by reset time, report the scarcest as github_rate_limit_remaining, and expose every window as github_rate_limit_window_remaining{reset}. Drops one request per org refresh. --- CLAUDE.md | 10 ++-- README.md | 16 +++++- .../github-actions-runner-exporter.json | 14 +++-- internal/collector/collector.go | 33 +++++++++--- internal/github/client.go | 52 ++++++++++++++++--- internal/github/client_test.go | 47 +++++++++++++++++ internal/github/types.go | 12 ++--- internal/orgstats/feed_test.go | 5 +- internal/orgstats/orgstats.go | 9 +--- main.go | 2 +- 10 files changed, 157 insertions(+), 43 deletions(-) diff --git a/CLAUDE.md b/CLAUDE.md index 890c605..9773cce 100644 --- a/CLAUDE.md +++ b/CLAUDE.md @@ -19,7 +19,9 @@ depending on which TTL "wins". - `internal/github` — REST client for every GitHub endpoint this exporter calls (`orgs/{org}/actions/runners`, `orgs/{org}/repos`, `repos/{owner}/{repo}/actions/workflows`, `.../actions/workflows/{id}/runs`, - `.../pulls`, `.../dependabot/alerts`, `/rate_limit`). `OpenPRCount` reads + `.../pulls`, `.../dependabot/alerts`). The rate-limit budget comes from + the `X-RateLimit-*` headers of those responses, not `GET /rate_limit`, + which reports used=0 for our tokens. `OpenPRCount` reads the page count off the `Link` response header instead of paginating — deliberate: it's one request regardless of how many open PRs a repo has, and stays on the *core* rate limit rather than the separately-throttled @@ -34,8 +36,8 @@ depending on which TTL "wins". the client's raw responses into each domain's `Summary`. `orgstats.Build` swallows per-repo call failures deliberately (one repo the token can't see, or with Dependabot disabled, shouldn't blank out every other - repo's data) — but propagates a failure of `Repos`/`RateLimit` - themselves, since nothing else can proceed without those. + repo's data) — but propagates a failure of `Repos` itself, since nothing else can + proceed without it. - `internal/collector` — adapts both `Summary` types into Prometheus metrics for `/metrics`, on one shared `prometheus.Collector`. - `internal/config` — env var parsing (see README's Configuration table). @@ -52,7 +54,7 @@ The watermark is pinned by the oldest in-progress run (capped at 24h) — don't add `status=completed` to the runs query, or a slow run created before a faster one gets skipped forever. Dedupe is by (run ID, attempt) and job ID, kept 48h. A restart starts from "now"; nothing is replayed. -Measured on drumandbytes (26 repos, 156 active workflows): ~106 calls per +Measured on drumandbytes (26 repos, 156 active workflows): ~105 calls per refresh vs ~236 with the old per-workflow polling. ## Build / test / run diff --git a/README.md b/README.md index 044686a..b2eab5a 100644 --- a/README.md +++ b/README.md @@ -43,8 +43,9 @@ scrape. | `github_runner_busy` | `runner`, `os` | 1 if the runner is currently executing a job, 0 if idle | | `github_org_up` | | 1 if the last org/repo stats poll succeeded, 0 if a stale cache is being served or the first poll is still running (about a minute after startup at ~25 repos) | | `github_org_repos_total` | `visibility` | Number of non-archived repos, by `public`/`private` | -| `github_rate_limit_remaining` | | Remaining core API rate-limit budget | -| `github_rate_limit_limit` | | Total core API rate-limit budget | +| `github_rate_limit_remaining` | | Remaining core API budget in the scarcest active window: the one that runs out first | +| `github_rate_limit_limit` | | Total core API budget of that window | +| `github_rate_limit_window_remaining` | `reset` | Remaining budget per active core window, labelled with its reset time. GitHub counts per region, so one token can have several windows at once, and which one a call counts against depends on the endpoint. A window's series disappears once it resets | | `github_repo_open_prs` | `repo` | Number of open pull requests | | `github_repo_ci_last_run_conclusion` | `repo`, `workflow`, `url`, `conclusion` | Always 1 - an "info" metric. `conclusion` is GitHub's own string verbatim (`success`, `failure`, `cancelled`, `skipped`, `neutral`, `timed_out`, `action_required`, `stale`), not collapsed to pass/fail here - what counts as "actually broken" is a dashboard-level call. `url` links to the run on github.com. Absent if the workflow has never run | | `github_repo_ci_last_run_timestamp_seconds` | `repo`, `workflow` | Unix timestamp of that workflow's latest completed run | @@ -97,6 +98,17 @@ histogram_quantile(0.95, sum by (le, runner_label) (rate(github_job_queue_second sum by (conclusion) (increase(github_workflow_runs_total[1d])) ``` +## Rate limit + +The budget comes from the `X-RateLimit-*` headers of the exporter's own API +responses, which GitHub documents as authoritative. `GET /rate_limit` isn't +used: it can disagree with the headers, and for our tokens it reported +`used: 0` throughout. Because GitHub serves requests from several regions, a +token can be counted in two `core` windows at once, with different reset +times and counts. On drumandbytes, runners, workflows and Dependabot alerts +landed in one window and everything else in another. The exporter tracks every +window it sees and alerts on the lowest. + ## Configuration | Env var | Default | Required | diff --git a/dashboards/github-actions-runner-exporter.json b/dashboards/github-actions-runner-exporter.json index 294eaf9..234571e 100644 --- a/dashboards/github-actions-runner-exporter.json +++ b/dashboards/github-actions-runner-exporter.json @@ -8,7 +8,7 @@ "prometheus" ], "schemaVersion": 39, - "version": 2, + "version": 3, "editable": true, "time": { "from": "now-24h", @@ -952,12 +952,20 @@ "targets": [ { "expr": "max(github_rate_limit_remaining)", - "legendFormat": "remaining", + "legendFormat": "remaining (scarcest window)", "range": true, "instant": false, "refId": "A" + }, + { + "expr": "max by (reset) (github_rate_limit_window_remaining)", + "legendFormat": "window resetting {{reset}}", + "range": true, + "instant": false, + "refId": "B" } - ] + ], + "description": "remaining: the scarcest core window, which is what runs out first. GitHub counts per region, so a token can have several windows at once; each shows as its own line until it resets." }, { "id": 18, diff --git a/internal/collector/collector.go b/internal/collector/collector.go index 475ae52..ce7398d 100644 --- a/internal/collector/collector.go +++ b/internal/collector/collector.go @@ -8,6 +8,7 @@ import ( "github.com/prometheus/client_golang/prometheus" "github.com/drumandbytes/github-actions-runner-exporter/internal/fetch" + "github.com/drumandbytes/github-actions-runner-exporter/internal/github" "github.com/drumandbytes/github-actions-runner-exporter/internal/orgstats" "github.com/drumandbytes/github-actions-runner-exporter/internal/runners" ) @@ -17,6 +18,7 @@ const namespace = "github" type Collector struct { runnerFetcher *fetch.Fetcher[runners.Summary] orgFetcher *fetch.Fetcher[orgstats.Summary] + rateLimitFn func() map[time.Time]github.RateLimit runnerUp *prometheus.Desc runnerBusy *prometheus.Desc @@ -25,6 +27,7 @@ type Collector struct { reposTotal *prometheus.Desc rateLimit *prometheus.Desc rateLimitCap *prometheus.Desc + rateWindow *prometheus.Desc repoOpenPRs *prometheus.Desc repoCILastRunConclusion *prometheus.Desc @@ -34,7 +37,7 @@ type Collector struct { } // New builds the collector. The intervals only feed HELP text; polling is the Fetchers' job. -func New(runnerFetcher *fetch.Fetcher[runners.Summary], orgFetcher *fetch.Fetcher[orgstats.Summary], runnerCacheTTL, orgCacheTTL time.Duration) *Collector { +func New(runnerFetcher *fetch.Fetcher[runners.Summary], orgFetcher *fetch.Fetcher[orgstats.Summary], rateLimits func() map[time.Time]github.RateLimit, runnerCacheTTL, orgCacheTTL time.Duration) *Collector { runnerNote := fmt.Sprintf(" Polled every %s.", runnerCacheTTL) orgNote := fmt.Sprintf(" Polled every %s - not real-time by design, see internal/orgstats.", orgCacheTTL) desc := func(subsystem, name, help string, labels []string) *prometheus.Desc { @@ -44,6 +47,7 @@ func New(runnerFetcher *fetch.Fetcher[runners.Summary], orgFetcher *fetch.Fetche return &Collector{ runnerFetcher: runnerFetcher, orgFetcher: orgFetcher, + rateLimitFn: rateLimits, runnersUp: desc("runners", "up", "Whether the last scrape of the runners API succeeded (1) or a stale cache is being served (0)."+runnerNote, nil), @@ -57,9 +61,11 @@ func New(runnerFetcher *fetch.Fetcher[runners.Summary], orgFetcher *fetch.Fetche reposTotal: desc("org", "repos_total", "Number of non-archived repos in the org."+orgNote, []string{"visibility"}), rateLimit: desc("rate_limit", "remaining", - "Remaining core API rate-limit budget."+orgNote, nil), + "Remaining core API rate-limit budget, as of the latest GitHub response.", nil), rateLimitCap: desc("rate_limit", "limit", - "Total core API rate-limit budget."+orgNote, nil), + "Total core API rate-limit budget.", nil), + rateWindow: desc("rate_limit", "window_remaining", + "Remaining budget per core rate-limit window. GitHub counts per region, so a token can have several windows at once; github_rate_limit_remaining is the lowest.", []string{"reset"}), repoOpenPRs: desc("repo", "open_prs", "Number of open pull requests."+orgNote, []string{"repo"}), @@ -82,7 +88,7 @@ func New(runnerFetcher *fetch.Fetcher[runners.Summary], orgFetcher *fetch.Fetche func (c *Collector) Describe(ch chan<- *prometheus.Desc) { for _, d := range []*prometheus.Desc{ c.runnersUp, c.runnerUp, c.runnerBusy, - c.orgUp, c.reposTotal, c.rateLimit, c.rateLimitCap, + c.orgUp, c.reposTotal, c.rateLimit, c.rateLimitCap, c.rateWindow, c.repoOpenPRs, c.repoCILastRunConclusion, c.repoCILastRunAt, c.repoCIDuration, c.repoDependabot, } { ch <- d @@ -92,6 +98,23 @@ func (c *Collector) Describe(ch chan<- *prometheus.Desc) { func (c *Collector) Collect(ch chan<- prometheus.Metric) { c.collectRunners(ch) c.collectOrgStats(ch) + c.collectRateLimits(ch) +} + +// collectRateLimits reports every active core window plus the scarcest one, +// which is what runs out first. Absent until the first GitHub response. +func (c *Collector) collectRateLimits(ch chan<- prometheus.Metric) { + var low github.RateLimit + for reset, rl := range c.rateLimitFn() { + ch <- prometheus.MustNewConstMetric(c.rateWindow, prometheus.GaugeValue, float64(rl.Remaining), reset.Format(time.RFC3339)) + if low.Limit == 0 || rl.Remaining < low.Remaining { + low = rl + } + } + if low.Limit > 0 { + ch <- prometheus.MustNewConstMetric(c.rateLimit, prometheus.GaugeValue, float64(low.Remaining)) + ch <- prometheus.MustNewConstMetric(c.rateLimitCap, prometheus.GaugeValue, float64(low.Limit)) + } } func (c *Collector) collectRunners(ch chan<- prometheus.Metric) { @@ -122,8 +145,6 @@ func (c *Collector) collectOrgStats(ch chan<- prometheus.Metric) { return } ch <- prometheus.MustNewConstMetric(c.orgUp, prometheus.GaugeValue, 1) - ch <- prometheus.MustNewConstMetric(c.rateLimit, prometheus.GaugeValue, float64(s.RateLimit.Remaining)) - ch <- prometheus.MustNewConstMetric(c.rateLimitCap, prometheus.GaugeValue, float64(s.RateLimit.Limit)) visibilityCounts := map[string]int{} for _, r := range s.Repos { diff --git a/internal/github/client.go b/internal/github/client.go index 3d87452..1f6f57c 100644 --- a/internal/github/client.go +++ b/internal/github/client.go @@ -9,6 +9,7 @@ import ( "net/url" "regexp" "strconv" + "sync" "time" ) @@ -16,14 +17,21 @@ const apiBase = "https://api.github.com" type Client struct { baseURL string // apiBase; tests point it at httptest + now func() time.Time org string token string httpClient *http.Client + + mu sync.Mutex // both pollers share the client + // core budgets by reset time: GitHub counts per region, so one token has + // more than one window at once, depending on which region serves an endpoint + budgets map[int64]RateLimit } func NewClient(org, token string, timeout time.Duration) *Client { return &Client{ baseURL: apiBase, + now: time.Now, org: org, token: token, httpClient: &http.Client{Timeout: timeout}, @@ -43,9 +51,34 @@ func (c *Client) request(ctx context.Context, url string) (*http.Response, error if err != nil { return nil, fmt.Errorf("requesting %s: %w", url, err) } + c.recordRateLimit(resp.Header) return resp, nil } +// recordRateLimit keeps the core budget from a response's headers. GET +// /rate_limit can't be trusted for this: it reports used=0 for our tokens +// while these headers show the real count. +func (c *Client) recordRateLimit(h http.Header) { + if r := h.Get("X-RateLimit-Resource"); r != "" && r != "core" { + return + } + limit, lErr := strconv.Atoi(h.Get("X-RateLimit-Limit")) + remaining, rErr := strconv.Atoi(h.Get("X-RateLimit-Remaining")) + reset, sErr := strconv.ParseInt(h.Get("X-RateLimit-Reset"), 10, 64) + if lErr != nil || rErr != nil || sErr != nil { + return + } + c.mu.Lock() + defer c.mu.Unlock() + if c.budgets == nil { + c.budgets = map[int64]RateLimit{} + } + // within one window, the lowest count seen is the latest + if b, ok := c.budgets[reset]; !ok || remaining < b.Remaining { + c.budgets[reset] = RateLimit{Limit: limit, Remaining: remaining} + } +} + func (c *Client) get(ctx context.Context, url string, out interface{}) error { _, err := c.getPage(ctx, url, out) return err @@ -192,11 +225,18 @@ func (c *Client) DependabotAlerts(ctx context.Context, repo string) ([]Dependabo return out, nil } -// RateLimit returns the core rate-limit budget this client draws from. -func (c *Client) RateLimit(ctx context.Context) (RateLimit, error) { - var out rateLimitResponse - if err := c.get(ctx, c.baseURL+"/rate_limit", &out); err != nil { - return RateLimit{}, err +// RateLimits returns every core window that hasn't reset yet, by reset time. +func (c *Client) RateLimits() map[time.Time]RateLimit { + c.mu.Lock() + defer c.mu.Unlock() + now := c.now().Unix() + out := map[time.Time]RateLimit{} + for reset, b := range c.budgets { + if reset <= now { + delete(c.budgets, reset) + continue + } + out[time.Unix(reset, 0).UTC()] = b } - return out.Resources.Core, nil + return out } diff --git a/internal/github/client_test.go b/internal/github/client_test.go index d664e96..8cda2cd 100644 --- a/internal/github/client_test.go +++ b/internal/github/client_test.go @@ -3,6 +3,7 @@ package github import ( "context" "fmt" + "maps" "net/http" "net/http/httptest" "testing" @@ -61,3 +62,49 @@ func TestRunJobsPaginates(t *testing.T) { t.Fatalf("jobs = %+v", jobs) } } + +func TestRateLimitFromHeaders(t *testing.T) { + now := time.Unix(1000, 0) + // path -> remaining, reset; two windows at once, like GitHub does + budgets := map[string][2]string{ + "/orgs/org/repos": {"3790", "2000"}, + "/repos/org/repo/dependabot/alerts": {"4900", "3000"}, + "/search": {"1", "2000"}, + } + srv := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) { + b := budgets[r.URL.Path] + w.Header().Set("X-RateLimit-Limit", "5000") + w.Header().Set("X-RateLimit-Remaining", b[0]) + w.Header().Set("X-RateLimit-Reset", b[1]) + w.Header().Set("X-RateLimit-Resource", "core") + if r.URL.Path == "/search" { + w.Header().Set("X-RateLimit-Resource", "search") + } + _, _ = w.Write([]byte(`[]`)) + })) + defer srv.Close() + + c := NewClient("org", "token", time.Second) + c.baseURL = srv.URL + c.now = func() time.Time { return now } + if got := c.RateLimits(); len(got) != 0 { + t.Fatalf("before any request: %+v", got) + } + ctx := context.Background() + var out []any + _, _ = c.Repos(ctx) + _, _ = c.DependabotAlerts(ctx, "repo") + _ = c.get(ctx, srv.URL+"/search", &out) // another resource's budget + want := map[time.Time]RateLimit{ + time.Unix(2000, 0).UTC(): {Limit: 5000, Remaining: 3790}, + time.Unix(3000, 0).UTC(): {Limit: 5000, Remaining: 4900}, + } + if got := c.RateLimits(); !maps.Equal(got, want) { + t.Fatalf("RateLimits = %+v, want both core windows and not search", got) + } + + now = time.Unix(2500, 0) // the 3790 window has reset + if got := c.RateLimits(); len(got) != 1 || got[time.Unix(3000, 0).UTC()].Remaining != 4900 { + t.Fatalf("after reset: %+v, want only the later window", got) + } +} diff --git a/internal/github/types.go b/internal/github/types.go index cf41918..f141da7 100644 --- a/internal/github/types.go +++ b/internal/github/types.go @@ -85,14 +85,8 @@ type DependabotAlert struct { } `json:"security_advisory"` } -// RateLimit is the "core" resource from GET /rate_limit. +// RateLimit is the "core" API budget, from X-RateLimit-* response headers. type RateLimit struct { - Limit int `json:"limit"` - Remaining int `json:"remaining"` -} - -type rateLimitResponse struct { - Resources struct { - Core RateLimit `json:"core"` - } `json:"resources"` + Limit int + Remaining int } diff --git a/internal/orgstats/feed_test.go b/internal/orgstats/feed_test.go index ea6bc92..0037e6d 100644 --- a/internal/orgstats/feed_test.go +++ b/internal/orgstats/feed_test.go @@ -28,9 +28,6 @@ type fakeClient struct { func (c *fakeClient) Repos(context.Context) ([]github.Repo, error) { return []github.Repo{{Name: "repo"}, {Name: "old", Archived: true}}, c.reposErr } -func (c *fakeClient) RateLimit(context.Context) (github.RateLimit, error) { - return github.RateLimit{Limit: 5000, Remaining: 4000}, nil -} func (c *fakeClient) OpenPRCount(context.Context, string) (int, error) { return 0, errors.New("forbidden") } @@ -245,7 +242,7 @@ func TestBuild(t *testing.T) { t.Fatal(err) } // archived repo dropped, disabled workflow dropped, per-repo errors swallowed - if len(s.Repos) != 1 || len(s.Repos[0].Workflows) != 1 || !s.Repos[0].Workflows[0].HasRun || s.RateLimit.Remaining != 4000 { + if len(s.Repos) != 1 || len(s.Repos[0].Workflows) != 1 || !s.Repos[0].Workflows[0].HasRun { t.Fatalf("summary = %+v", s) } diff --git a/internal/orgstats/orgstats.go b/internal/orgstats/orgstats.go index 327976b..992bb52 100644 --- a/internal/orgstats/orgstats.go +++ b/internal/orgstats/orgstats.go @@ -39,13 +39,11 @@ type RepoStats struct { type Summary struct { GeneratedAt time.Time Repos []RepoStats - RateLimit github.RateLimit } // Client is the subset of *github.Client this package calls; tests fake it. type Client interface { Repos(ctx context.Context) ([]github.Repo, error) - RateLimit(ctx context.Context) (github.RateLimit, error) OpenPRCount(ctx context.Context, repo string) (int, error) Workflows(ctx context.Context, repo string) ([]github.Workflow, error) DependabotAlerts(ctx context.Context, repo string) ([]github.DependabotAlert, error) @@ -60,12 +58,7 @@ func Build(ctx context.Context, client Client, feed *Feed) (Summary, error) { return Summary{}, err } - rateLimit, err := client.RateLimit(ctx) - if err != nil { - return Summary{}, err - } - - s := Summary{GeneratedAt: time.Now(), RateLimit: rateLimit} + s := Summary{GeneratedAt: time.Now()} for _, r := range repos { if r.Archived { continue diff --git a/main.go b/main.go index 74ba789..bab5f9a 100644 --- a/main.go +++ b/main.go @@ -41,7 +41,7 @@ func main() { }, cfg.OrgCacheTTL, cfg.OrgCacheMaxStale, log) registry := prometheus.NewRegistry() - registry.MustRegister(collector.New(runnerFetcher, orgFetcher, cfg.RunnerCacheTTL, cfg.OrgCacheTTL), feed) + registry.MustRegister(collector.New(runnerFetcher, orgFetcher, client.RateLimits, cfg.RunnerCacheTTL, cfg.OrgCacheTTL), feed) mux := http.NewServeMux() mux.Handle("/metrics", promhttp.HandlerFor(registry, promhttp.HandlerOpts{}))