From 984897518efa19e175597a65023329b50d065788 Mon Sep 17 00:00:00 2001 From: baron unread Date: Sat, 26 Sep 2026 03:37:51 +0200 Subject: [PATCH 1/5] Explain selections: whole-suite rules, a why line, fuller console output Feedback from leanest's first real run (rdyrct #263: 37/37 selected, with no way to tell why from the log). - Whole-suite rules, decided before the judge: a change to runner config (playwright/vitest/vite config, package.json, lockfiles, CI workflows) runs everything; a Markdown-only change skips everything. No judge call, so matrix shards can't disagree on these. - "Why they run: N by rule, N judge unsure (c < 0.5), N judged affected." in the console and the PR comment. - Console: each RUN line carries its reason, and the RUN and Changed lists say "... and N more" past 20. - inspect during a judge outage selects every test and says why, instead of showing nothing selected. Co-Authored-By: Claude Opus 5.5 --- README.md | 1 + src/cli.test.ts | 9 +++++++- src/cli.ts | 45 ++++++++++++++++++++++++++++-------- src/leanest.ts | 31 ++++++++++++++++++++++--- src/selection-policy.test.ts | 32 ++++++++++++++++++++++++- src/selection-policy.ts | 28 +++++++++++++++++++++- src/types.ts | 4 ++++ 7 files changed, 135 insertions(+), 15 deletions(-) diff --git a/README.md b/README.md index a51a171..8ec7603 100644 --- a/README.md +++ b/README.md @@ -66,6 +66,7 @@ Tests that are confidently irrelevant get skipped. Everything else runs through - **Fail open**: uncertainty means RUN. A missing API key, an API timeout, or a malformed response always falls back to running the full suite, loudly (`⚠ Judge unavailable (...), running the full suite.`). Finding no tests at all is an error (exit 1), not a silent pass. - **Deterministic overrides**: no threshold decides these, the judge isn't even asked. A test whose own file changed always runs, as does one that statically imports a changed file, or that navigates a route a changed file's own path names (e.g. `page.goto("/admin/users")` against a changed `routes/admin/users.tsx`) -- a heuristic that catches e2e route coupling no import graph can see, since a browser test never imports the page it drives. +- **Whole-suite rules**: before any per-test decision, a change to the runner's own setup (`playwright.config.*`, `vitest.config.*`, `vite.config.*`, `package.json`, a lockfile, or anything in `.github/workflows/`) runs every test, and a change that only touches Markdown files skips every test. Neither asks the judge, so every shard of a matrix gets the same answer. - **Leanest doesn't run tests itself**: it selects file paths and hands them to your actual runner (`playwright test `, `vitest run `). It leaves reporters, retries, sharding, and CI-required-check behavior alone. Anything after `--` goes straight to the runner: `leanest playwright -- --shard=1/3`. - **Static checks are out of scope on purpose**: lint/format/typecheck are already fast at full scope, and semantic per-rule selection would add latency for no real payoff. Leanest spends its Jev budget only on suites that are expensive to run in full: e2e today, more later. diff --git a/src/cli.test.ts b/src/cli.test.ts index dc83a79..fca6e18 100644 --- a/src/cli.test.ts +++ b/src/cli.test.ts @@ -1,5 +1,5 @@ import { describe, expect, test } from "bun:test"; -import { parseFlags, renderReport, reportMarker } from "./cli.js"; +import { explainRuns, parseFlags, renderReport, reportMarker } from "./cli.js"; import type { SelectionResult, TestCase } from "./types.js"; describe("parseFlags", () => { @@ -103,6 +103,13 @@ describe("renderReport", () => { expect(md).toContain("| `a\\|b.spec.ts` | RUN | judge unavailable retry |"); }); + test("says why the selected tests run, in the comment too", () => { + const r = result({ runBreakdown: { rule: 1, judgeUnsure: 30, judgeLikely: 0 } }); + expect(explainRuns(r)).toBe("Why they run: 1 by rule, 30 judge unsure (c < 0.5)."); + expect(renderReport("playwright", ".", r, false)).toContain("Why they run: 1 by rule"); + expect(explainRuns(result({ selectedTests: [] }))).toBeNull(); + }); + test("marker differs per framework and dir, so each run keeps its own comment", () => { expect(reportMarker("playwright", ".")).not.toBe(reportMarker("vitest", ".")); expect(reportMarker("playwright", "apps/web")).not.toBe(reportMarker("playwright", ".")); diff --git a/src/cli.ts b/src/cli.ts index b406d01..793e038 100755 --- a/src/cli.ts +++ b/src/cli.ts @@ -3,7 +3,7 @@ import { appendFileSync, realpathSync } from "node:fs"; import { fileURLToPath } from "node:url"; import { Leanest } from "./leanest.js"; -import { SelectionPolicy } from "./selection-policy.js"; +import { MIN_CONFIDENCE, SelectionPolicy } from "./selection-policy.js"; import { runTests } from "./runner.js"; import type { SelectionResult, TestCase } from "./types.js"; @@ -165,6 +165,11 @@ function printInspect(result: any): void { } console.log(`Discovering ${discovered.framework} tests...`); console.log(` ${discovered.count} tests found`); + if (result.error) { + console.log( + `\n⚠ Judge unavailable (${result.error}), all ${discovered.count} tests would run.`, + ); + } if (result.evaluated.length > 0) { console.log(`\nEvaluating semantic impact...`); console.log(` ${result.evaluated.length} tests evaluated`); @@ -182,25 +187,45 @@ function printInspect(result: any): void { console.log(`\nSkipping ${result.skipped} tests.`); } -function printSelect(result: any): void { +/** One line on why the selected tests run, or null when there's nothing to explain. */ +export function explainRuns(result: SelectionResult): string | null { + const b = result.runBreakdown; + if (!b || result.selectedTests.length === 0) return null; + const parts = [ + b.rule > 0 ? `${b.rule} by rule` : "", + b.judgeUnsure > 0 ? `${b.judgeUnsure} judge unsure (c < ${MIN_CONFIDENCE})` : "", + b.judgeLikely > 0 ? `${b.judgeLikely} judged affected` : "", + ].filter(Boolean); + return `Why they run: ${parts.join(", ")}.`; +} + +function printList(lines: string[]): void { + for (const line of lines.slice(0, 20)) console.log(` ${line}`); + if (lines.length > 20) console.log(` ... and ${lines.length - 20} more`); +} + +function printSelect(result: SelectionResult): void { const noChanges = result.changedFiles.length === 0; if (noChanges && result.totalTests > 0) { console.log(`No changes detected.`); console.log(`Evaluating all ${result.totalTests} tests conservatively...\n`); } else { console.log(`Changed:`); - for (const f of result.changedFiles.slice(0, 20)) { - console.log(` ${f}`); - } + printList(result.changedFiles); console.log(``); } console.log(`${result.totalTests} tests found`); console.log(`\nSelected ${result.selectedTests.length} / ${result.totalTests} tests`); - for (const test of result.selectedTests.slice(0, 20)) { - console.log(` RUN ${test.identity.path}`); - } + const why = explainRuns(result); + if (why) console.log(why); + printList( + result.selectedTests.map((t) => `RUN ${t.identity.path} (${result.reasons[t.identity.path]})`), + ); if (result.skippedTests > 0) { - console.log(`\nSkipping ${result.skippedTests} tests.`); + const reasons = new Set(result.skipped.map((t) => result.reasons[t.identity.path])); + // A suite rule gives every skip the same reason; say it once. + const shared = reasons.size === 1 ? ` (${[...reasons][0]})` : ""; + console.log(`\nSkipping ${result.skippedTests} tests.${shared}`); } } @@ -239,6 +264,7 @@ export function renderReport( : [`
${summary}`, "", ...table(rows), "", "
", ""]; const runRows = result.selectedTests.map((t) => row(t, "RUN")); const skipRows = result.skipped.map((t) => row(t, "SKIP")); + const why = explainRuns(result); if (result.status === "error") { return [ @@ -259,6 +285,7 @@ export function renderReport( reportMarker(command, dir), `### leanest: ${result.selectedTests.length} of ${result.totalTests} ${command} test files selected`, "", + ...(why ? [why, ""] : []), ...(runningAll ? ["The full suite runs anyway (`--shadow` or `--full`).", ""] : []), ...(runRows.length > 0 ? [...table(runRows), ""] : []), ...details(`${skipRows.length} skipped`, skipRows), diff --git a/src/leanest.ts b/src/leanest.ts index fcaf641..a96bb71 100644 --- a/src/leanest.ts +++ b/src/leanest.ts @@ -6,7 +6,7 @@ import { import { ChangeResolver, type GitChange } from "./git-diff.js"; import { TestDiscovery } from "./test-discovery.js"; import { ContextBuilder } from "./context-builder.js"; -import { SelectionPolicy } from "./selection-policy.js"; +import { MIN_CONFIDENCE, SelectionPolicy, suiteRule } from "./selection-policy.js"; import { importsChangedFile } from "./import-graph.js"; import { touchesSameRoute } from "./route-heuristic.js"; import type { TestCase, SelectionResult, PipelineResult } from "./types.js"; @@ -58,14 +58,15 @@ export class Leanest { let answers: Record; try { answers = await this.judge.evaluate(state, questions); - } catch { + } catch (error) { return { change, discovered: { framework, count: discovery.tests.length, tests: discovery.tests }, evaluated: [], - selected: [], + selected: discovery.tests, skipped: 0, decision: "RUN", + error: error instanceof Error ? error.message : String(error), }; } @@ -120,6 +121,25 @@ export class Leanest { }; } + const rule = suiteRule(change.changedFiles); + if (rule) { + const run = rule.decision === "RUN" ? tests : []; + return { + command: "select", + args: [framework], + status: "complete", + totalTests: tests.length, + selectedTests: run, + skippedTests: tests.length - run.length, + runTests: run, + skipped: rule.decision === "SKIP" ? tests : [], + reasons: Object.fromEntries(tests.map((t) => [t.identity.path, rule.reason])), + runBreakdown: { rule: run.length, judgeUnsure: 0, judgeLikely: 0 }, + changedFiles: change.changedFiles, + diff: change.diff, + }; + } + const state = this.context.buildState(change, tests); const questions = this.buildQuestions(tests, change.changedFiles.length === 0); let answers: Record; @@ -155,6 +175,7 @@ export class Leanest { const runTests: TestCase[] = []; const skipTests: TestCase[] = []; const reasons: Record = {}; + const runBreakdown = { rule: 0, judgeUnsure: 0, judgeLikely: 0 }; for (const entry of ranked) { const deterministic = this.deterministicReason(entry.test, change); @@ -164,6 +185,9 @@ export class Leanest { this.policy.decide(entry.probability, entry.confidence, deterministic !== null) === "RUN" ) { runTests.push(entry.test); + if (deterministic) runBreakdown.rule++; + else if (entry.confidence < MIN_CONFIDENCE) runBreakdown.judgeUnsure++; + else runBreakdown.judgeLikely++; } else { skipTests.push(entry.test); } @@ -179,6 +203,7 @@ export class Leanest { runTests, skipped: skipTests, reasons, + runBreakdown, changedFiles: change.changedFiles, diff: change.diff, }; diff --git a/src/selection-policy.test.ts b/src/selection-policy.test.ts index 58c191b..0cf60ce 100644 --- a/src/selection-policy.test.ts +++ b/src/selection-policy.test.ts @@ -1,5 +1,5 @@ import { describe, expect, test } from "bun:test"; -import { SelectionPolicy } from "./selection-policy.js"; +import { SelectionPolicy, suiteRule } from "./selection-policy.js"; describe("SelectionPolicy.decide", () => { const policy = new SelectionPolicy(); @@ -25,3 +25,33 @@ describe("SelectionPolicy.decide", () => { expect(policy.decide(0.1, undefined, false)).toBe("RUN"); }); }); + +describe("suiteRule", () => { + test("runs everything when the runner's setup changed", () => { + for (const f of [ + "playwright.config.ts", + "apps/web/vitest.config.mts", + "package.json", + "bun.lock", + ".github/workflows/test.yml", + ]) { + expect(suiteRule(["src/a.ts", f])?.decision).toBe("RUN"); + } + }); + + test("skips everything when only Markdown changed", () => { + expect(suiteRule(["AGENTS.md", "docs/guide.md"])).toEqual({ + decision: "SKIP", + reason: "only Markdown changed", + }); + }); + + test("asks per test otherwise, and on no changes", () => { + expect(suiteRule(["README.md", "src/a.ts"])).toBeNull(); + expect(suiteRule([])).toBeNull(); + }); + + test("runner config wins over Markdown", () => { + expect(suiteRule(["README.md", ".github/workflows/test.yml"])?.decision).toBe("RUN"); + }); +}); diff --git a/src/selection-policy.ts b/src/selection-policy.ts index e10119b..a757753 100644 --- a/src/selection-policy.ts +++ b/src/selection-policy.ts @@ -1,5 +1,31 @@ import type { TestCase } from "./types.js"; +/** Below this judge confidence the probability isn't trusted, and the test runs. */ +export const MIN_CONFIDENCE = 0.5; + +// Files every test depends on: the runner's config, dependencies, and CI workflows. +const RUNNER_CONFIG = [ + /(^|\/)(playwright|vitest|vite)\.config\.[cm]?[jt]s$/, + /(^|\/)package\.json$/, + /(^|\/)(package-lock\.json|bun\.lockb?|pnpm-lock\.yaml|yarn\.lock)$/, + /^\.github\/workflows\//, +]; + +/** + * A decision for the whole suite that needs no judge: run everything when the runner's + * own setup changed, skip everything when only Markdown changed. Null means ask per test. + */ +export function suiteRule( + changedFiles: string[], +): { decision: "RUN" | "SKIP"; reason: string } | null { + const config = changedFiles.find((f) => RUNNER_CONFIG.some((re) => re.test(f))); + if (config) return { decision: "RUN", reason: `runner config changed: ${config}` }; + if (changedFiles.length > 0 && changedFiles.every((f) => f.endsWith(".md"))) { + return { decision: "SKIP", reason: "only Markdown changed" }; + } + return null; +} + export class SelectionPolicy { decide( probability: number | undefined | null, @@ -9,7 +35,7 @@ export class SelectionPolicy { if (testChanged) return "RUN"; if (probability === undefined || probability === null) return "RUN"; if (confidence === undefined || confidence === null) return "RUN"; - if (confidence < 0.5) return "RUN"; + if (confidence < MIN_CONFIDENCE) return "RUN"; if (probability < 0.3) return "SKIP"; return "RUN"; } diff --git a/src/types.ts b/src/types.ts index 9a96558..5e39e79 100644 --- a/src/types.ts +++ b/src/types.ts @@ -43,6 +43,8 @@ export interface SelectionResult { skipped: TestCase[]; /** Why each test path was run or skipped. */ reasons: Record; + /** Why the selected tests run: forced by a rule, the judge too unsure to skip, or judged affected. */ + runBreakdown?: { rule: number; judgeUnsure: number; judgeLikely: number }; decision?: string; changedFiles: string[]; diff: string; @@ -59,4 +61,6 @@ export interface PipelineResult { selected: TestCase[]; skipped: number; decision: "RUN" | "SKIP"; + /** Set when the judge failed; every test is then selected. */ + error?: string; } From 0c44c649d51637be9d542167e53a7db9af8f4685 Mon Sep 17 00:00:00 2001 From: baron unread Date: Sun, 27 Sep 2026 01:47:38 +0200 Subject: [PATCH 2/5] Say why tests run in a plain sentence MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit "37 by rule" read like a log line. Whole-suite rules now say their reason once ("All 37 run because the runner setup changed (…)", "Skipping all 37 tests: only Markdown changed.") instead of on every test, and mixed runs get a sentence: "2 touch the change directly, the judge wasn't sure enough to skip 9 and it thinks 1 is affected." Co-Authored-By: Claude Opus 5.5 --- src/cli.test.ts | 17 +++++++++++++---- src/cli.ts | 41 +++++++++++++++++++++++++++++------------ src/leanest.ts | 2 +- src/selection-policy.ts | 2 +- src/types.ts | 2 ++ 5 files changed, 46 insertions(+), 18 deletions(-) diff --git a/src/cli.test.ts b/src/cli.test.ts index fca6e18..c8b252f 100644 --- a/src/cli.test.ts +++ b/src/cli.test.ts @@ -103,11 +103,20 @@ describe("renderReport", () => { expect(md).toContain("| `a\\|b.spec.ts` | RUN | judge unavailable retry |"); }); - test("says why the selected tests run, in the comment too", () => { - const r = result({ runBreakdown: { rule: 1, judgeUnsure: 30, judgeLikely: 0 } }); - expect(explainRuns(r)).toBe("Why they run: 1 by rule, 30 judge unsure (c < 0.5)."); - expect(renderReport("playwright", ".", r, false)).toContain("Why they run: 1 by rule"); + test("says why the selected tests run, in plain words, in the comment too", () => { + const why = (rule: number, judgeUnsure: number, judgeLikely: number) => + explainRuns(result({ runBreakdown: { rule, judgeUnsure, judgeLikely } })); + expect(why(2, 9, 1)).toBe( + "2 touch the change directly, the judge wasn't sure enough to skip 9 and it thinks 1 is affected.", + ); + expect(why(0, 37, 0)).toBe("The judge wasn't sure enough to skip 37."); + expect(why(1, 0, 3)).toBe("1 touches the change directly and the judge thinks 3 are affected."); + expect(explainRuns(result({ suiteReason: "runner setup changed (package.json)" }))).toBe( + "The only test runs because the runner setup changed (package.json).", + ); expect(explainRuns(result({ selectedTests: [] }))).toBeNull(); + const r = result({ runBreakdown: { rule: 1, judgeUnsure: 0, judgeLikely: 0 } }); + expect(renderReport("playwright", ".", r, false)).toContain("1 touches the change directly."); }); test("marker differs per framework and dir, so each run keeps its own comment", () => { diff --git a/src/cli.ts b/src/cli.ts index 793e038..bf3d6dd 100755 --- a/src/cli.ts +++ b/src/cli.ts @@ -3,7 +3,7 @@ import { appendFileSync, realpathSync } from "node:fs"; import { fileURLToPath } from "node:url"; import { Leanest } from "./leanest.js"; -import { MIN_CONFIDENCE, SelectionPolicy } from "./selection-policy.js"; +import { SelectionPolicy } from "./selection-policy.js"; import { runTests } from "./runner.js"; import type { SelectionResult, TestCase } from "./types.js"; @@ -187,16 +187,26 @@ function printInspect(result: any): void { console.log(`\nSkipping ${result.skipped} tests.`); } -/** One line on why the selected tests run, or null when there's nothing to explain. */ +/** One sentence on why the selected tests run, or null when there's nothing to explain. */ export function explainRuns(result: SelectionResult): string | null { + const n = result.selectedTests.length; + if (n === 0) return null; + // The only whole-suite rule that runs tests is "runner setup changed (…)". + if (result.suiteReason) { + return `${n === 1 ? "The only test runs" : `All ${n} run`} because the ${result.suiteReason}.`; + } const b = result.runBreakdown; - if (!b || result.selectedTests.length === 0) return null; + if (!b) return null; const parts = [ - b.rule > 0 ? `${b.rule} by rule` : "", - b.judgeUnsure > 0 ? `${b.judgeUnsure} judge unsure (c < ${MIN_CONFIDENCE})` : "", - b.judgeLikely > 0 ? `${b.judgeLikely} judged affected` : "", + b.rule > 0 ? `${b.rule} ${b.rule === 1 ? "touches" : "touch"} the change directly` : "", + b.judgeUnsure > 0 ? `the judge wasn't sure enough to skip ${b.judgeUnsure}` : "", + b.judgeLikely > 0 + ? `${b.judgeUnsure > 0 ? "it" : "the judge"} thinks ${b.judgeLikely === 1 ? "1 is" : `${b.judgeLikely} are`} affected` + : "", ].filter(Boolean); - return `Why they run: ${parts.join(", ")}.`; + const sentence = + parts.length > 1 ? `${parts.slice(0, -1).join(", ")} and ${parts.at(-1)}` : parts[0]!; + return `${sentence.charAt(0).toUpperCase()}${sentence.slice(1)}.`; } function printList(lines: string[]): void { @@ -218,14 +228,21 @@ function printSelect(result: SelectionResult): void { console.log(`\nSelected ${result.selectedTests.length} / ${result.totalTests} tests`); const why = explainRuns(result); if (why) console.log(why); + // A whole-suite rule already said why in one line; don't repeat it per test. printList( - result.selectedTests.map((t) => `RUN ${t.identity.path} (${result.reasons[t.identity.path]})`), + result.selectedTests.map((t) => + result.suiteReason + ? `RUN ${t.identity.path}` + : `RUN ${t.identity.path} (${result.reasons[t.identity.path]})`, + ), ); if (result.skippedTests > 0) { - const reasons = new Set(result.skipped.map((t) => result.reasons[t.identity.path])); - // A suite rule gives every skip the same reason; say it once. - const shared = reasons.size === 1 ? ` (${[...reasons][0]})` : ""; - console.log(`\nSkipping ${result.skippedTests} tests.${shared}`); + const n = result.skippedTests; + console.log( + result.suiteReason + ? `\nSkipping all ${n} tests: ${result.suiteReason}.` + : `\nSkipping ${n} tests.`, + ); } } diff --git a/src/leanest.ts b/src/leanest.ts index a96bb71..c62958d 100644 --- a/src/leanest.ts +++ b/src/leanest.ts @@ -134,7 +134,7 @@ export class Leanest { runTests: run, skipped: rule.decision === "SKIP" ? tests : [], reasons: Object.fromEntries(tests.map((t) => [t.identity.path, rule.reason])), - runBreakdown: { rule: run.length, judgeUnsure: 0, judgeLikely: 0 }, + suiteReason: rule.reason, changedFiles: change.changedFiles, diff: change.diff, }; diff --git a/src/selection-policy.ts b/src/selection-policy.ts index a757753..7ba6cd5 100644 --- a/src/selection-policy.ts +++ b/src/selection-policy.ts @@ -19,7 +19,7 @@ export function suiteRule( changedFiles: string[], ): { decision: "RUN" | "SKIP"; reason: string } | null { const config = changedFiles.find((f) => RUNNER_CONFIG.some((re) => re.test(f))); - if (config) return { decision: "RUN", reason: `runner config changed: ${config}` }; + if (config) return { decision: "RUN", reason: `runner setup changed (${config})` }; if (changedFiles.length > 0 && changedFiles.every((f) => f.endsWith(".md"))) { return { decision: "SKIP", reason: "only Markdown changed" }; } diff --git a/src/types.ts b/src/types.ts index 5e39e79..1de0756 100644 --- a/src/types.ts +++ b/src/types.ts @@ -45,6 +45,8 @@ export interface SelectionResult { reasons: Record; /** Why the selected tests run: forced by a rule, the judge too unsure to skip, or judged affected. */ runBreakdown?: { rule: number; judgeUnsure: number; judgeLikely: number }; + /** Set when a whole-suite rule decided every test, e.g. "only Markdown changed". */ + suiteReason?: string; decision?: string; changedFiles: string[]; diff: string; From 9725c91496d9c7954a510464d8b5d54b3106777c Mon Sep 17 00:00:00 2001 From: baron unread Date: Sun, 27 Sep 2026 01:54:03 +0200 Subject: [PATCH 3/5] Report: fold the RUN table when a whole-suite rule decided Every row repeats the reason the sentence above already gives, so the 37 rows go into a collapsed section like the skipped ones. Co-Authored-By: Claude Opus 5.5 --- src/cli.test.ts | 9 +++++++++ src/cli.ts | 7 ++++++- 2 files changed, 15 insertions(+), 1 deletion(-) diff --git a/src/cli.test.ts b/src/cli.test.ts index c8b252f..21795e6 100644 --- a/src/cli.test.ts +++ b/src/cli.test.ts @@ -115,6 +115,15 @@ describe("renderReport", () => { "The only test runs because the runner setup changed (package.json).", ); expect(explainRuns(result({ selectedTests: [] }))).toBeNull(); + const suite = renderReport( + "playwright", + ".", + result({ suiteReason: "runner setup changed (package.json)" }), + false, + ); + expect(suite.indexOf("
1 test file")).toBeLessThan( + suite.indexOf("| `a.spec.ts` | RUN |"), + ); const r = result({ runBreakdown: { rule: 1, judgeUnsure: 0, judgeLikely: 0 } }); expect(renderReport("playwright", ".", r, false)).toContain("1 touches the change directly."); }); diff --git a/src/cli.ts b/src/cli.ts index bf3d6dd..e0ab541 100755 --- a/src/cli.ts +++ b/src/cli.ts @@ -304,7 +304,12 @@ export function renderReport( "", ...(why ? [why, ""] : []), ...(runningAll ? ["The full suite runs anyway (`--shadow` or `--full`).", ""] : []), - ...(runRows.length > 0 ? [...table(runRows), ""] : []), + // A whole-suite rule gives every row the same reason, already said above: fold them. + ...(result.suiteReason + ? details(`${runRows.length} test file${runRows.length === 1 ? "" : "s"}`, runRows) + : runRows.length > 0 + ? [...table(runRows), ""] + : []), ...details(`${skipRows.length} skipped`, skipRows), ].join("\n"); } From 237ed426b4b17adcafbd657adf23286b6eec7625 Mon Sep 17 00:00:00 2001 From: baron unread Date: Sun, 27 Sep 2026 02:12:00 +0200 Subject: [PATCH 4/5] Say why nothing runs when only Markdown changed The Markdown-only comment showed "0 of 37" and a collapsed table with no sentence; it now reads "Nothing runs because only Markdown changed." Co-Authored-By: Claude Opus 5.5 --- src/cli.test.ts | 3 +++ src/cli.ts | 10 +++------- 2 files changed, 6 insertions(+), 7 deletions(-) diff --git a/src/cli.test.ts b/src/cli.test.ts index 21795e6..35819d4 100644 --- a/src/cli.test.ts +++ b/src/cli.test.ts @@ -115,6 +115,9 @@ describe("renderReport", () => { "The only test runs because the runner setup changed (package.json).", ); expect(explainRuns(result({ selectedTests: [] }))).toBeNull(); + expect(explainRuns(result({ selectedTests: [], suiteReason: "only Markdown changed" }))).toBe( + "Nothing runs because only Markdown changed.", + ); const suite = renderReport( "playwright", ".", diff --git a/src/cli.ts b/src/cli.ts index e0ab541..47e74b4 100755 --- a/src/cli.ts +++ b/src/cli.ts @@ -190,8 +190,8 @@ function printInspect(result: any): void { /** One sentence on why the selected tests run, or null when there's nothing to explain. */ export function explainRuns(result: SelectionResult): string | null { const n = result.selectedTests.length; - if (n === 0) return null; - // The only whole-suite rule that runs tests is "runner setup changed (…)". + // Whole-suite rules: "only Markdown changed" skips all, "runner setup changed (…)" runs all. + if (n === 0) return result.suiteReason ? `Nothing runs because ${result.suiteReason}.` : null; if (result.suiteReason) { return `${n === 1 ? "The only test runs" : `All ${n} run`} because the ${result.suiteReason}.`; } @@ -238,11 +238,7 @@ function printSelect(result: SelectionResult): void { ); if (result.skippedTests > 0) { const n = result.skippedTests; - console.log( - result.suiteReason - ? `\nSkipping all ${n} tests: ${result.suiteReason}.` - : `\nSkipping ${n} tests.`, - ); + console.log(`\nSkipping ${n} tests.`); } } From 7754dd10a6a9c49aa924bf72d72d9498819fc796 Mon Sep 17 00:00:00 2001 From: baron unread Date: Sun, 27 Sep 2026 23:55:55 +0200 Subject: [PATCH 5/5] Suite rules: Markdown-only still runs forced tests; inspect agrees MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit From a cold review of this PR: - "Only Markdown changed" skipped every test, overriding the rules that force a test to run: a test importing the changed .md file was skipped. Those tests now run with their own reason; the rest skip. - inspect never applied the whole-suite rules, so it could disagree with select and the real run. It now does, and says so. - Under --shadow/--full the comment said "Nothing runs because…" and then "The full suite runs anyway". The sentence is left out there. Adds an end-to-end test on a real git repo, which fails without the fix. Co-Authored-By: Claude Opus 5.5 --- src/cli.ts | 7 +++++- src/leanest.test.ts | 58 +++++++++++++++++++++++++++++++++++++++++++++ src/leanest.ts | 51 ++++++++++++++++++++++++++++++++------- src/types.ts | 3 ++- 4 files changed, 108 insertions(+), 11 deletions(-) create mode 100644 src/leanest.test.ts diff --git a/src/cli.ts b/src/cli.ts index 47e74b4..41cdd8f 100755 --- a/src/cli.ts +++ b/src/cli.ts @@ -170,6 +170,10 @@ function printInspect(result: any): void { `\n⚠ Judge unavailable (${result.error}), all ${discovered.count} tests would run.`, ); } + if (result.suiteReason) { + console.log(`\nNo judge needed: ${result.suiteReason}.`); + console.log(`Selected ${result.selected.length} / ${discovered.count} tests`); + } if (result.evaluated.length > 0) { console.log(`\nEvaluating semantic impact...`); console.log(` ${result.evaluated.length} tests evaluated`); @@ -298,7 +302,8 @@ export function renderReport( reportMarker(command, dir), `### leanest: ${result.selectedTests.length} of ${result.totalTests} ${command} test files selected`, "", - ...(why ? [why, ""] : []), + // Under --shadow/--full everything runs, so "Nothing runs because…" would contradict it. + ...(why && !runningAll ? [why, ""] : []), ...(runningAll ? ["The full suite runs anyway (`--shadow` or `--full`).", ""] : []), // A whole-suite rule gives every row the same reason, already said above: fold them. ...(result.suiteReason diff --git a/src/leanest.test.ts b/src/leanest.test.ts new file mode 100644 index 0000000..3aa3e2d --- /dev/null +++ b/src/leanest.test.ts @@ -0,0 +1,58 @@ +import { afterAll, beforeAll, describe, expect, test } from "bun:test"; +import { execFileSync } from "child_process"; +import { mkdtempSync, rmSync, writeFileSync } from "fs"; +import { tmpdir } from "os"; +import { join } from "path"; +import { Leanest } from "./leanest.js"; + +// A real git repo, so the whole-suite rules run end to end. Neither case calls the judge. +// Runs from the repo root with dir ".", like the Action's default. +describe("Leanest.select with a Markdown-only change", () => { + const root = mkdtempSync(join(tmpdir(), "leanest-suite-")); + const git = (...args: string[]) => + execFileSync("git", ["-c", "user.name=t", "-c", "user.email=t@t", ...args], { cwd: root }); + + const startDir = process.cwd(); + + beforeAll(() => { + process.chdir(root); + git("init", "-q"); + writeFileSync(join(root, "terms.md"), "v1"); + writeFileSync(join(root, "README.md"), "v1"); + writeFileSync( + join(root, "terms.test.ts"), + `import terms from "./terms.md";\ntest("x", () => {});`, + ); + writeFileSync(join(root, "math.test.ts"), `test("y", () => {});`); + git("add", "."); + git("commit", "-qm", "base"); + }); + + afterAll(() => { + process.chdir(startDir); + rmSync(root, { recursive: true, force: true }); + }); + + const change = (file: string) => { + writeFileSync(join(root, file), `${Date.now()}`); + git("commit", "-qam", file); + }; + const names = (tests: { identity: { path: string } }[]) => + tests.map((t) => t.identity.path.split("/").pop()); + + test("skips everything when no test depends on the Markdown", async () => { + change("README.md"); + const result = await new Leanest(".", "HEAD~1").select("vitest"); + expect(result.selectedTests).toEqual([]); + expect(result.suiteReason).toBe("only Markdown changed"); + }); + + test("still runs a test that imports the changed Markdown", async () => { + change("terms.md"); + const result = await new Leanest(".", "HEAD~1").select("vitest"); + expect(names(result.selectedTests)).toEqual(["terms.test.ts"]); + expect(names(result.skipped)).toEqual(["math.test.ts"]); + // The rule no longer explains every test, so no shared reason. + expect(result.suiteReason).toBeUndefined(); + }); +}); diff --git a/src/leanest.ts b/src/leanest.ts index c62958d..e325d83 100644 --- a/src/leanest.ts +++ b/src/leanest.ts @@ -53,6 +53,19 @@ export class Leanest { }; } + const suite = this.suiteDecision(discovery.tests, change); + if (suite) { + return { + change, + discovered: { framework, count: discovery.tests.length, tests: discovery.tests }, + evaluated: [], + selected: suite.run, + skipped: suite.skip.length, + decision: "RUN", + suiteReason: suite.rule.reason, + }; + } + const state = this.context.buildState(change, discovery.tests); const questions = this.buildQuestions(discovery.tests, change.changedFiles.length === 0); let answers: Record; @@ -121,20 +134,20 @@ export class Leanest { }; } - const rule = suiteRule(change.changedFiles); - if (rule) { - const run = rule.decision === "RUN" ? tests : []; + const suite = this.suiteDecision(tests, change); + if (suite) { return { command: "select", args: [framework], status: "complete", totalTests: tests.length, - selectedTests: run, - skippedTests: tests.length - run.length, - runTests: run, - skipped: rule.decision === "SKIP" ? tests : [], - reasons: Object.fromEntries(tests.map((t) => [t.identity.path, rule.reason])), - suiteReason: rule.reason, + selectedTests: suite.run, + skippedTests: suite.skip.length, + runTests: suite.run, + skipped: suite.skip, + reasons: suite.reasons, + suiteReason: suite.sharedReason, + runBreakdown: { rule: suite.run.length, judgeUnsure: 0, judgeLikely: 0 }, changedFiles: change.changedFiles, diff: change.diff, }; @@ -209,6 +222,26 @@ export class Leanest { }; } + /** + * A whole-suite rule's outcome, or null to ask the judge. A Markdown-only change still runs + * the tests a deterministic rule forces, e.g. one that imports the changed .md file. + */ + private suiteDecision(tests: TestCase[], change: GitChange) { + const rule = suiteRule(change.changedFiles); + if (!rule) return null; + const run: TestCase[] = []; + const skip: TestCase[] = []; + const reasons: Record = {}; + for (const test of tests) { + const forced = rule.decision === "SKIP" ? this.deterministicReason(test, change) : null; + reasons[test.identity.path] = forced ?? rule.reason; + (rule.decision === "RUN" || forced ? run : skip).push(test); + } + // Only when the rule decided every test does its reason explain the whole selection. + const sharedReason = rule.decision === "RUN" || run.length === 0 ? rule.reason : undefined; + return { rule, run, skip, reasons, sharedReason }; + } + private deterministicReason(test: TestCase, change: GitChange): string | null { if (change.changedFiles.includes(test.identity.path)) return "test file changed"; if (importsChangedFile(test, change.changedFiles, this.cwd)) return "imports a changed file"; diff --git a/src/types.ts b/src/types.ts index 1de0756..0b5337d 100644 --- a/src/types.ts +++ b/src/types.ts @@ -64,5 +64,6 @@ export interface PipelineResult { skipped: number; decision: "RUN" | "SKIP"; /** Set when the judge failed; every test is then selected. */ - error?: string; + error?: string /** Set when a whole-suite rule decided instead of the judge. */; + suiteReason?: string; }