From f64ecfaf97200deb519c75ae8a6b5d9eb51aef17 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Mauricio=20S=C3=A1nchez?= Date: Fri, 2 Oct 2026 00:24:42 -0500 Subject: [PATCH] feat: /cost-report shows what Opus sessions would have cost on Sonnet MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit A pilot's report showed every session on Opus although the project's model setting said sonnet: the desktop app's model picker sets each session's model. The report now re-prices each Opus or Fable session's tokens at Sonnet's prices — an "on sonnet" column, and a model line with the total and the difference — and the skill ties it to the cost model's rule. Cache reads cost the same on both, so the report notes that context size matters as much in long sessions. --json adds cost_on_sonnet. aplyca-framework 0.2.6. Co-Authored-By: Claude Opus 5.5 --- CHANGELOG.md | 10 ++++++++ README.md | 2 +- evals/static/test-plugin.sh | 4 +++ .../.claude-plugin/plugin.json | 2 +- plugins/aplyca-framework/README.md | 2 +- .../skills/cost-report/SKILL.md | 13 ++++++++-- .../skills/cost-report/session_cost.py | 25 +++++++++++++++---- 7 files changed, 48 insertions(+), 10 deletions(-) diff --git a/CHANGELOG.md b/CHANGELOG.md index f7d4ef8..cdfdb3e 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -11,6 +11,16 @@ For each entry, **Upgrade impact** classifies the change against the [three-buck ## Unreleased +### `/cost-report` shows what Opus sessions would have cost on Sonnet + +A pilot's report showed every session on Opus, though the project's `"model"` setting said `sonnet`: +the desktop app's model picker sets each session's model. The report now prices each Opus or Fable +session's tokens at Sonnet's prices too — an `on sonnet` column and a model line with the total and +the difference — and the skill ties it to the cost model's rule: Sonnet for work with a clear spec +and a way to check it. Cache reads cost the same on both, so the difference is in output and cache +writes; the report says so. `--json` adds `cost_on_sonnet` (`aplyca-framework` 0.2.6). +**Upgrade impact:** framework-internal; update the plugin. + ### Claude Code's own worktrees are workers too (parallel-agents) ([0015](docs/decisions/0015-tool-worktrees-are-workers.md), amending diff --git a/README.md b/README.md index 6c855dd..f1f13d0 100644 --- a/README.md +++ b/README.md @@ -30,7 +30,7 @@ Most of what's here was proven in real client projects first — some built on t - **9 engineering standards** — code quality (including "write almost no comments"), testing, security, git workflow, plus customizable architecture, UI/UX, deployment, performance, observability. - **Process records** — a constitution that gates every spec and review, Process Decision Records for how the team works, ADRs for the application, and on-demand code-level reference pages. - **Optional modules** — `github` (PR template with the lane, traceability, and constitution gates; issue forms, secret scan, base-branch policy), `git-hooks` (tool-agnostic `pre-push`), `clickup` (ClickUp's MCP server, so `/triage` reads tasks directly; a read-only allowlist, and each developer signs in with OAuth), `parallel-agents` (one worktree, branch, and session per task — plus its own port when the app runs locally; the main checkout only dispatches). ([Modules](modules/README.md)) -- **Installer plugin** — `/adopt` and `/upgrade` for Claude Code, plus `/cost-report`: what each agent session on a project cost — calls, context, tokens, estimated cost — with flags for long context, cache-expiring pauses, and spec-heavy small changes. ([Plugin](plugins/aplyca-framework/README.md)) +- **Installer plugin** — `/adopt` and `/upgrade` for Claude Code, plus `/cost-report`: what each agent session on a project cost — calls, context, tokens, estimated cost, and what Opus sessions would have cost on Sonnet — with flags for long context, cache-expiring pauses, and spec-heavy small changes. ([Plugin](plugins/aplyca-framework/README.md)) - **Evals** — structural checks plus functional tests of the hooks, module scripts, and plugin, run in CI on every pull request at zero token cost; routing evals that run `/triage` in real Claude Code sessions on Sonnet and Opus, with graded reports. ([Evals](evals/README.md) · [latest report](evals/dynamic/reports/2026-10-01-triage-routing.md)) - **Onboarding, worked examples, scenario playbooks** — see [Team onboarding](#team-onboarding). diff --git a/evals/static/test-plugin.sh b/evals/static/test-plugin.sh index 0876b0f..0f0dc2a 100755 --- a/evals/static/test-plugin.sh +++ b/evals/static/test-plugin.sh @@ -80,6 +80,10 @@ cost=$(printf '%s' "$json" | python3 -c 'import json,sys; s={x["title"]: x for x check "cost-report: prices Opus calls at list prices (\$0.25 for the quick fix, got \$$cost)" "[ '$cost' = '0.25' ]" sonnet=$(printf '%s' "$json" | python3 -c 'import json,sys; s={x["title"]: x for x in json.load(sys.stdin)}; print(s["Spec heavy"]["model"])') check "cost-report: recognizes the model tier (got $sonnet)" "[ '$sonnet' = 'sonnet' ]" +on_sonnet=$(printf '%s' "$json" | python3 -c 'import json,sys; s={x["title"]: x for x in json.load(sys.stdin)}; print(s["Quick fix"]["cost_on_sonnet"], s["Spec heavy"]["cost_on_sonnet"] == s["Spec heavy"]["cost"])') +# The same tokens at Sonnet's prices: 10×(10×2 + 500×10 + 50,000×0.20 + 1,000×2.5)/1e6 = 0.18 +check "cost-report: re-prices an Opus session at Sonnet's prices (\$0.18; a Sonnet session unchanged — got $on_sonnet)" "[ '$on_sonnet' = '0.18 True' ]" +check "cost-report: shows the Sonnet estimate for Opus sessions only, and totals it" "echo \"\$out\" | grep 'Quick fix' | grep -q '≈0.18' && echo \"\$out\" | grep 'Spec heavy' | grep -q ' - ' && echo \"\$out\" | grep -q '^Model: 3 sessions ran above Sonnet'" missing=$(python3 "$REPORT" "$WORK/elsewhere" --projects-dir "$WORK/projects" 2>&1); code=$? check "cost-report: a project without transcripts says so and exits non-zero" "[ $code -ne 0 ] && echo \"\$missing\" | grep -q 'No Claude Code transcripts'" diff --git a/plugins/aplyca-framework/.claude-plugin/plugin.json b/plugins/aplyca-framework/.claude-plugin/plugin.json index 098de60..5c0c4e8 100644 --- a/plugins/aplyca-framework/.claude-plugin/plugin.json +++ b/plugins/aplyca-framework/.claude-plugin/plugin.json @@ -1,7 +1,7 @@ { "name": "aplyca-framework", "description": "Installer and upgrader for the Aplyca Agentic Development Framework. /adopt bootstraps a repository for agentic development — skeleton, optional modules (GitHub harness, git hooks, parallel-agent worktrees), guardrail hooks, verified facts; /upgrade syncs an adopted repository to a newer skeleton version; /cost-report shows what agent sessions on a project cost, from local transcripts. The framework itself ships as committed files in each repo (AGENTS.md standard, multi-tool); this plugin is the tooling that installs and maintains them.", - "version": "0.2.5", + "version": "0.2.6", "author": { "name": "Aplyca", "email": "dev@aplyca.com" diff --git a/plugins/aplyca-framework/README.md b/plugins/aplyca-framework/README.md index 099f518..14c95bb 100644 --- a/plugins/aplyca-framework/README.md +++ b/plugins/aplyca-framework/README.md @@ -93,7 +93,7 @@ claude plugin marketplace remove aplyca --scope user |---|---| | `/adopt` | Bootstrap a repo: inspect it (stack, commands, branching model, tracker, Git host), copy the skeleton and the [optional modules](../../modules/README.md) you choose, fill placeholders from verified repo facts, configure the guardrail hooks, record the adoption as PDR-0001, stamp the baseline SHA and modules, verify (settings schema, hook smoke tests, the `@AGENTS.md` import), and prepare a draft adoption PR. On an already-adopted repo it adds modules. Automates [docs/SETUP.md](../../docs/SETUP.md). | | `/upgrade` | Sync an adopted repo — skeleton and installed modules — to a newer version via the three-bucket taxonomy, OLD_SHA → NEW_SHA discipline, and the changelog's migration steps, and offer the modules it doesn't have yet (installed in the same pull request when chosen). Automates [docs/UPGRADING.md](../../docs/UPGRADING.md). | -| `/cost-report` | What agent sessions on a project cost — calls, active time, context size, tokens, estimated cost — from Claude Code's local transcripts, with the expensive patterns flagged (long context, pauses past the cache lifetime, browser loops, spec-heavy small changes). Read-only; nothing leaves the machine. See the skeleton's [COST-MODEL.md](../../skeleton/docs/COST-MODEL.md). | +| `/cost-report` | What agent sessions on a project cost — calls, active time, context size, tokens, estimated cost — from Claude Code's local transcripts, with what Opus sessions would have cost on Sonnet and the expensive patterns flagged (long context, pauses past the cache lifetime, browser loops, spec-heavy small changes). Read-only; nothing leaves the machine. See the skeleton's [COST-MODEL.md](../../skeleton/docs/COST-MODEL.md). | `/adopt` and `/upgrade` work branch-and-PR only — they never commit to a default branch, and never push without explicit approval. `/cost-report` only reads. diff --git a/plugins/aplyca-framework/skills/cost-report/SKILL.md b/plugins/aplyca-framework/skills/cost-report/SKILL.md index 711682a..438854d 100644 --- a/plugins/aplyca-framework/skills/cost-report/SKILL.md +++ b/plugins/aplyca-framework/skills/cost-report/SKILL.md @@ -1,6 +1,6 @@ --- name: cost-report -description: Report what Claude Code sessions on a project cost — calls, active time, context size, tokens, and estimated cost per session from Claude Code's local transcripts — and flag the patterns that make sessions expensive (long context, pauses past the cache lifetime, browser loops, spec-heavy small changes). Use when asked how much agent work costs, why a session was expensive, or whether a change to the lanes or habits paid off. +description: Report what Claude Code sessions on a project cost — calls, active time, context size, tokens, and estimated cost per session from Claude Code's local transcripts, with what Opus sessions would have cost on Sonnet — and flag the patterns that make sessions expensive (long context, pauses past the cache lifetime, browser loops, spec-heavy small changes). Use when asked how much agent work costs, why a session was expensive, or whether a change to the lanes or habits paid off. argument-hint: "[project path — default: this repository] [--days N] [--siblings]" --- @@ -22,10 +22,16 @@ stays local: the script reads `~/.claude/projects/` and prints a report; nothing module); worktrees under `.claude/worktrees/` are always included. - `--days 0` reads every session; `--top N` lists more; `--json` gives every session's numbers. -2. **Present the result:** totals, the most expensive sessions, the size bands, and the flags. +2. **Present the result:** totals, the most expensive sessions, the size bands, the model line, and + the flags. 3. **Explain the drivers** with `docs/COST-MODEL.md` (in the project, or the framework's skeleton): - **Calls × context** — cost grows with the number of steps and with how much each step re-reads. + - **The model** — the `on sonnet` column re-prices an Opus or Fable session's tokens at Sonnet's + prices. Where the session's work had a clear spec and a way to check it — the fast and careful + lanes, bug fixes, reviews, an approved plan — Sonnet fits: pick it in the desktop app's model + picker or with `/model`, since the picker overrides the project's `"model"` setting. Cache reads + cost the same on both, so in long sessions the context is as large a lever. - **`long-context`** — one task per session, `/clear` between tasks; a 1M-context model lets routine sessions grow far past what they need. - **`pauses`** — each wait longer than the cache lifetime wrote the whole context again; batch @@ -40,6 +46,9 @@ stays local: the script reads `~/.claude/projects/` and prints a report; nothing - **Estimates, not invoices.** Prices are list prices in the script; subscriptions, discounts, and provider pricing differ. Say so. +- **The Sonnet column is a price comparison, not a prediction.** It re-prices the same tokens; a + session on Sonnet could take more or fewer turns. It can't say whether a session's work needed + Opus — read the session's title and lane for that. - **Session titles can contain client or task details.** Keep the report in the conversation; don't paste titles into commits, pull requests, or shared documents without the developer's say-so. - Read-only: the skill never edits transcripts or settings. diff --git a/plugins/aplyca-framework/skills/cost-report/session_cost.py b/plugins/aplyca-framework/skills/cost-report/session_cost.py index 0189906..d5f89dc 100755 --- a/plugins/aplyca-framework/skills/cost-report/session_cost.py +++ b/plugins/aplyca-framework/skills/cost-report/session_cost.py @@ -2,8 +2,9 @@ """Summarize what Claude Code sessions on a project cost, from Claude Code's local transcripts. Reads ~/.claude/projects//*.jsonl (nothing leaves the machine) and reports, per session: -API calls, active time, context size, tokens, an estimated cost at list prices, and flags for the -patterns that make sessions expensive. Python 3 standard library only. +API calls, active time, context size, tokens, an estimated cost at list prices — and, for sessions +on Opus or Fable, what the same tokens cost on Sonnet — and flags for the patterns that make sessions +expensive. Python 3 standard library only. Usage: session_cost.py [project-path] [--days 30] [--top 15] [--siblings] [--json] [--projects-dir ~/.claude/projects] @@ -97,9 +98,11 @@ def analyze(path): totals = collections.Counter() contexts, models = [], collections.Counter() - cost = 0.0 + cost = on_sonnet = 0.0 for model, usage in calls.values(): price = PRICES[tier(model)] + # The same tokens at Sonnet's prices — for calls above Sonnet only; cheaper calls keep theirs. + sonnet = PRICES["sonnet"] if tier(model) in ("opus", "fable") else price fresh = usage.get("input_tokens", 0) read = usage.get("cache_read_input_tokens", 0) write = usage.get("cache_creation_input_tokens", 0) @@ -108,6 +111,7 @@ def analyze(path): contexts.append(fresh + read + write) models[tier(model)] += 1 cost += (fresh * price[0] + out * price[1] + read * price[2] + write * price[3]) / 1e6 + on_sonnet += (fresh * sonnet[0] + out * sonnet[1] + read * sonnet[2] + write * sonnet[3]) / 1e6 times.sort() active = sum(min((b - a).total_seconds(), IDLE_CAP) for a, b in zip(times, times[1:])) @@ -138,6 +142,7 @@ def analyze(path): "cache_read_m": round(totals["read"] / 1e6, 2), "cache_write_k": round(totals["write"] / 1000, 1), "cost": round(cost, 2), + "cost_on_sonnet": round(on_sonnet, 2), "share": { "cache_read": totals["read"], "cache_write": totals["write"], "output": totals["output"], "input": totals["input"], @@ -199,10 +204,11 @@ def main(): print(f"{len(sessions)} sessions · {calls} calls · ≈ ${total:.2f} at list prices " f"(median ${statistics.median(s['cost'] / s['calls'] for s in sessions):.3f} per call)") print(f"Folders: {', '.join(os.path.basename(d) for d in dirs)}\n") - print(f"{'date':<11}{'calls':>6}{'active':>8}{'ctx avg':>9}{'cost':>9} {'model':<12}{'flags':<28}title") + print(f"{'date':<11}{'calls':>6}{'active':>8}{'ctx avg':>9}{'cost':>9}{'on sonnet':>11} {'model':<12}{'flags':<28}title") for s in sessions[: args.top]: + sonnet = f"≈{s['cost_on_sonnet']:.2f}" if s["cost_on_sonnet"] < s["cost"] else "-" print(f"{s['date']:<11}{s['calls']:>6}{s['active_min']:>7.0f}m{s['context_mean_k']:>8.0f}k" - f"{s['cost']:>9.2f} {s['model']:<12}{' '.join(s['flags']):<28}{s['title']}") + f"{s['cost']:>9.2f}{sonnet:>11} {s['model']:<12}{' '.join(s['flags']):<28}{s['title']}") if len(sessions) > args.top: print(f"… {len(sessions) - args.top} cheaper sessions not listed (--top)") print() @@ -212,6 +218,15 @@ def main(): label = f"{low}+" if high == 10**9 else f"{low}–{high - 1}" print(f"{label:>7} calls: {len(band):>3} sessions, median ${statistics.median(s['cost'] for s in band):.2f}, " f"median {statistics.median(s['active_min'] for s in band):.0f} min active") + above = [s for s in sessions if s["cost_on_sonnet"] < s["cost"]] + if above: + spent = sum(s["cost"] for s in above) + saved = spent - sum(s["cost_on_sonnet"] for s in above) + print(f"\nModel: {len(above)} session{'s' * (len(above) != 1)} ran above Sonnet (≈ ${spent:.2f}); the same tokens on " + f"Sonnet ≈ ${spent - saved:.2f}, {saved / spent:.0%} less.") + print(" Sonnet suits work with a clear spec and a way to check it — the fast and careful lanes, bug") + print(" fixes, reviews, an approved plan; Opus, the full lane's spec and plan (docs/COST-MODEL.md).") + print(" Cache reads cost the same on both: in long sessions, context size matters as much as the model.") flagged = collections.Counter(f.split(":")[0] for s in sessions for f in s["flags"]) if flagged: print("\nFlags: " + ", ".join(f"{name} ×{n}" for name, n in flagged.most_common()))