From 1180523adf84b17374141d7fb6b468c28665003a Mon Sep 17 00:00:00 2001 From: Ivar Vong Date: Mon, 29 Jun 2026 19:59:03 -0400 Subject: [PATCH] feat(api): carry the capability ledger + footprint on a failed run's error MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit A failed `Pyex.run` now returns a `%Pyex.Error{}` that carries `runtime_spans` (the capability ledger of what the program touched before it broke) and `footprint` (its resource usage). Auditing, policy-gating, and billing a *failed* run no longer requires attaching a telemetry handler — it's a value on the return. Both are empty/nil for errors raised before execution (syntax), which is correct: nothing ran, nothing was touched. The error path already computed both for the [:pyex, :run, :exception] event; this just stops throwing them away. On success the same is read off `ctx` via `Pyex.Turn`/`Pyex.Ctx.runtime_spans`, so the ledger is now first-class on every outcome. The library exposes the value; callers compose audit/policy/billing. Simplifies examples/sandbox_server.exs accordingly: it reads the ledger straight off the error instead of a per-worker telemetry-capture handler. Pre-release; additive struct fields, non-breaking for `{:error, %Pyex.Error{}}` matchers. Full suite + Dialyzer green. Co-Authored-By: Claude Opus 4.8 (1M context) Claude-Session: https://claude.ai/code/session_019NokzcR7BiAigPgC78zpk9 --- examples/sandbox_server.exs | 73 ++++++++++++++++++------------------- lib/pyex.ex | 29 ++++++++++++--- lib/pyex/error.ex | 11 +++++- test/pyex/error_test.exs | 19 ++++++++++ 4 files changed, 87 insertions(+), 45 deletions(-) diff --git a/examples/sandbox_server.exs b/examples/sandbox_server.exs index 5a15212..adf0434 100644 --- a/examples/sandbox_server.exs +++ b/examples/sandbox_server.exs @@ -89,19 +89,31 @@ defmodule SandboxServer do # Fresh, isolated in-memory store per request. run_opts = [limits: @limits, storage: Pyex.Storage.Memory.new()] + # The footprint + capability ledger come straight off the return value + # on both paths — `ctx` on success, the `%Pyex.Error{}` on failure — so + # "what did it touch before it crashed?" needs no telemetry plumbing. body = case Pyex.run(source, run_opts) do {:ok, value, ctx} -> - with_telemetry(%{verdict: "ok", stdout: Pyex.output(ctx), value: inspect(value)}) + %{ + verdict: "ok", + stdout: Pyex.output(ctx), + value: inspect(value), + usage: usage(Pyex.Turn.footprint(ctx)), + trace: Pyex.Turn.render(ctx) + } - {:error, %Pyex.Error{kind: :timeout, message: m}} -> - with_telemetry(%{verdict: "timeout", detail: m}) + {:error, %Pyex.Error{kind: :timeout} = e} -> + audited(%{verdict: "timeout", detail: e.message}, e) {:error, %Pyex.Error{} = e} -> - with_telemetry(%{ - verdict: "error", - error: %{type: e.exception_type, kind: e.kind, message: e.message, line: e.line} - }) + audited( + %{ + verdict: "error", + error: %{type: e.exception_type, kind: e.kind, message: e.message, line: e.line} + }, + e + ) end send(parent, {:done, self(), body}) @@ -132,39 +144,26 @@ defmodule SandboxServer do end end - # Folds the run's footprint + capability ledger (captured from Pyex's telemetry - # in THIS worker process) into the verdict body. Present whenever the run - # produced a result — including failures. - defp with_telemetry(body) do - case Process.get(:pyex_run) do - {footprint, metadata} -> - spans = Map.get(metadata, :runtime_spans, []) - - body - |> Map.put(:usage, %{ - steps: footprint[:steps], - compute_ms: footprint[:compute], - duration_ms: footprint[:duration_ms], - memory_bytes: footprint[:memory_bytes], - output_bytes: footprint[:output_bytes] - }) - |> Map.put(:trace, Pyex.SpanTree.render(spans, title: "runtime · scope=pyex")) + # Folds a failed run's footprint + capability ledger (carried on the error) + # into the verdict body. Both are empty/nil for errors raised before execution. + defp audited(body, %Pyex.Error{footprint: nil}), do: body - _ -> - body - end + defp audited(body, %Pyex.Error{footprint: footprint, runtime_spans: spans}) do + body + |> Map.put(:usage, usage(footprint)) + |> Map.put(:trace, Pyex.SpanTree.render(spans, title: "runtime · scope=pyex")) end -end -# One telemetry handler captures each run's footprint + capability ledger into -# the emitting worker's process dictionary (handlers run in the caller's -# process), for `with_telemetry/1` to read back. Covers success and failure. -:telemetry.attach_many( - "sandbox-capture", - [[:pyex, :run, :stop], [:pyex, :run, :exception]], - fn _event, measurements, metadata, _ -> Process.put(:pyex_run, {measurements, metadata}) end, - nil -) + defp usage(footprint) do + %{ + steps: footprint[:steps], + compute_ms: footprint[:compute], + duration_ms: footprint[:duration_ms], + memory_bytes: footprint[:memory_bytes], + output_bytes: footprint[:output_bytes] + } + end +end port = String.to_integer(System.get_env("PORT", "4599")) {:ok, _} = Bandit.start_link(plug: SandboxServer, port: port) diff --git a/lib/pyex.ex b/lib/pyex.ex index 12e6e49..50b3ed1 100644 --- a/lib/pyex.ex +++ b/lib/pyex.ex @@ -72,7 +72,14 @@ defmodule Pyex do The optional second argument can be a `Pyex.Ctx` struct or a keyword list of options (forwarded to `Pyex.Ctx.new/1`). - Returns `{:ok, value, ctx}` on success, or `{:error, reason}`. + Returns `{:ok, value, ctx}` on success, or `{:error, %Pyex.Error{}}`. + + On failure the error carries the run's `runtime_spans` (the capability + ledger of what the program touched before it broke) and `footprint` (its + resource usage), so a failed run is auditable from the return value alone — + no telemetry handler required. Both are empty/`nil` for errors raised before + execution (e.g. syntax). On success the same is read from `ctx` via + `Pyex.Turn` and `Pyex.Ctx.runtime_spans/1`. ## Options (when passing keyword list) @@ -218,15 +225,25 @@ defmodule Pyex do 1000.0 final_ctx = %{final_ctx | duration_ms: duration_ms} - error = Error.from_message(msg) # A failed turn still surfaces what it touched: the capability ledger - # of everything done before the error rides on the exception event, - # so a host/agent can audit a crash even though `run` returns no ctx. + # and resource footprint of everything done before the error are + # carried on the returned error (and the exception event), so a host + # or agent can audit a crash from the return value alone — no + # telemetry handler required. + footprint = Pyex.Turn.footprint(final_ctx) + runtime_spans = Ctx.runtime_spans(final_ctx) + + error = %{ + Error.from_message(msg) + | footprint: footprint, + runtime_spans: runtime_spans + } + :telemetry.execute( [:pyex, :run, :exception], - Pyex.Turn.footprint(final_ctx), - %{error: error, runtime_spans: Ctx.runtime_spans(final_ctx)} + footprint, + %{error: error, runtime_spans: runtime_spans} ) {:error, error} diff --git a/lib/pyex/error.ex b/lib/pyex/error.ex index d1b9346..416a8c8 100644 --- a/lib/pyex/error.ex +++ b/lib/pyex/error.ex @@ -45,14 +45,21 @@ defmodule Pyex.Error do limit: limit_type(), message: String.t(), line: pos_integer() | nil, - exception_type: String.t() | nil + exception_type: String.t() | nil, + runtime_spans: [map()], + footprint: map() | nil } defstruct kind: :internal, limit: nil, message: "", line: nil, - exception_type: nil + exception_type: nil, + # The capability ledger and resource footprint at the point of + # failure — what the program touched and how much it spent before it + # broke. Empty/nil for errors raised before execution (e.g. syntax). + runtime_spans: [], + footprint: nil @doc """ Classifies a raw error string into a structured error. diff --git a/test/pyex/error_test.exs b/test/pyex/error_test.exs index 6acd09a..3aca2f0 100644 --- a/test/pyex/error_test.exs +++ b/test/pyex/error_test.exs @@ -226,4 +226,23 @@ defmodule Pyex.ErrorTest do assert err.kind == :timeout end end + + describe "the error carries the capability ledger and footprint of a failed run" do + test "a failed run's error carries the span ledger of what it touched first" do + src = "import store\nstore.set('k', 1)\nstore.get('k')\n1 / 0" + {:error, %Error{} = err} = Pyex.run(src, storage: Pyex.Storage.Memory.new()) + + # Auditable from the return value alone — no telemetry handler needed. + assert Enum.map(err.runtime_spans, & &1.name) == ["db.set", "db.get"] + assert err.footprint[:steps] > 0 + end + + test "an error raised before execution carries an empty ledger and no footprint" do + {:error, %Error{} = err} = Pyex.run("def (:") + + assert err.kind == :syntax + assert err.runtime_spans == [] + assert err.footprint == nil + end + end end