From 4e86e25a72a2ed875cad392ab1989f098111866a Mon Sep 17 00:00:00 2001 From: Ivar Vong Date: Mon, 29 Jun 2026 20:35:10 -0400 Subject: [PATCH] test: deflake the two intermittently-failing 1.19 CI tests MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Two pre-existing flaky tests were intermittently failing on the loaded 1.19 CI runners, blocking unrelated PRs. Both are fixed at the root, not by loosening blindly: - OtelAdversarialFuzzTest "linear, not quadratic": the single-shot :timer.tc ratio flaked when a GC pause inflated one sample. Now takes the MIN of 5 runs per size — the fastest run is the least-contended measurement, so noise can't inflate the ratio, while a genuinely quadratic accumulation still shows ~16x (the algorithmic factor dominates the min). - MathOracleTest float sum/mean vs Polars: the tolerance was relative to the RESULT, which collapses under catastrophic cancellation — naive left-to-right summation (pyex/CPython) and Polars' pairwise/SIMD summation legitimately diverge by ~N·ε·Σ|xᵢ| when signed terms cancel (the result is near zero while the partial sums are large). New `assert_sum_close/3` scales the tolerance to the INPUT magnitude (Σ|xᵢ|), bounding the algorithmic divergence regardless of cancellation while still catching a genuinely wrong sum. Test-only; no library changes. Full suite green. Co-Authored-By: Claude Opus 4.8 (1M context) Claude-Session: https://claude.ai/code/session_019NokzcR7BiAigPgC78zpk9 --- test/pyex/math_oracle_test.exs | 25 +++++++++++++++--- test/pyex/otel_adversarial_fuzz_test.exs | 32 +++++++++++++++++------- 2 files changed, 44 insertions(+), 13 deletions(-) diff --git a/test/pyex/math_oracle_test.exs b/test/pyex/math_oracle_test.exs index f33f1c7..d7b5c3d 100644 --- a/test/pyex/math_oracle_test.exs +++ b/test/pyex/math_oracle_test.exs @@ -496,7 +496,7 @@ defmodule Pyex.MathOracleTest do check all(xs <- float_list(), max_runs: @float_runs) do opts = Oracle.float_csv_opts(xs) py = Oracle.run_with_csv(Oracle.float_preamble() <> "sum(xs)", opts) - assert_float_close(py, Oracle.polars_float(:sum, xs)) + assert_sum_close(py, Oracle.polars_float(:sum, xs), xs) end end @@ -516,7 +516,7 @@ defmodule Pyex.MathOracleTest do opts ) - assert_float_close(py, Oracle.polars_float(:sum, xs)) + assert_sum_close(py, Oracle.polars_float(:sum, xs), xs) end end @@ -532,7 +532,7 @@ defmodule Pyex.MathOracleTest do opts ) - assert_float_close(py, Oracle.polars_float(:sum, xs)) + assert_sum_close(py, Oracle.polars_float(:sum, xs), xs) end end end @@ -602,7 +602,7 @@ defmodule Pyex.MathOracleTest do check all(xs <- float_list(), max_runs: @float_runs) do opts = Oracle.float_csv_opts(xs) py = Oracle.run_with_csv(Oracle.float_preamble() <> "sum(xs) / len(xs)", opts) - assert_float_close(py, Oracle.polars_float(:mean, xs)) + assert_sum_close(py, Oracle.polars_float(:mean, xs), xs) end end end @@ -776,4 +776,21 @@ defmodule Pyex.MathOracleTest do tol = max(@float_tol, scale * @float_tol) assert diff <= tol, "expected #{b}, got #{a}, diff=#{diff}, tol=#{tol}" end + + # For comparisons of a *summation* (sum/mean) against Polars, the tolerance + # must scale to the INPUT magnitude (Σ|xᵢ|), not the result. Naive + # left-to-right summation (pyex's `sum()`, like CPython) and Polars' + # pairwise/SIMD summation legitimately diverge by ~N·ε·Σ|xᵢ| when signed + # terms cancel — the result can be near zero while the partial sums are large, + # so a result-relative tolerance collapses and the test flakes. The input + # magnitude bounds the algorithmic divergence regardless of cancellation, + # while remaining tight enough to catch a genuinely wrong sum. + defp assert_sum_close(a, b, xs) when is_number(a) and is_number(b) do + a = a * 1.0 + b = b * 1.0 + diff = abs(a - b) + magnitude = xs |> Enum.map(&abs(&1 * 1.0)) |> Enum.sum() + tol = max(@float_tol, magnitude * 1.0e-9) + assert diff <= tol, "expected #{b}, got #{a}, diff=#{diff}, tol=#{tol}" + end end diff --git a/test/pyex/otel_adversarial_fuzz_test.exs b/test/pyex/otel_adversarial_fuzz_test.exs index dcb7469..2cf1410 100644 --- a/test/pyex/otel_adversarial_fuzz_test.exs +++ b/test/pyex/otel_adversarial_fuzz_test.exs @@ -188,25 +188,39 @@ defmodule Pyex.OtelAdversarialFuzzTest do end test "attribute/event accumulation is linear, not quadratic" do - time = fn n -> - src = """ + src = fn n -> + """ from opentelemetry import trace tracer = trace.get_tracer("t") with tracer.start_as_current_span("s") as span: for i in range(#{n}): span.add_event("e") """ + end - {us, _} = - :timer.tc(fn -> Pyex.run(src, limits: [max_steps: 100_000_000, timeout: 30_000]) end) - - us / 1000 + # Min of several runs, not a single shot: the fastest run is the + # least-contended measurement, so a GC pause or scheduler hiccup on a + # loaded CI box can't inflate the ratio (the old single-sample timing + # flaked). A genuinely quadratic accumulation would still show ~16x — far + # above the threshold — because the algorithmic factor dominates the min. + min_ms = fn n -> + 1..5 + |> Enum.map(fn _ -> + {us, _} = + :timer.tc(fn -> + Pyex.run(src.(n), limits: [max_steps: 100_000_000, timeout: 30_000]) + end) + + us / 1000 + end) + |> Enum.min() end - _warm = time.(4_000) - ratio = time.(40_000) / time.(10_000) + _warm = min_ms.(4_000) + ratio = min_ms.(40_000) / min_ms.(10_000) # 4x the events should cost ~4x; quadratic would be ~16x. - assert ratio < 8.0, "add_event scaled #{Float.round(ratio, 2)}x for 4x input (expected ~4x)" + assert ratio < 8.0, + "add_event scaled #{Float.round(ratio, 2)}x (min-of-5) for 4x input (expected ~4x)" end end