Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
25 changes: 21 additions & 4 deletions test/pyex/math_oracle_test.exs
Original file line number Diff line number Diff line change
Expand Up @@ -496,7 +496,7 @@ defmodule Pyex.MathOracleTest do
check all(xs <- float_list(), max_runs: @float_runs) do
opts = Oracle.float_csv_opts(xs)
py = Oracle.run_with_csv(Oracle.float_preamble() <> "sum(xs)", opts)
assert_float_close(py, Oracle.polars_float(:sum, xs))
assert_sum_close(py, Oracle.polars_float(:sum, xs), xs)
end
end

Expand All @@ -516,7 +516,7 @@ defmodule Pyex.MathOracleTest do
opts
)

assert_float_close(py, Oracle.polars_float(:sum, xs))
assert_sum_close(py, Oracle.polars_float(:sum, xs), xs)
end
end

Expand All @@ -532,7 +532,7 @@ defmodule Pyex.MathOracleTest do
opts
)

assert_float_close(py, Oracle.polars_float(:sum, xs))
assert_sum_close(py, Oracle.polars_float(:sum, xs), xs)
end
end
end
Expand Down Expand Up @@ -602,7 +602,7 @@ defmodule Pyex.MathOracleTest do
check all(xs <- float_list(), max_runs: @float_runs) do
opts = Oracle.float_csv_opts(xs)
py = Oracle.run_with_csv(Oracle.float_preamble() <> "sum(xs) / len(xs)", opts)
assert_float_close(py, Oracle.polars_float(:mean, xs))
assert_sum_close(py, Oracle.polars_float(:mean, xs), xs)
end
end
end
Expand Down Expand Up @@ -776,4 +776,21 @@ defmodule Pyex.MathOracleTest do
tol = max(@float_tol, scale * @float_tol)
assert diff <= tol, "expected #{b}, got #{a}, diff=#{diff}, tol=#{tol}"
end

# For comparisons of a *summation* (sum/mean) against Polars, the tolerance
# must scale to the INPUT magnitude (Σ|xᵢ|), not the result. Naive
# left-to-right summation (pyex's `sum()`, like CPython) and Polars'
# pairwise/SIMD summation legitimately diverge by ~N·ε·Σ|xᵢ| when signed
# terms cancel — the result can be near zero while the partial sums are large,
# so a result-relative tolerance collapses and the test flakes. The input
# magnitude bounds the algorithmic divergence regardless of cancellation,
# while remaining tight enough to catch a genuinely wrong sum.
defp assert_sum_close(a, b, xs) when is_number(a) and is_number(b) do
a = a * 1.0
b = b * 1.0
diff = abs(a - b)
magnitude = xs |> Enum.map(&abs(&1 * 1.0)) |> Enum.sum()
tol = max(@float_tol, magnitude * 1.0e-9)
assert diff <= tol, "expected #{b}, got #{a}, diff=#{diff}, tol=#{tol}"
end
end
32 changes: 23 additions & 9 deletions test/pyex/otel_adversarial_fuzz_test.exs
Original file line number Diff line number Diff line change
Expand Up @@ -188,25 +188,39 @@ defmodule Pyex.OtelAdversarialFuzzTest do
end

test "attribute/event accumulation is linear, not quadratic" do
time = fn n ->
src = """
src = fn n ->
"""
from opentelemetry import trace
tracer = trace.get_tracer("t")
with tracer.start_as_current_span("s") as span:
for i in range(#{n}):
span.add_event("e")
"""
end

{us, _} =
:timer.tc(fn -> Pyex.run(src, limits: [max_steps: 100_000_000, timeout: 30_000]) end)

us / 1000
# Min of several runs, not a single shot: the fastest run is the
# least-contended measurement, so a GC pause or scheduler hiccup on a
# loaded CI box can't inflate the ratio (the old single-sample timing
# flaked). A genuinely quadratic accumulation would still show ~16x — far
# above the threshold — because the algorithmic factor dominates the min.
min_ms = fn n ->
1..5
|> Enum.map(fn _ ->
{us, _} =
:timer.tc(fn ->
Pyex.run(src.(n), limits: [max_steps: 100_000_000, timeout: 30_000])
end)

us / 1000
end)
|> Enum.min()
end

_warm = time.(4_000)
ratio = time.(40_000) / time.(10_000)
_warm = min_ms.(4_000)
ratio = min_ms.(40_000) / min_ms.(10_000)
# 4x the events should cost ~4x; quadratic would be ~16x.
assert ratio < 8.0, "add_event scaled #{Float.round(ratio, 2)}x for 4x input (expected ~4x)"
assert ratio < 8.0,
"add_event scaled #{Float.round(ratio, 2)}x (min-of-5) for 4x input (expected ~4x)"
end
end

Expand Down
Loading