From 336e4960fec462fc2dc934f349bd3d56fda57ebc Mon Sep 17 00:00:00 2001 From: feiiiiii5 Date: Mon, 28 Sep 2026 17:18:47 +0800 Subject: [PATCH] fix(ragas): resolve the judge model through the library default _get_model is what every RAGAS scorer calls to resolve its judge model. It fell back to DEFAULT_RAGAS_MODEL, a hardcoded "gpt-5-nano", when the caller passed no model and had not called init(default_model=...): init() # resets to the library defaults get_default_model() # 'gpt-5-mini' _get_model(None) # 'gpt-5-nano' <- not the library default So a RAGAS scorer silently judged with a different model than every other scorer in the library. The TypeScript implementation resolves through getDefaultModel() and never had a separate fallback, so the same evaluation produced different scores on the two implementations. The constant was already labelled deprecated, by the commit that added init(default_model=...) (#161): "This was previously 'gpt-5-mini' but now defaults to the configured model." Only the first half of that change landed -- the scorers stopped taking model=DEFAULT_RAGAS_MODEL as a default argument, but _get_model still returned the constant. _get_model now delegates to get_default_model(), which also removes the manual _default_model_var lookup. The constant had no other reader, so it and that import go with it. Test: py/autoevals/test_ragas_default_model.py fails on main with "assert 'gpt-5-nano' == 'gpt-5-mini'" and passes here. Excluding the LLM-backed files, the suite goes from 25 to 28 passed; the 4 failures and 8 errors are unchanged and are all missing OPENAI_API_KEY or a missing litellm. --- py/autoevals/ragas.py | 19 +++------------- py/autoevals/test_ragas_default_model.py | 29 ++++++++++++++++++++++++ 2 files changed, 32 insertions(+), 16 deletions(-) create mode 100644 py/autoevals/test_ragas_default_model.py diff --git a/py/autoevals/ragas.py b/py/autoevals/ragas.py index 4726159..04430f5 100644 --- a/py/autoevals/ragas.py +++ b/py/autoevals/ragas.py @@ -115,7 +115,6 @@ def context_relevancy_scorer(output, expected, input, metadata): from .llm import OpenAILLMScorer from .oai import ( Client, - _default_model_var, arun_cached_request, get_default_embedding_model, get_default_model, @@ -130,28 +129,16 @@ def check_required(name, **kwargs): raise ValueError(f"{name} requires {key} value") -# Deprecated: Use init(default_model="...") to configure the default model instead. -# This was previously "gpt-5-nano" but now defaults to the configured model. -DEFAULT_RAGAS_MODEL = "gpt-5-nano" - - def _get_model(model: str | None) -> str: """Get the model to use, respecting init(default_model=...) configuration. - Falls back to DEFAULT_RAGAS_MODEL if no model is specified and no custom - default has been configured. + Falls back to the library-wide default when no model is specified and no + custom default has been configured. """ if model is not None: return model - # Check if user configured a custom default via init(default_model=...) - # If they did (even if it's "gpt-5-mini"), respect it for consistency - configured_default = _default_model_var.get(None) - if configured_default is not None: - return configured_default - - # Fall back to RAGAS-specific default when user hasn't configured anything - return DEFAULT_RAGAS_MODEL + return get_default_model() ENTITY_PROMPT = """Given a text, extract unique entities without repetition. Ensure you consider different forms or mentions of the same entity as a single entity. diff --git a/py/autoevals/test_ragas_default_model.py b/py/autoevals/test_ragas_default_model.py new file mode 100644 index 0000000..c292c1a --- /dev/null +++ b/py/autoevals/test_ragas_default_model.py @@ -0,0 +1,29 @@ +"""The RAGAS scorers must not fall back to their own hardcoded model. + +`_get_model` is what every RAGAS scorer calls to resolve its judge model. +Before this change it returned `DEFAULT_RAGAS_MODEL` ("gpt-5-nano") when the +caller passed no model and had not called `init(default_model=...)`, so a +RAGAS scorer silently used a different judge than every other scorer in the +library -- and than the TypeScript implementation, which resolves through +`getDefaultModel()`. The comment above the constant already said the +hardcoded fallback was deprecated. +""" + +from autoevals import init +from autoevals.oai import get_default_model +from autoevals.ragas import _get_model + + +def test_ragas_default_matches_library_default(): + init() + assert _get_model(None) == get_default_model() + + +def test_ragas_respects_explicit_model(): + init() + assert _get_model("gpt-4o") == "gpt-4o" + + +def test_ragas_respects_configured_default(): + init(default_model="gpt-4-turbo") + assert _get_model(None) == "gpt-4-turbo"