Skip to content
Open
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
19 changes: 3 additions & 16 deletions py/autoevals/ragas.py
Original file line number Diff line number Diff line change
Expand Up @@ -115,7 +115,6 @@ def context_relevancy_scorer(output, expected, input, metadata):
from .llm import OpenAILLMScorer
from .oai import (
Client,
_default_model_var,
arun_cached_request,
get_default_embedding_model,
get_default_model,
Expand All @@ -130,28 +129,16 @@ def check_required(name, **kwargs):
raise ValueError(f"{name} requires {key} value")


# Deprecated: Use init(default_model="...") to configure the default model instead.
# This was previously "gpt-5-nano" but now defaults to the configured model.
DEFAULT_RAGAS_MODEL = "gpt-5-nano"


def _get_model(model: str | None) -> str:
"""Get the model to use, respecting init(default_model=...) configuration.

Falls back to DEFAULT_RAGAS_MODEL if no model is specified and no custom
default has been configured.
Falls back to the library-wide default when no model is specified and no
custom default has been configured.
"""
if model is not None:
return model

# Check if user configured a custom default via init(default_model=...)
# If they did (even if it's "gpt-5-mini"), respect it for consistency
configured_default = _default_model_var.get(None)
if configured_default is not None:
return configured_default

# Fall back to RAGAS-specific default when user hasn't configured anything
return DEFAULT_RAGAS_MODEL
return get_default_model()


ENTITY_PROMPT = """Given a text, extract unique entities without repetition. Ensure you consider different forms or mentions of the same entity as a single entity.
Expand Down
29 changes: 29 additions & 0 deletions py/autoevals/test_ragas_default_model.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,29 @@
"""The RAGAS scorers must not fall back to their own hardcoded model.

`_get_model` is what every RAGAS scorer calls to resolve its judge model.
Before this change it returned `DEFAULT_RAGAS_MODEL` ("gpt-5-nano") when the
caller passed no model and had not called `init(default_model=...)`, so a
RAGAS scorer silently used a different judge than every other scorer in the
library -- and than the TypeScript implementation, which resolves through
`getDefaultModel()`. The comment above the constant already said the
hardcoded fallback was deprecated.
"""

from autoevals import init
from autoevals.oai import get_default_model
from autoevals.ragas import _get_model


def test_ragas_default_matches_library_default():
init()
assert _get_model(None) == get_default_model()


def test_ragas_respects_explicit_model():
init()
assert _get_model("gpt-4o") == "gpt-4o"


def test_ragas_respects_configured_default():
init(default_model="gpt-4-turbo")
assert _get_model(None) == "gpt-4-turbo"