diff --git a/py/autoevals/ragas.py b/py/autoevals/ragas.py index 4726159..3bd27b2 100644 --- a/py/autoevals/ragas.py +++ b/py/autoevals/ragas.py @@ -1179,8 +1179,8 @@ async def _run_eval_async(self, output, expected=None, input=None, context=None, ) similarity = await asyncio.gather( *[ - EmbeddingSimilarity(client=self.client).eval_async( - output=q["question"], expected=input, model=self.embedding_model + EmbeddingSimilarity(client=self.client, model=self.embedding_model).eval_async( + output=q["question"], expected=input ) for q in questions ] @@ -1199,8 +1199,8 @@ def _run_eval_sync(self, output, expected=None, input=None, context=None, **kwar for _ in range(self.strictness) ] similarity = [ - EmbeddingSimilarity(client=self.client).eval( - output=q["question"], expected=input, model=self.embedding_model + EmbeddingSimilarity(client=self.client, model=self.embedding_model).eval( + output=q["question"], expected=input ) for q in questions ] diff --git a/py/autoevals/test_ragas.py b/py/autoevals/test_ragas.py index 74db4e5..9f3315b 100644 --- a/py/autoevals/test_ragas.py +++ b/py/autoevals/test_ragas.py @@ -4,12 +4,13 @@ import pytest import respx from httpx import Response -from openai import OpenAI +from openai import AsyncOpenAI, OpenAI import autoevals.ragas as ragas_module from autoevals import init from autoevals.ragas import * + data = { "input": "Can starred docs from different workspaces be accessed in one place?", "output": "Yes, all starred docs, even from multiple different workspaces, will live in the My Shortcuts section.", @@ -262,3 +263,135 @@ def mock_responses_api(request): ) assert captured_embedding_model == "text-embedding-3-large" + + +def _make_answer_relevancy_mocks(question: str): + """Build respx side effects for AnswerRelevancy: question generation + embeddings capture. + + Returns (captured_embedding_models, chat_mock, responses_mock, embeddings_mock). + The question must be unique per test because EmbeddingSimilarity caches embeddings + by input text, so a repeated question would skip the embeddings API entirely. + """ + captured_embedding_models = [] + + def mock_chat_completions(request): + return Response( + 200, + json={ + "id": "test-id", + "object": "chat.completion", + "created": 1234567890, + "model": "gpt-5-mini", + "choices": [ + { + "index": 0, + "message": { + "role": "assistant", + "tool_calls": [ + { + "id": "call_test", + "type": "function", + "function": { + "name": "generate_question", + "arguments": json.dumps({"question": question, "noncommittal": 0}), + }, + } + ], + }, + "finish_reason": "tool_calls", + } + ], + }, + ) + + def mock_responses_api(request): + return Response( + 200, + json={ + "id": "test-id", + "object": "response", + "created": 1234567890, + "model": "gpt-5-mini", + "output": [ + { + "type": "function_call", + "call_id": "call_test", + "name": "generate_question", + "arguments": json.dumps({"question": question, "noncommittal": 0}), + } + ], + }, + ) + + def mock_embeddings(request): + data = json.loads(request.content.decode()) + captured_embedding_models.append(data.get("model")) + return Response( + 200, + json={ + "object": "list", + "data": [ + { + "object": "embedding", + "embedding": [0.1] * 1536, + "index": 0, + } + ], + "model": data.get("model"), + "usage": {"prompt_tokens": 5, "total_tokens": 5}, + }, + ) + + return captured_embedding_models, mock_chat_completions, mock_responses_api, mock_embeddings + + +@respx.mock +def test_answer_relevancy_uses_custom_embedding_model(): + """Test that AnswerRelevancy passes embedding_model through to the embeddings API (#166).""" + captured, chat_mock, responses_mock, embeddings_mock = _make_answer_relevancy_mocks( + question="What is the sync capital of France?" + ) + + respx.post("https://api.openai.com/v1/chat/completions").mock(side_effect=chat_mock) + respx.post("https://api.openai.com/v1/responses").mock(side_effect=responses_mock) + respx.post("https://api.openai.com/v1/embeddings").mock(side_effect=embeddings_mock) + + init(OpenAI(api_key="test-api-key", base_url="https://api.openai.com/v1")) + + # Pin a chat-completions model so question generation hits the mocked endpoint above. + metric = AnswerRelevancy(model="gpt-4o-mini", embedding_model="text-embedding-3-large", strictness=1) + metric.eval( + input="What is the sync capital of France?!", + output="Paris", + context="Paris is the capital of France.", + ) + + assert captured, "expected at least one embeddings API call" + assert all(model == "text-embedding-3-large" for model in captured), captured + + +@respx.mock +def test_answer_relevancy_uses_custom_embedding_model_async(): + """Test that AnswerRelevancy passes embedding_model through to the embeddings API in the async path (#166).""" + captured, chat_mock, responses_mock, embeddings_mock = _make_answer_relevancy_mocks( + question="What is the async capital of France?" + ) + + respx.post("https://api.openai.com/v1/chat/completions").mock(side_effect=chat_mock) + respx.post("https://api.openai.com/v1/responses").mock(side_effect=responses_mock) + respx.post("https://api.openai.com/v1/embeddings").mock(side_effect=embeddings_mock) + + init(AsyncOpenAI(api_key="test-api-key", base_url="https://api.openai.com/v1")) + + # Pin a chat-completions model so question generation hits the mocked endpoint above. + metric = AnswerRelevancy(model="gpt-4o-mini", embedding_model="text-embedding-3-large", strictness=1) + asyncio.run( + metric.eval_async( + input="What is the async capital of France?!", + output="Paris", + context="Paris is the capital of France.", + ) + ) + + assert captured, "expected at least one embeddings API call" + assert all(model == "text-embedding-3-large" for model in captured), captured