Inference gateway integration (#22)
This commit is contained in:
+2
-115
@@ -1,12 +1,11 @@
|
||||
import pytest
|
||||
from livekit.agents import AgentSession, llm, mock_tools
|
||||
from livekit.plugins import openai
|
||||
from livekit.agents import AgentSession, inference, llm
|
||||
|
||||
from agent import Assistant
|
||||
|
||||
|
||||
def _llm() -> llm.LLM:
|
||||
return openai.LLM(model="gpt-4o-mini")
|
||||
return inference.LLM(model="openai/gpt-4.1-mini")
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
@@ -41,118 +40,6 @@ async def test_offers_assistance() -> None:
|
||||
result.expect.no_more_events()
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_weather_tool() -> None:
|
||||
"""Unit test for the weather tool combined with an evaluation of the agent's ability to incorporate its results."""
|
||||
async with (
|
||||
_llm() as llm,
|
||||
AgentSession(llm=llm) as session,
|
||||
):
|
||||
await session.start(Assistant())
|
||||
|
||||
# Run an agent turn following the user's request for weather information
|
||||
result = await session.run(user_input="What's the weather in Tokyo?")
|
||||
|
||||
# Test that the agent calls the weather tool with the correct arguments
|
||||
result.expect.next_event().is_function_call(
|
||||
name="lookup_weather", arguments={"location": "Tokyo"}
|
||||
)
|
||||
|
||||
# Test that the tool invocation works and returns the correct output
|
||||
# To mock the tool output instead, see https://docs.livekit.io/agents/build/testing/#mock-tools
|
||||
result.expect.next_event().is_function_call_output(
|
||||
output="sunny with a temperature of 70 degrees."
|
||||
)
|
||||
|
||||
# Evaluate the agent's response for accurate weather information
|
||||
await (
|
||||
result.expect.next_event()
|
||||
.is_message(role="assistant")
|
||||
.judge(
|
||||
llm,
|
||||
intent="""
|
||||
Informs the user that the weather is sunny with a temperature of 70 degrees.
|
||||
|
||||
Optional context that may or may not be included (but the response must not contradict these facts)
|
||||
- The location for the weather report is Tokyo
|
||||
""",
|
||||
)
|
||||
)
|
||||
|
||||
# Ensures there are no function calls or other unexpected events
|
||||
result.expect.no_more_events()
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_weather_unavailable() -> None:
|
||||
"""Evaluation of the agent's ability to handle tool errors."""
|
||||
async with (
|
||||
_llm() as llm,
|
||||
AgentSession(llm=llm) as sess,
|
||||
):
|
||||
await sess.start(Assistant())
|
||||
|
||||
# Simulate a tool error
|
||||
with mock_tools(
|
||||
Assistant,
|
||||
{"lookup_weather": lambda: RuntimeError("Weather service is unavailable")},
|
||||
):
|
||||
result = await sess.run(user_input="What's the weather in Tokyo?")
|
||||
result.expect.skip_next_event_if(type="message", role="assistant")
|
||||
result.expect.next_event().is_function_call(
|
||||
name="lookup_weather", arguments={"location": "Tokyo"}
|
||||
)
|
||||
result.expect.next_event().is_function_call_output()
|
||||
await result.expect.next_event(type="message").judge(
|
||||
llm,
|
||||
intent="""
|
||||
Acknowledges that the weather request could not be fulfilled and communicates this to the user.
|
||||
|
||||
The response should convey that there was a problem getting the weather information, but can be expressed in various ways such as:
|
||||
- Mentioning an error, service issue, or that it couldn't be retrieved
|
||||
- Suggesting alternatives or asking what else they can help with
|
||||
- Being apologetic or explaining the situation
|
||||
|
||||
The response does not need to use specific technical terms like "weather service error" or "temporary".
|
||||
""",
|
||||
)
|
||||
|
||||
# leaving this commented, some LLMs may occasionally try to retry.
|
||||
# result.expect.no_more_events()
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_unsupported_location() -> None:
|
||||
"""Evaluation of the agent's ability to handle a weather response with an unsupported location."""
|
||||
async with (
|
||||
_llm() as llm,
|
||||
AgentSession(llm=llm) as sess,
|
||||
):
|
||||
await sess.start(Assistant())
|
||||
|
||||
with mock_tools(Assistant, {"lookup_weather": lambda: "UNSUPPORTED_LOCATION"}):
|
||||
result = await sess.run(user_input="What's the weather in Tokyo?")
|
||||
|
||||
# Evaluate the agent's response for an unsupported location
|
||||
await result.expect.next_event(type="message").judge(
|
||||
llm,
|
||||
intent="""
|
||||
Communicates that the weather request for the specific location could not be fulfilled.
|
||||
|
||||
The response should indicate that weather information is not available for the requested location, but can be expressed in various ways such as:
|
||||
- Saying they can't get weather for that location
|
||||
- Explaining the location isn't supported or available
|
||||
- Suggesting alternatives or asking what else they can help with
|
||||
- Being apologetic about the limitation
|
||||
|
||||
The response does not need to explicitly state "unsupported" or discourage retrying.
|
||||
""",
|
||||
)
|
||||
|
||||
# Ensures there are no function calls or other unexpected events
|
||||
result.expect.no_more_events()
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_grounding() -> None:
|
||||
"""Evaluation of the agent's ability to refuse to answer when it doesn't know something."""
|
||||
|
||||
Reference in New Issue
Block a user