agent testing.md
Automated Testing for Guava Agents
Guava Agent tests are defined in code. You can write them using your existing testing framework, such as pytest or Python's built-in unittest module. This lets you test your Agent end-to-end alongside your Expert's code, using patterns you're already familiar with.
Test individual Agent handlers with MockCall
import unittest
from guava.testing import MockCall
# Import a handler from the help desk example.
from guava.examples.help_desk import on_action_request
class TestHandlers(unittest.TestCase):
def test_routes_to_sales(self):
# We can directly invoke that handler with a mocked Call object.
actions = on_action_request(MockCall(), "I want to buy a new sofa")
self.assertEqual("sales", actions[0].key)
import { MockCall } from "@guava-ai/guava-sdk";
// Import the agent from the help-desk example.
import { agent } from "@guava-ai/guava-sdk/examples/help-desk";
test("routes to sales", async () => {
// We can directly invoke that handler with a mocked Call object.
const suggestion = await agent.handlers.onActionRequest(new MockCall(), "I want to buy a new sofa");
expect(suggestion).toHaveProperty("key", "sales");
});
The most basic way to test Guava Agents is to test individual handlers like on_action_request or on_question. You can do so by constructing a guava.testing.MockCall object and passing it in directly to the handler.
Run a roleplay session and analyze the result
import unittest
# Import the agent from the help desk example.
from guava.examples.help_desk import agent
class TestHelpDeskAgent(unittest.TestCase):
def test_purchase_roleplay(self):
session = agent.roleplay("You are a caller who wants to buy a new dining table.")
# Use session.get_transcript() to retrieve the session's transcript.
print(session.get_transcript())
# Use session.evaluate() to assess the Agent's performance against a rubric.
session.evaluate(
# These criteria must all be true.
pass_criteria=["The agent identified itself as working for Clearfield Home & Living."],
# These criteria must all be false.
fail_criteria=["The agent directly offered to search the inventory."],
)
# You can assert some values directly on the session object.
self.assertIn("sales", session.executed_actions)
self.assertEqual("bot-transfer", session.termination_reason)
import { agent } from "@guava-ai/guava-sdk/examples/help-desk";
test("purchase roleplay", async () => {
const session = await agent.roleplay("You are a caller who wants to buy a new dining table.");
// Use session.getTranscript() to retrieve the session's transcript.
console.log(session.getTranscript());
// Use session.evaluate() to assess the Agent's performance against a rubric.
await session.evaluate({
// These criteria must all be true.
passCriteria: ["The agent identified itself as working for Clearfield Home & Living."],
// These criteria must all be false.
failCriteria: ["The agent directly offered to search the inventory."],
});
// You can assert some values directly on the session object.
expect(session.executedActions).toContain("sales");
expect(session.terminationReason).toBe("bot-transfer");
});
The returned session object contains the transcript, useful helpers that can be asserted against, and a session.evaluate function that evaluates the conversation against a rubric.
Patch Agent handlers before testing
In some cases, you may not want every handler to run during a test. Use agent.patch() to create a clone of the Agent where you can override callbacks without modifying the original Agent.
import unittest
import guava
# Import the agent from the help desk example.
from guava.examples.help_desk import agent
class TestHelpDeskAgent(unittest.TestCase):
def test_sales_closed(self):
# Create a clone of the agent and patch the on_action handler.
patched = agent.patch()
@patched.on_action("sales")
def patched_sales(call: guava.Call):
call.hangup(
"Tell the caller that the sales department is closed and that "
"they should call back tomorrow between 9am and 5pm."
)
# Run the roleplay test with our patched agent.
session = patched.roleplay("You are a caller who wants to buy a new dining table.")
session.evaluate(["The agent informed the caller of the business hours from 9am to 5pm."])
import type { Call } from "@guava-ai/guava-sdk";
import { agent } from "@guava-ai/guava-sdk/examples/help-desk";
test("sales closed", async () => {
// Create a clone of the agent and patch the onAction handler.
const patched = agent.patch();
patched.onAction("sales", async (call: Call) => {
call.hangup(
"Tell the caller that the sales department is closed and that " +
"they should call back tomorrow between 9am and 5pm."
);
});
// Run the roleplay test with our patched agent.
const session = await patched.roleplay("You are a caller who wants to buy a new dining table.");
await session.evaluate({
passCriteria: ["The agent informed the caller of the business hours from 9am to 5pm."],
});
});
For any function that isn't an Agent handler, you can patch it using Python's builtin patching system.
Run a session with full control
import unittest
# Import the agent from the help desk example.
from guava.examples.help_desk import agent
class TestHelpDeskAgent(unittest.TestCase):
def test_purchase_routes_to_sales(self):
with agent.test() as session:
# Wait until the agent has finished its opening turn.
session.wait_for_turn();
# Inject a caller utterance and let the agent respond.
session.say("Hi, I'm looking to make a new purchase.");
# Wait until the session ends (transfer, hangup, etc.).
session.wait_for_end();
# The TestSession here is the same type as those returned from roleplay sessions.
# Assert against any of its values, read the transcript, or use 'session.evaluate(...)'
self.assertIn("sales", session.executed_actions)
self.assertEqual("bot-transfer", session.termination_reason)
import { agent } from "@guava-ai/guava-sdk/examples/help-desk";
test("purchase routes to sales", async () => {
const session = await agent.test(async (session) => {
// Wait until the agent has finished its opening turn.
await session.waitForTurn();
// Inject a caller utterance and let the agent respond.
session.say("Hi, I'm looking to make a new purchase.");
// Wait until the session ends (transfer, hangup, etc.).
await session.waitForEnd();
});
// The TestSession here is the same type as those returned from roleplay sessions.
// Assert against any of its values, read the transcript, or use 'session.evaluate(...)'
expect(session.executedActions).toContain("sales");
expect(session.terminationReason).toBe("bot-transfer");
});
agent.test() gives you a live test session where you have full control over timing and caller utterances. Use this when you need precise control over how the conversation unfolds.