Source code for pipecat.evals.persona

#
# Copyright (c) 2024-2026, Daily
#
# SPDX-License-Identifier: BSD 2-Clause License
#

"""The simulated caller: who they are, what they want, and how they end the call.

An :class:`EvalPersona` is the instruction the persona LLM runs under, the
LLM service itself, which rides in the eval pipeline, and the context it
runs on, where the bot's turns are the ``user`` messages and its own the
``assistant`` ones. The context offers one tool, ``end_call``, which the
persona calls once its goal is achieved or clearly out of reach; its
``success`` claim is its own view, and the judge decides the outcome.
"""

from collections.abc import Awaitable, Callable

from pipecat.adapters.schemas.function_schema import FunctionSchema
from pipecat.adapters.schemas.tools_schema import ToolsSchema
from pipecat.frames.frames import LLMContextFrame
from pipecat.processors.aggregators.llm_context import LLMContext
from pipecat.services.llm_service import FunctionCallParams, LLMService

END_CALL_FUNCTION = "end_call"

END_CALL_SCHEMA = FunctionSchema(
    name=END_CALL_FUNCTION,
    description=(
        "Hang up the phone. This is the only way to end the call; saying goodbye "
        "does not end it. Call it as soon as your goal is achieved, or as soon as "
        "it is clear the assistant cannot help you achieve it. A short goodbye may "
        "be spoken in the same turn."
    ),
    properties={
        "success": {
            "type": "boolean",
            "description": "Whether you achieved your goal on this call.",
        },
        "reason": {
            "type": "string",
            "description": "One sentence on why you are ending the call.",
        },
    },
    required=["success", "reason"],
)

_INSTRUCTION_TEMPLATE = """\
You are playing a person on a phone call with a voice assistant. Stay in \
character throughout; never mention being simulated, a test, or an AI.

Who you are: {persona}

What you want from this call: {goal}

The assistant's words arrive as the user's messages. Reply with only what you \
would say next: one short spoken turn, in the first person, in plain sentences \
(no lists, markdown, or stage directions). Ask for or give one thing at a time, \
as a real caller would, and do not repeat what the assistant has already \
understood.

Saying goodbye does not hang up: the call ends only when you call the \
{end_call} tool. As soon as your goal is achieved, or it is clear the \
assistant cannot help, call {end_call} with whether you succeeded and why. You \
may say one short goodbye sentence in the same turn, but you must make the call."""


[docs] class EvalPersona: """The simulated caller: its instruction, its LLM, and the context it runs on."""
[docs] def __init__(self, description: str, goal: str, llm: LLMService): """Initialize the persona. Args: description: Who the caller is, as free text. goal: What the caller wants from the call. llm: The LLM service that plays the caller, run inside the eval pipeline. """ self._description = description self._goal = goal self._llm = llm self._context = LLMContext(tools=ToolsSchema(standard_tools=[END_CALL_SCHEMA])) # What the persona has heard of the bot's current turn (text mode). self._heard: list[str] = [] self._calls_in_progress = 0 self._hung_up = False
@property def instruction(self) -> str: """The system instruction the persona LLM runs under.""" return _INSTRUCTION_TEMPLATE.format( persona=self._description, goal=self._goal, end_call=END_CALL_FUNCTION ) @property def llm(self) -> LLMService: """The LLM service that plays the caller, a processor in the eval pipeline.""" return self._llm @property def context(self) -> LLMContext: """The context the persona runs on, with the ``end_call`` tool; empty until the call starts.""" return self._context @property def hung_up(self) -> bool: """Whether the persona has ended its part of the call.""" return self._hung_up
[docs] def hang_up(self) -> None: """End the persona's part of the call: it answers nothing more.""" self._hung_up = True
[docs] def hear(self, event: dict) -> LLMContextFrame | None: """Take in one of the bot's events; the frame that has the persona answer, once the bot's turn is done. A response the bot gives while one of its function calls is still running is held and joined with the response after the call, so the persona answers the bot's whole turn rather than its "let me check". In audio mode the aggregator does this instead. Args: event: An event from the stream. Returns: The context frame to push to the persona LLM, or ``None``. """ if self._hung_up: return None match event["type"]: case "function_call": self._calls_in_progress += 1 case "function_call_stopped": self._calls_in_progress = max(0, self._calls_in_progress - 1) case "llm_response": if event["text"]: self._heard.append(event["text"]) if self._calls_in_progress or not self._heard: return None text = " ".join(self._heard) self._heard = [] self._context.add_message({"role": "user", "content": text}) return LLMContextFrame(self._context) return None
[docs] def on_end_call(self, handler: Callable[[FunctionCallParams], Awaitable[None]]) -> None: """Register what happens when the persona calls ``end_call``. Args: handler: Awaited with the call's params; its ``success`` and ``reason`` arguments are the persona's own claim. """ self._llm.register_function(END_CALL_FUNCTION, handler)