Source code for pipecat.evals.audio

#
# Copyright (c) 2024-2026, Daily
#
# SPDX-License-Identifier: BSD 2-Clause License
#

"""Audio files played as a user's turn.

A turn's ``audio:`` names a recording to play instead of synthesizing its
text. Any format libsndfile reads works: WAV, MP3, FLAC, OGG.
"""

import asyncio

import numpy as np
import soundfile as sf

__all__ = ["load_user_audio"]


[docs] async def load_user_audio(path: str) -> tuple[bytes, int]: """Read an audio file as the PCM a user turn is sent as. The file keeps its own sample rate; the bot resamples on its way in. Args: path: Path to the audio file. Returns: Tuple of ``(pcm_bytes, sample_rate)`` -- raw 16-bit little-endian mono PCM. Raises: ValueError: If the file cannot be read as audio. """ return await asyncio.to_thread(_read, path)
def _read(path: str) -> tuple[bytes, int]: """Read and downmix the file (blocking; run off the event loop).""" data: np.ndarray try: # always_2d keeps the channel axis so mono and multi-channel read alike. data, sample_rate = sf.read(path, dtype="int16", always_2d=True) except Exception as e: raise ValueError(f"Could not read user audio {path!r}: {e}") from e if data.shape[0] == 0: raise ValueError(f"User audio {path!r} is empty") # A recording of one speaker carries the same voice on every channel, so the # mean is the mono take on it. Averaging in float avoids int16 overflow. mono = data[:, 0] if data.shape[1] == 1 else data.mean(axis=1).astype(np.int16) return mono.tobytes(), int(sample_rate)