pepperplus-cb/test/unit/agents/test_vad_streaming.py

from unittest.mock import AsyncMock, MagicMock

import numpy as np
import pytest

from control_backend.agents.vad_agent import Streaming


@pytest.fixture
def audio_in_socket():
    return AsyncMock()


@pytest.fixture
def audio_out_socket():
    return AsyncMock()


@pytest.fixture
def streaming(audio_in_socket, audio_out_socket):
    return Streaming(audio_in_socket, audio_out_socket)


async def simulate_streaming_with_probabilities(streaming, probabilities: list[float]):
    """
    Simulates a streaming scenario with given VAD model probabilities for testing purposes.

    :param streaming: The streaming component to be tested.
    :param probabilities: A list of probabilities representing the outputs of the VAD model.
    """
    model_item = MagicMock()
    model_item.item.side_effect = probabilities
    streaming.model = MagicMock()
    streaming.model.return_value = model_item

    audio_in_poller = AsyncMock()
    audio_in_poller.poll.return_value = np.empty(shape=512, dtype=np.float32)
    streaming.audio_in_poller = audio_in_poller

    for _ in probabilities:
        await streaming.run()


@pytest.mark.asyncio
async def test_voice_activity_detected(audio_in_socket, audio_out_socket, streaming):
    """
    Test a scenario where there is voice activity detected between silences.
    :return:
    """
    speech_chunk_count = 5
    probabilities = [0.0] * 5 + [1.0] * speech_chunk_count + [0.0] * 5
    await simulate_streaming_with_probabilities(streaming, probabilities)

    audio_out_socket.send.assert_called_once()
    data = audio_out_socket.send.call_args[0][0]
    assert isinstance(data, bytes)
    # each sample has 512 frames of 4 bytes, expecting 7 chunks (5 with speech, 2 as padding)
    assert len(data) == 512 * 4 * (speech_chunk_count + 2)


@pytest.mark.asyncio
async def test_voice_activity_short_pause(audio_in_socket, audio_out_socket, streaming):
    """
    Test a scenario where there is a short pause between speech, checking whether it ignores the
    short pause.
    """
    speech_chunk_count = 5
    probabilities = (
        [0.0] * 5 + [1.0] * speech_chunk_count + [0.0] + [1.0] * speech_chunk_count + [0.0] * 5
    )
    await simulate_streaming_with_probabilities(streaming, probabilities)

    audio_out_socket.send.assert_called_once()
    data = audio_out_socket.send.call_args[0][0]
    assert isinstance(data, bytes)
    # Expecting 13 chunks (2*5 with speech, 1 pause between, 2 as padding)
    assert len(data) == 512 * 4 * (speech_chunk_count * 2 + 1 + 2)


@pytest.mark.asyncio
async def test_no_data(audio_in_socket, audio_out_socket, streaming):
    """
    Test a scenario where there is no data received. This should not cause errors.
    """
    audio_in_poller = AsyncMock()
    audio_in_poller.poll.return_value = None
    streaming.audio_in_poller = audio_in_poller

    assert streaming.i_since_data == 0

    await streaming.run()

    audio_out_socket.send.assert_not_called()
    assert streaming.i_since_data == 1