Skip to content
Closed
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
1 change: 1 addition & 0 deletions changelog/+aic-sdk-021-vad.changed.md
Original file line number Diff line number Diff line change
@@ -0,0 +1 @@
Updated the ai-coustics integration to SDK 0.21 (bumped `aic-sdk` to `~=2.5.0`). `AICQuailVADAnalyzer` now reports the model's continuous raw VAD probability (`VadContext.raw_vad_probability()`) gated by Pipecat's `VADParams` instead of a binary speech flag, and defaults to the new `quail-vf-vad-2.0-s-16khz` model. The ai-coustics voice examples now use the `quail-vf-2.2-l-16khz` enhancement model.
1 change: 1 addition & 0 deletions changelog/+aic-tyto-analyzer.added.md
Original file line number Diff line number Diff line change
@@ -0,0 +1 @@
Added `AICTytoAnalyzer`, a real-time audio-quality processor backed by the ai-coustics Tyto model (aic-sdk 2.4.0+). It taps the pipeline's input audio and periodically emits an `AICAudioQualityMetricsData` (seven `0.0`–`1.0` scores predicting downstream STT/VAD/turn-taking degradation) via a `MetricsFrame` and an `on_audio_analysis` event. The scores are also forwarded to RTVI clients under the `audio_quality` metrics key.
90 changes: 90 additions & 0 deletions examples/voice/voice-aicoustics-audio-quality.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,90 @@
#
# Copyright (c) 2024-2026, Daily
#
# SPDX-License-Identifier: BSD 2-Clause License
#

"""Minimal audio-quality test bot for AICTytoAnalyzer.

Runs only the ai-coustics Tyto analysis model on the input audio so you can
watch its quality scores react to speech, silence, and background noise without
paying for STT/LLM/TTS API calls. The AICTytoAnalyzer is placed right after
``transport.input()`` so it scores the raw microphone signal (move it after an
AICFilter to score enhanced audio instead).

Logging:
- INFO "audio quality" once per ``analysis_interval`` with the seven Tyto
scores. ``risk_score`` (and ``noise`` / ``interfering_speech``) rising
under poor conditions is the signal that the analyzer is working.
- DEBUG init lines from AICTytoAnalyzer (run with LOGURU_LEVEL=DEBUG).

Required env vars:
AIC_SDK_LICENSE ai-coustics SDK license key
Plus whatever credentials the chosen transport needs (DAILY_*, etc.)

Run:
LOGURU_LEVEL=DEBUG uv run python examples/voice/voice-aicoustics-audio-quality.py daily
"""

import os

from dotenv import load_dotenv
from loguru import logger

from pipecat.metrics.metrics import AICAudioQualityMetricsData
from pipecat.pipeline.pipeline import Pipeline
from pipecat.pipeline.worker import PipelineParams, PipelineWorker
from pipecat.processors.audio.aic_tyto_analyzer import AICTytoAnalyzer
from pipecat.runner.types import RunnerArguments
from pipecat.runner.utils import create_transport
from pipecat.transports.base_transport import BaseTransport, TransportParams
from pipecat.transports.daily.transport import DailyParams
from pipecat.transports.websocket.fastapi import FastAPIWebsocketParams
from pipecat.workers.runner import WorkerRunner

load_dotenv(override=True)


aic_tyto_analyzer = AICTytoAnalyzer(
license_key=os.environ["AIC_SDK_LICENSE"],
analysis_interval=1.0,
)


@aic_tyto_analyzer.event_handler("on_audio_analysis")
async def on_audio_analysis(_processor, scores: AICAudioQualityMetricsData) -> None:
logger.info(
"audio quality: "
f"risk={scores.risk_score:.2f} noise={scores.noise:.2f} "
f"interfering_speech={scores.interfering_speech:.2f} "
f"media_speech={scores.media_speech:.2f} reverb={scores.speaker_reverb:.2f} "
f"loudness={scores.speaker_loudness:.2f} packet_loss={scores.packet_loss:.2f}"
)


transport_params = {
"daily": lambda: DailyParams(audio_in_enabled=True),
"twilio": lambda: FastAPIWebsocketParams(audio_in_enabled=True),
"webrtc": lambda: TransportParams(audio_in_enabled=True),
}


async def run_bot(transport: BaseTransport, runner_args: RunnerArguments) -> None:
logger.info("Audio-quality test bot starting")
pipeline = Pipeline([transport.input(), aic_tyto_analyzer])
worker = PipelineWorker(pipeline, params=PipelineParams(enable_metrics=True))
runner = WorkerRunner(handle_sigint=runner_args.handle_sigint)
await runner.add_workers(worker)
await runner.run()


async def bot(runner_args: RunnerArguments) -> None:
"""Main bot entry point compatible with Pipecat Cloud."""
transport = await create_transport(runner_args, transport_params)
await run_bot(transport, runner_args)


if __name__ == "__main__":
from pipecat.runner.run import main

main()
2 changes: 1 addition & 1 deletion examples/voice/voice-aicoustics-vad-only.py
Original file line number Diff line number Diff line change
Expand Up @@ -144,7 +144,7 @@ async def process_frame(self, frame: Frame, direction: FrameDirection) -> None:

aic_filter = AICFilter(
license_key=os.environ["AIC_SDK_LICENSE"],
model_id="quail-vf-2.1-l-16khz",
model_id="quail-vf-2.2-l-16khz",
enhancement_level=0.8,
)
aic_vad_analyzer = LoggingAICQuailVADAnalyzer(
Expand Down
2 changes: 1 addition & 1 deletion examples/voice/voice-aicoustics.py
Original file line number Diff line number Diff line change
Expand Up @@ -42,7 +42,7 @@ def _create_aic_filter() -> AICFilter:

return AICFilter(
license_key=license_key,
model_id="quail-vf-2.1-l-16khz",
model_id="quail-vf-2.2-l-16khz",
enhancement_level=0.8,
)

Expand Down
2 changes: 1 addition & 1 deletion pyproject.toml
Original file line number Diff line number Diff line change
Expand Up @@ -62,7 +62,7 @@ pipecat = "pipecat.cli.main:run"
pc = "pipecat.cli.main:run"

[project.optional-dependencies]
aic = [ "aic-sdk~=2.3.0" ]
aic = [ "aic-sdk~=2.5.0" ]
anthropic = [ "anthropic>=0.49.0,<1" ]
assemblyai = []
asyncai = []
Expand Down
96 changes: 29 additions & 67 deletions src/pipecat/audio/vad/aic_quail_vad.py
Original file line number Diff line number Diff line change
Expand Up @@ -4,16 +4,17 @@
# SPDX-License-Identifier: BSD 2-Clause License
#

"""Standalone Quail VAD 2.0 analyzer for Pipecat.
"""Standalone Quail VAD analyzer for Pipecat.

Runs the standalone Quail VAD 2.0 model from the ai-coustics SDK as a dedicated
VAD-only processor. Unlike :class:`pipecat.audio.vad.aic_vad.AICVADAnalyzer`,
which queries the model-internal VAD of :class:`pipecat.audio.filters.aic_filter.AICFilter`,
this analyzer owns its own :class:`aic_sdk.Processor` instance and can be placed
Runs a standalone Quail VAD-only model from the ai-coustics SDK (e.g. Quail VAD
2.0 or VF VAD 2.0) as a dedicated VAD processor. Unlike
:class:`pipecat.audio.vad.aic_vad.AICVADAnalyzer`, which queries the
model-internal VAD of :class:`pipecat.audio.filters.aic_filter.AICFilter`, this
analyzer owns its own :class:`aic_sdk.Processor` instance and can be placed
anywhere in the pipeline.

Classes:
AICQuailVADAnalyzer: Standalone Quail VAD 2.0 analyzer.
AICQuailVADAnalyzer: Standalone Quail VAD analyzer.
"""

from __future__ import annotations
Expand All @@ -26,7 +27,6 @@
Model,
Processor,
ProcessorConfig,
VadParameter,
set_sdk_id,
)
from loguru import logger
Expand All @@ -36,7 +36,7 @@
if TYPE_CHECKING:
from aic_sdk import VadContext

DEFAULT_QUAIL_VAD_MODEL_ID = "quail-vad-2.0-xxs-16khz"
DEFAULT_QUAIL_VAD_MODEL_ID = "quail-vf-vad-2.0-s-16khz"

# Telemetry identifier registered with the AIC SDK; identifies pipecat to the
# vendor's usage pipeline. Mirrors the value used by AICFilter; kept private
Expand All @@ -53,35 +53,27 @@ class AICQuailVADAnalyzer(VADAnalyzer):

The analyzer owns a dedicated :class:`aic_sdk.Processor` initialized with a
Quail VAD-only model. Each :meth:`voice_confidence` call processes one audio
window through the processor and queries the resulting
:class:`aic_sdk.VadContext` for the speech-detected boolean, which is mapped
to ``1.0`` / ``0.0`` to satisfy the :class:`VADAnalyzer` interface.
window through the processor and returns the model's raw speech probability
in ``[0.0, 1.0]`` (:meth:`aic_sdk.VadContext.raw_vad_probability`). The base
:class:`VADAnalyzer` state machine then gates speech start/stop using its own
:class:`VADParams` (``confidence`` threshold, ``start_secs``, ``stop_secs``),
so the SDK's own VAD post-processing (sensitivity thresholding, speech-hold)
is intentionally bypassed — Pipecat owns the thresholding.

Comparison to :class:`pipecat.audio.vad.aic_vad.AICVADAnalyzer` (deprecated):

- **Model:** Quail VAD-only model (e.g. ``quail-vad-2.0-xxs-16khz``); the
- **Model:** Quail VAD-only model (e.g. ``quail-vf-vad-2.0-s-16khz``); the
deprecated analyzer uses the enhancement model's internal VAD as a
side-channel.
- **Audio path:** runs on whatever the pipeline feeds it (raw or enhanced).
The deprecated analyzer reads post-enhancement VAD state from
:class:`AICFilter`'s processor.
- **Sensitivity semantics:** speech-probability threshold in ``[0.0, 1.0]``
on dedicated VAD models. The deprecated analyzer's enhancement-model VAD
uses an energy threshold in ``[1.0, 15.0]``.
- **Confidence:** a continuous raw probability gated by Pipecat's
``VADParams.confidence``. The deprecated analyzer exposes only a boolean
gated by the enhancement model's energy threshold (``[1.0, 15.0]``).
- **Coupling:** independent — owns its own ``Processor``. The deprecated
analyzer is bound to an :class:`AICFilter` instance.

Quail VAD parameters (applied via :class:`aic_sdk.VadParameter`):

- **speech_hold_duration**: seconds the VAD continues reporting speech after
the signal stops containing speech. Range 0.0 to 300x the model window
length. Default 0.03s.
- **minimum_speech_duration**: seconds of speech required before the VAD
reports speech detected. Range 0.0 to 1.0. Default 0.0s.
- **sensitivity**: speech-probability threshold on dedicated Quail VAD
models (range 0.0 to 1.0). Energy-based VADs keep the 1.0 to 15.0 range.
Default is model-specific.

Example::

analyzer = AICQuailVADAnalyzer(license_key=os.environ["AIC_SDK_LICENSE"])
Expand All @@ -96,9 +88,6 @@ def __init__(
model_id: str | None = DEFAULT_QUAIL_VAD_MODEL_ID,
model_path: Path | None = None,
model_download_dir: Path | None = None,
speech_hold_duration: float | None = None,
minimum_speech_duration: float | None = None,
sensitivity: float | None = None,
sample_rate: int | None = None,
params: VADParams | None = None,
) -> None:
Expand All @@ -111,21 +100,13 @@ def __init__(
Args:
license_key: ai-coustics SDK license key.
model_id: Quail VAD model identifier. Defaults to the published
standalone VAD model ``"quail-vad-2.0-xxs-16khz"``. See
standalone VAD model ``"quail-vf-vad-2.0-s-16khz"``. See
https://artifacts.ai-coustics.io/ for the catalogue. Ignored if
``model_path`` is provided.
model_path: Optional path to a local ``.aicmodel`` file. Overrides
``model_id`` when set.
model_download_dir: Directory for downloaded models. Defaults to
``~/.cache/pipecat/aic-models``.
speech_hold_duration: Optional override for the SDK's
``VadParameter.SpeechHoldDuration``.
minimum_speech_duration: Optional override for the SDK's
``VadParameter.MinimumSpeechDuration``.
sensitivity: Optional override for the SDK's
``VadParameter.Sensitivity``. This is a probability threshold
in ``[0.0, 1.0]``. Values above this threshold are considered
speech.
sample_rate: Initial sample rate; the pipeline will set this via
:meth:`set_sample_rate` once the transport rate is known.
params: Optional :class:`VADParams` for the base state machine.
Expand All @@ -148,10 +129,6 @@ def __init__(
Path.home() / ".cache" / "pipecat" / "aic-models"
)

self._pending_speech_hold_duration = speech_hold_duration
self._pending_minimum_speech_duration = minimum_speech_duration
self._pending_sensitivity = sensitivity

self._model: Model | None = None
self._processor: Processor | None = None
self._vad_ctx: VadContext | None = None
Expand Down Expand Up @@ -235,28 +212,8 @@ def _initialize_processor(self, sample_rate: int) -> None:
self._in_f32 = np.zeros((1, num_frames), dtype=np.float32)
self._inference_error_logged = False
self._buffer_size_warning_logged = False
self._apply_vad_parameters()
logger.debug(f"AICQuailVADAnalyzer initialized at {sample_rate} Hz, frames={num_frames}")

def _apply_vad_parameters(self) -> None:
vad_ctx = self._vad_ctx
if vad_ctx is None:
return
# Per-parameter try/except so a single SDK rejection doesn't silently
# drop the remaining params.
pending: list[tuple[VadParameter, float | None]] = [
(VadParameter.SpeechHoldDuration, self._pending_speech_hold_duration),
(VadParameter.MinimumSpeechDuration, self._pending_minimum_speech_duration),
(VadParameter.Sensitivity, self._pending_sensitivity),
]
for parameter, value in pending:
if value is None:
continue
try:
vad_ctx.set_parameter(parameter, value)
except Exception as e: # noqa: BLE001 - one bad param shouldn't drop the others
logger.warning(f"Quail VAD parameter {parameter} application failed: {e}")

def set_sample_rate(self, sample_rate: int) -> None:
"""Set the sample rate. Recreates the SDK processor if the rate changed.

Expand Down Expand Up @@ -305,10 +262,12 @@ def voice_confidence(self, buffer: bytes) -> float:
:meth:`num_frames_required` samples.

Returns:
``1.0`` if the VAD reports speech, ``0.0`` otherwise. Returns
``0.0`` if the processor is not yet initialized (i.e.
:meth:`set_sample_rate` has not run), if the buffer size does not
match the expected window, or if an SDK inference error occurs.
The model's raw speech probability in ``[0.0, 1.0]``. The base
:class:`VADAnalyzer` compares this against ``VADParams.confidence``
to decide speech. Returns ``0.0`` if the processor is not yet
initialized (i.e. :meth:`set_sample_rate` has not run), if the buffer
size does not match the expected window, or if an SDK inference error
occurs.
"""
if self._processor is None or self._vad_ctx is None or self._in_f32 is None:
return 0.0
Expand All @@ -330,7 +289,10 @@ def voice_confidence(self, buffer: bytes) -> float:
# Successful inference re-arms the error latch so a fresh error
# after a recovery is reported at ERROR rather than buried at DEBUG.
self._inference_error_logged = False
return 1.0 if self._vad_ctx.is_speech_detected() else 0.0
# Raw model probability (no SDK post-processing); clamp defensively
# to the [0.0, 1.0] the VADAnalyzer state machine expects.
probability = float(self._vad_ctx.raw_vad_probability())
return max(0.0, min(1.0, probability))
except Exception as e: # noqa: BLE001 - keep the pipeline alive on SDK errors
if not self._inference_error_logged:
logger.error(f"Quail VAD inference error: {e}")
Expand Down
28 changes: 28 additions & 0 deletions src/pipecat/metrics/metrics.py
Original file line number Diff line number Diff line change
Expand Up @@ -155,3 +155,31 @@ class SmartTurnMetricsData(TurnMetricsData):

inference_time_ms: float = 0.0
server_total_time_ms: float = 0.0


class AICAudioQualityMetricsData(MetricsData):
"""Audio-quality scores from the ai-coustics Tyto analysis model.

Each score is in the range ``0.0``–``1.0``. For every field **except**
``speaker_loudness``, lower values indicate less problematic audio; the
scores predict the likelihood that the analyzed audio degrades downstream
models (speech-to-text, VAD, turn-taking, speech-to-speech). Emitted by
:class:`pipecat.processors.audio.aic_tyto_analyzer.AICTytoAnalyzer`.

Parameters:
risk_score: Overall likelihood of downstream-model degradation.
speaker_reverb: Reverberation on the primary speaker.
speaker_loudness: Primary-speaker loudness (not a lower-is-better score).
interfering_speech: Presence of competing/background speech.
media_speech: Presence of speech from media (TV, music, etc.).
noise: Non-speech background noise.
packet_loss: Audio artifacts consistent with network packet loss.
"""

risk_score: float
speaker_reverb: float
speaker_loudness: float
interfering_speech: float
media_speech: float
noise: float
packet_loss: float
Loading
Loading