diff --git a/examples/README.md b/examples/README.md index c9bb52e28..76355619b 100644 --- a/examples/README.md +++ b/examples/README.md @@ -79,6 +79,7 @@ Unless noted otherwise, every example below uses `braintrust.auto_instrument()`. | `openai_agents/` | OpenAI Agents SDK `Runner` running an agent | | `openrouter/` | OpenRouter chat completion routed to OpenAI | | `otel/` | OpenTelemetry interop — `BraintrustSpanProcessor`, filtering, distributed tracing | +| [pipecat/](pipecat/) | Cascade and realtime voice agents with tool calls and recordings — uses `setup_pipecat()` | | `pydantic_ai/` | Pydantic AI agent run inside a `start_span` for a permalink | | `strands/` | Strands `Agent` against `gpt-4o-mini` | | `temporal/` | Distributed Temporal workflow tracing via `BraintrustPlugin` | diff --git a/examples/pipecat/README.md b/examples/pipecat/README.md new file mode 100644 index 000000000..3693a0319 --- /dev/null +++ b/examples/pipecat/README.md @@ -0,0 +1,28 @@ +# Pipecat voice tracing + +Two small voice agents using real OpenAI services and the local Braintrust SDK: + +| Example | Pipeline | +| --- | --- | +| `cascade.py` | Speech → OpenAI transcription → LLM + order lookup → synthesized speech | +| `realtime.py` | Speech → OpenAI Realtime + order lookup → speech | + +Requires Python 3.11–3.13, `uv`, `OPENAI_API_KEY`, and `BRAINTRUST_API_KEY` in your environment. From the repository root, run either command: + +```sh +uv run --project examples/pipecat python examples/pipecat/cascade.py +uv run --project examples/pipecat python examples/pipecat/realtime.py +``` + +Each command installs its dependencies, runs one conversation, and prints a trace link in the `example-pipecat` project. OpenAI usage is billable. + +The bundled `order.wav` (about 82 KiB) asks “Where is my order number one zero four two?” `common.py` supplies file input and silent output so no microphone, speaker, browser, or phone setup is needed. The local `lookup_order` tool returns a fictional delivery date. Listen to the captured audio in the trace. + +Instrumentation is enabled with: + +```python +logger = braintrust.init_logger(project="example-pipecat") +setup_pipecat(capture_audio_attachments=True) +``` + +Pipecat metrics are enabled in `PipelineParams`. The trace includes turns, model calls, tool execution, metrics, and Ogg audio attachments. Set `capture_audio_attachments=False` to trace without recording audio. diff --git a/examples/pipecat/cascade.py b/examples/pipecat/cascade.py new file mode 100644 index 000000000..2c5aa3169 --- /dev/null +++ b/examples/pipecat/cascade.py @@ -0,0 +1,62 @@ +"""Audio → transcription → model/tool → synthesized speech, traced by Braintrust.""" + +import asyncio +import os + +import braintrust +from braintrust.integrations.pipecat import setup_pipecat +from common import INSTRUCTIONS, RecordedInput, SilentOutput, order_context, run_conversation + +# Pipecat requires Python 3.11+ and is installed by this example, not the shared lint environment. +# pylint: disable=import-error +from pipecat.audio.vad.silero import SileroVADAnalyzer +from pipecat.pipeline.pipeline import Pipeline +from pipecat.processors.aggregators.llm_response_universal import LLMContextAggregatorPair, LLMUserAggregatorParams +from pipecat.services.openai.llm import OpenAILLMService +from pipecat.services.openai.stt import OpenAISTTService +from pipecat.services.openai.tts import OpenAITTSService +from pipecat.transports.base_transport import TransportParams + + +async def main(): + logger = braintrust.init_logger(project="example-pipecat") + setup_pipecat(capture_audio_attachments=True) + + api_key = os.environ["OPENAI_API_KEY"] + transport = TransportParams(audio_in_enabled=True, audio_out_enabled=True) + source, output = RecordedInput(transport), SilentOutput(transport) + aggregators = LLMContextAggregatorPair( + order_context(), + user_params=LLMUserAggregatorParams(vad_analyzer=SileroVADAnalyzer()), + ) + pipeline = Pipeline( + [ + source, + OpenAISTTService(api_key=api_key), + aggregators.user(), + OpenAILLMService( + api_key=api_key, + settings=OpenAILLMService.Settings( + model="gpt-4.1-mini", + system_instruction=INSTRUCTIONS, + ), + ), + OpenAITTSService( + api_key=api_key, + settings=OpenAITTSService.Settings( + model="gpt-4o-mini-tts", + voice="alloy", + ), + ), + output, + aggregators.assistant(), + ] + ) + with logger.start_span(name="cascade") as trace: + await run_conversation(pipeline, aggregators, source) + await asyncio.to_thread(logger.flush) + print(f"Trace: {trace.link()}") + + +if __name__ == "__main__": + asyncio.run(main()) diff --git a/examples/pipecat/common.py b/examples/pipecat/common.py new file mode 100644 index 000000000..339cdec9a --- /dev/null +++ b/examples/pipecat/common.py @@ -0,0 +1,115 @@ +"""Example application plumbing: play a WAV instead of opening a microphone.""" + +import asyncio +import wave +from pathlib import Path + +# Pipecat requires Python 3.11+ and is installed by this example, not the shared lint environment. +# pylint: disable=import-error +from pipecat.adapters.schemas.function_schema import FunctionSchema +from pipecat.adapters.schemas.tools_schema import ToolsSchema +from pipecat.audio.utils import create_stream_resampler +from pipecat.frames.frames import InputAudioRawFrame, LLMRunFrame +from pipecat.pipeline.worker import PipelineParams, PipelineWorker +from pipecat.processors.aggregators.llm_context import LLMContext +from pipecat.transports.base_input import BaseInputTransport +from pipecat.transports.base_output import BaseOutputTransport +from pipecat.workers.runner import WorkerRunner + + +async def lookup_order(params): + """A fictional local order database; replace with your application's tool.""" + await params.result_callback({"order_id": params.arguments["order_id"], "delivery": "Friday"}) + + +def order_context(): + return LLMContext( + tools=ToolsSchema( + standard_tools=[ + FunctionSchema( + name="lookup_order", + description="Look up an order's delivery date", + properties={"order_id": {"type": "string"}}, + required=["order_id"], + handler=lookup_order, + ) + ] + ), + ) + + +INSTRUCTIONS = ( + "Speak English. Greet the caller briefly and ask for their order number. " + "When they supply it, always call lookup_order. " + "After its result, say only: Your order arrives Friday." +) + + +class RecordedInput(BaseInputTransport): + async def start(self, frame): + await super().start(frame) + await self.set_transport_ready(frame) + + async def play(self, sample_rate): + with wave.open(str(Path(__file__).with_name("order.wav"))) as audio: + pcm = audio.readframes(audio.getnframes()) + pcm = await create_stream_resampler().resample(pcm, 16000, sample_rate) + # Include silence around the request so native VAD determines its boundaries. + pcm = b"\0" * sample_rate + pcm + b"\0" * (sample_rate * 2) + frame_bytes = sample_rate // 50 * 2 + for offset in range(0, len(pcm), frame_bytes): + await self.push_audio_frame(InputAudioRawFrame(pcm[offset : offset + frame_bytes], sample_rate, 1)) + await asyncio.sleep(0.02) + + +class SilentOutput(BaseOutputTransport): + """Accept agent audio without requiring an audio device; listen in the trace.""" + + async def start(self, frame): + await super().start(frame) + await self.set_transport_ready(frame) + + async def write_audio_frame(self, frame): + return True + + +async def run_conversation(pipeline, aggregators, source, *, realtime=False): + sample_rate = 24000 if realtime else 16000 + worker = PipelineWorker( + pipeline, + params=PipelineParams( + audio_in_sample_rate=sample_rate, + audio_out_sample_rate=24000, + enable_metrics=True, + enable_usage_metrics=True, + ), + ) + greeted, finished = asyncio.Event(), asyncio.Event() + + @aggregators.assistant().event_handler("on_assistant_turn_stopped") + async def assistant_turn(aggregator, message): + print(f"Agent: {message.content}", flush=True) + if "friday" in message.content.lower(): + finished.set() + else: + greeted.set() + + @aggregators.user().event_handler("on_user_turn_message_added") + async def user_turn(aggregator, message): + print(f"Caller: {message.content}", flush=True) + + @worker.event_handler("on_pipeline_started") + async def started(worker, frame): + await worker.queue_frame(LLMRunFrame()) + await asyncio.wait_for(greeted.wait(), 30) + await source.play(sample_rate) + + task = asyncio.create_task(WorkerRunner(handle_sigint=False).run(worker)) + try: + await asyncio.wait_for(finished.wait(), 60) + await worker.stop_when_done() + await task + finally: + if not task.done(): + await worker.cancel() + await task diff --git a/examples/pipecat/order.wav b/examples/pipecat/order.wav new file mode 100644 index 000000000..d7230f1ec Binary files /dev/null and b/examples/pipecat/order.wav differ diff --git a/examples/pipecat/pyproject.toml b/examples/pipecat/pyproject.toml new file mode 100644 index 000000000..7a322a5df --- /dev/null +++ b/examples/pipecat/pyproject.toml @@ -0,0 +1,12 @@ +[project] +name = "braintrust-pipecat-example" +version = "0.1.0" +description = "Trace cascade and realtime Pipecat voice agents" +requires-python = ">=3.11,<3.14" +dependencies = [ + "braintrust[audio]", + "pipecat-ai[openai,silero]==1.12.0", +] + +[tool.uv.sources] +braintrust = { path = "../../py", editable = true } diff --git a/examples/pipecat/realtime.py b/examples/pipecat/realtime.py new file mode 100644 index 000000000..3d38e93b0 --- /dev/null +++ b/examples/pipecat/realtime.py @@ -0,0 +1,50 @@ +"""OpenAI Realtime speech-to-speech with a local tool, traced by Braintrust.""" + +import asyncio +import os + +import braintrust +from braintrust.integrations.pipecat import setup_pipecat +from common import INSTRUCTIONS, RecordedInput, SilentOutput, order_context, run_conversation + +# Pipecat requires Python 3.11+ and is installed by this example, not the shared lint environment. +# pylint: disable=import-error +from pipecat.pipeline.pipeline import Pipeline +from pipecat.processors.aggregators.llm_response_universal import LLMContextAggregatorPair +from pipecat.services.openai.realtime import events +from pipecat.services.openai.realtime.llm import OpenAIRealtimeLLMService +from pipecat.transports.base_transport import TransportParams + + +async def main(): + logger = braintrust.init_logger(project="example-pipecat") + setup_pipecat(capture_audio_attachments=True) + + transport = TransportParams(audio_in_enabled=True, audio_out_enabled=True) + source, output = RecordedInput(transport), SilentOutput(transport) + aggregators = LLMContextAggregatorPair(order_context()) + model = OpenAIRealtimeLLMService( + api_key=os.environ["OPENAI_API_KEY"], + settings=OpenAIRealtimeLLMService.Settings( + model="gpt-realtime", + system_instruction=INSTRUCTIONS, + session_properties=events.SessionProperties( + audio=events.AudioConfiguration( + input=events.AudioInput( + transcription=events.InputAudioTranscription(model="gpt-4o-mini-transcribe", language="en"), + turn_detection=events.TurnDetection(), + ), + output=events.AudioOutput(voice="alloy"), + ) + ), + ), + ) + pipeline = Pipeline([source, aggregators.user(), model, output, aggregators.assistant()]) + with logger.start_span(name="realtime") as trace: + await run_conversation(pipeline, aggregators, source, realtime=True) + await asyncio.to_thread(logger.flush) + print(f"Trace: {trace.link()}") + + +if __name__ == "__main__": + asyncio.run(main()) diff --git a/py/noxfile.py b/py/noxfile.py index dc06bb9d1..a9eed96d3 100644 --- a/py/noxfile.py +++ b/py/noxfile.py @@ -467,7 +467,7 @@ def test_pipecat(session, version): # loaded from Nox's virtualenv when it is beneath the current directory. _run_tests( session, - f"{INTEGRATION_DIR}/pipecat/test_pipecat.py", + f"{INTEGRATION_DIR}/pipecat" if version == LATEST else f"{INTEGRATION_DIR}/pipecat/test_pipecat.py", version=version, run_from_temp_dir=True, ) @@ -786,6 +786,13 @@ def test_pytest_plugin(session, version): _run_tests(session, f"{WRAPPER_DIR}/pytest_plugin/test_plugin.py") +@nox.session() +def test_audio(session): + _install_test_deps(session) + _install_group_locked(session, "test-audio") + _run_tests(session, "braintrust/_audio") + + @nox.session() def test_core(session): _install_test_deps(session) @@ -956,6 +963,8 @@ def _run_core_tests(session): SRC_DIR, ignore_paths=[ WRAPPER_DIR, + "braintrust/_audio/test_recording.py", + "braintrust/_audio/test_segments.py", *_integration_subdirs_to_ignore(), CONTRIB_DIR, DEVSERVER_DIR, diff --git a/py/pyproject.toml b/py/pyproject.toml index fdfbe9af3..e67947b82 100644 --- a/py/pyproject.toml +++ b/py/pyproject.toml @@ -45,6 +45,7 @@ braintrust = "braintrust.wrappers.pytest_plugin.plugin" braintrust = "braintrust.integrations.harbor:HarborPlugin" [project.optional-dependencies] +audio = ["numpy>=1.26", "soundfile>=0.13.1"] cli = ["boto3", "python-dotenv", "uv", "starlette", "uvicorn"] # TODO: remove the doc extra in the next major version. doc = [] @@ -195,8 +196,15 @@ test-livekit-agents = [ "opentelemetry-sdk<1.39", ] +test-audio = [ + {include-group = "test"}, + "numpy>=1.26", + "soundfile>=0.13.1", +] + test-pipecat = [ {include-group = "test"}, + "soundfile>=0.13.1", # pipecat-ai 1.3.0 imports websockets but does not install it transitively. "websockets==15.0.1", ] diff --git a/py/src/braintrust/_audio/__init__.py b/py/src/braintrust/_audio/__init__.py new file mode 100644 index 000000000..33cb43c93 --- /dev/null +++ b/py/src/braintrust/_audio/__init__.py @@ -0,0 +1,33 @@ +"""Internal audio API shared by integrations; codecs remain optional and lazy.""" + +from .alignment import InputRanges +from .attachments import UploadFailed, prepare_recording, upload_recording +from .budget import source_budget +from .export import AlignmentPublisher, segment_descriptor +from .jobs import RecordingJobs +from .options import RecordingOptions +from .recording import encode_audio +from .segments import SegmentedRecording +from .timeline import ClipTimeline, ms_to_samples, pcm_bytes_to_ms, samples_to_ms +from .worker import RecordingBusy, encode_in_worker + + +__all__ = [ + "AlignmentPublisher", + "ClipTimeline", + "InputRanges", + "RecordingBusy", + "RecordingJobs", + "RecordingOptions", + "SegmentedRecording", + "UploadFailed", + "encode_audio", + "encode_in_worker", + "ms_to_samples", + "pcm_bytes_to_ms", + "prepare_recording", + "samples_to_ms", + "segment_descriptor", + "source_budget", + "upload_recording", +] diff --git a/py/src/braintrust/_audio/alignment.py b/py/src/braintrust/_audio/alignment.py new file mode 100644 index 000000000..d7d6b5a5f --- /dev/null +++ b/py/src/braintrust/_audio/alignment.py @@ -0,0 +1,204 @@ +"""Bounded sample ranges and recording selections shared by voice integrations.""" + +from collections import deque +from collections.abc import Iterator +from dataclasses import dataclass, field + +from .constants import MAX_PENDING_RANGES +from .timeline import ClipTimeline, ms_to_samples, samples_to_ms +from .types import SegmentInterval, Selection + + +def merge_ranges(ranges: list[list[int]]) -> list[list[int]]: + result = [] + for start, end in sorted(ranges): + if end <= start: + continue + if result and start <= result[-1][1]: + result[-1][1] = max(result[-1][1], end) + else: + result.append([start, end]) + return result + + +def resolve_selections( + ranges: list[list[int]], + segments: list[SegmentInterval], + resolved: list[list[int]], + recording_span_id: str, + channel: int, +) -> tuple[list[Selection], list[list[int]]]: + """Map call samples into ready files; retain ranges whose files are pending.""" + selections = [] + merged = merge_ranges(ranges) + pending = [] + for segment in segments: + if segment["state"] != "ready": + continue + lower, upper = ms_to_samples(segment["start_ms"]), ms_to_samples(segment["end_ms"]) + for start, end in merged: + start, end = max(start, lower), min(end, upper) + if end > start: + selections.append( + { + "recording_span_id": recording_span_id, + "recording_id": segment["id"], + "start_offset_ms": samples_to_ms(start - lower), + "end_offset_ms": samples_to_ms(end - lower), + "channel_index": channel, + } + ) + for start, end in ranges: + for lower, upper in resolved: + if lower > start: + pending.append([start, min(lower, end)]) + start = max(start, upper) if lower < end else end + if start >= end: + break + if start < end: + pending.append([start, end]) + return selections, pending + + +class InputRanges: + def __init__(self): + self.runs = deque() + self.size = 0 + + def append(self, size, interval): + self.runs.append([size, interval, size, 0]) + self.size += size + + def trim(self, retained): + while self.size > retained and self.runs: + run = self.runs[0] + count = min(self.size - retained, run[0]) + run[0] -= count + run[3] += count + self.size -= count + if not run[0]: + self.runs.popleft() + + def drain_mapped(self): + pieces = [] + position = 0 + for count, interval, original, offset in self.runs: + if interval: + length = interval["end"] - interval["start"] + pieces.append( + ( + position, + position + count, + interval["start"] + round(offset * length / original), + interval["start"] + round((offset + count) * length / original), + ) + ) + position += count + self.runs.clear() + self.size = 0 + return pieces + + def drain(self): + return merge_ranges([[a, b] for _, _, a, b in self.drain_mapped()]) + + +@dataclass +class _SelectionOwner: + channel: int + ranges: list[list[int]] = field(default_factory=list) + published: list[Selection] = field(default_factory=list) + + +@dataclass +class _Clip: + timeline: ClipTimeline = field(default_factory=ClipTimeline) + descriptor: dict | None = None + + +class Alignment: + """Resolve recording positions into metadata updates, without SDK objects or I/O.""" + + def __init__(self, recording_span_id: str): + self.recording_span_id = recording_span_id + self._owners: dict[str, _SelectionOwner] = {} + self._clips: dict[str, _Clip] = {} + self._dirty_clips: set[str] = set() + self.count = 0 + self.omitted = 0 + self.pending_owners = set() + + def add(self, span_id: str, ranges: list[list[int]], channel: int) -> None: + item = self._owners.setdefault(span_id, _SelectionOwner(channel)) + for interval in ranges: + if item.ranges and item.ranges[-1][0] <= interval[0] <= item.ranges[-1][1]: + item.ranges[-1][1] = max(item.ranges[-1][1], interval[1]) + continue + if self.count >= MAX_PENDING_RANGES: + self.omitted += 1 + continue + item.ranges.append(list(interval)) + self.count += 1 + + if item.ranges: + self.pending_owners.add(span_id) + + def begin_clip(self, span_id: str) -> None: + self._clips[span_id] = _Clip() + + def add_clip_range( + self, span_id: str, start_ms: float, end_ms: float, call_start_ms: float, call_end_ms: float + ) -> None: + clip = self._clips.get(span_id) + if clip is not None: + clip.timeline.add(start_ms, end_ms, call_start_ms, call_end_ms) + self._dirty_clips.add(span_id) + + def set_clip(self, span_id: str, descriptor: dict) -> None: + clip = self._clips.get(span_id) + if clip is not None: + clip.descriptor = descriptor + self._dirty_clips.add(span_id) + + def clip_updates(self, origin: float | None) -> Iterator[tuple[str, dict]]: + if origin is None: + return + for span_id in tuple(self._dirty_clips): + clip = self._clips[span_id] + if clip.descriptor is None: + continue + timeline = clip.timeline.descriptor(origin) + if timeline != clip.descriptor.get("timeline"): + yield span_id, {"audio.recordings": [{**clip.descriptor, "timeline": timeline}]} + # Iteration resumes only after the publisher logs successfully. + clip.descriptor["timeline"] = timeline + self._dirty_clips.discard(span_id) + + def selection_updates(self, segments: list[SegmentInterval]) -> Iterator[tuple[str, dict]]: + resolved = merge_ranges( + [ + [ms_to_samples(s["start_ms"]), ms_to_samples(s["end_ms"])] + for s in segments + if s["state"] in {"ready", "omitted"} + ] + ) + for span_id in tuple(self.pending_owners): + state = self._owners[span_id] + channel, ranges = state.channel, state.ranges + selections = list(state.published) + additions, pending = resolve_selections(ranges, segments, resolved, self.recording_span_id, channel) + selections.extend(additions) + selections = [dict(items) for items in dict.fromkeys(tuple(s.items()) for s in selections)] + if selections and selections != state.published: + yield ( + span_id, + { + "audio.selections": selections, + "audio.selection": selections[0] if len(selections) == 1 else None, + }, + ) + state.published = selections + # Keep unresolved publication work intact if logging raises at yield. + self.count -= len(ranges) - len(pending) + ranges[:] = pending + if not pending: + self.pending_owners.discard(span_id) diff --git a/py/src/braintrust/_audio/attachments.py b/py/src/braintrust/_audio/attachments.py new file mode 100644 index 000000000..986c839d6 --- /dev/null +++ b/py/src/braintrust/_audio/attachments.py @@ -0,0 +1,34 @@ +"""Prepare encoded recordings for the standard attachment exporter.""" + +import asyncio + +from braintrust.logger import Attachment + +from .worker import await_background + + +def prepare_recording(name, function, *args): + encoded = function(*args) + if encoded is not None: + encoded["attachment"] = Attachment( + data=encoded["data"], + filename=f"{name}.{encoded['extension']}", + content_type=encoded["mime_type"], + ) + return encoded + + +class UploadFailed(RuntimeError): + """The attachment did not reach confirmed upload completion.""" + + +async def upload_recording(encoded): + def upload(): + try: + status = encoded["attachment"].upload() + except Exception as error: + raise UploadFailed("upload_failed") from error + if status.get("upload_status") != "done": + raise UploadFailed("upload_failed") + + await await_background(asyncio.to_thread(upload)) diff --git a/py/src/braintrust/_audio/budget.py b/py/src/braintrust/_audio/budget.py new file mode 100644 index 000000000..4d4b1dccd --- /dev/null +++ b/py/src/braintrust/_audio/budget.py @@ -0,0 +1,26 @@ +"""Conservative process-wide bound on retained recording source bytes.""" + +import threading + + +class ByteBudget: + def __init__(self, maximum=64 * 1024 * 1024): + self.maximum = maximum + self.used = 0 + self.lock = threading.Lock() + + def reserve(self, size): + with self.lock: + if self.used + size > self.maximum: + return False + self.used += size + return True + + def release(self, size): + with self.lock: + self.used -= size + if self.used < 0: + raise RuntimeError("Unbalanced audio budget release") + + +source_budget = ByteBudget() diff --git a/py/src/braintrust/_audio/conftest.py b/py/src/braintrust/_audio/conftest.py new file mode 100644 index 000000000..3a0e539de --- /dev/null +++ b/py/src/braintrust/_audio/conftest.py @@ -0,0 +1,10 @@ +"""Audio tests retain attachments locally instead of contacting Braintrust storage.""" + +import pytest +from braintrust.logger import Attachment + + +@pytest.fixture(autouse=True) +def local_attachment_uploads(monkeypatch): + # Override in lifecycle tests to hold/fail uploads. Keep real encoding and payloads. + monkeypatch.setattr(Attachment, "upload", lambda self: {"upload_status": "done"}) diff --git a/py/src/braintrust/_audio/constants.py b/py/src/braintrust/_audio/constants.py new file mode 100644 index 000000000..8b4cc7413 --- /dev/null +++ b/py/src/braintrust/_audio/constants.py @@ -0,0 +1,11 @@ +"""Internal safety bounds; recording policy remains configurable in RecordingOptions.""" + +# Leave room for output writes that arrive behind the input sample clock. +SEGMENT_HEADROOM_MS = 1000 +# One encoding job and one queued segment; a slow exporter cannot grow a backlog. +MAX_SEGMENT_JOBS = 2 +# Independent object-count guard for unusually small PCM packets in an active buffer. +MAX_BUFFERED_PACKETS = 100000 +# Bound unresolved alignment work, not the lifetime number of turns or frames. +MAX_PENDING_RANGES = 32000 +MAX_OUTPUT_CONTEXTS = 16000 diff --git a/py/src/braintrust/_audio/export.py b/py/src/braintrust/_audio/export.py new file mode 100644 index 000000000..b65169f1d --- /dev/null +++ b/py/src/braintrust/_audio/export.py @@ -0,0 +1,95 @@ +"""Serialize manifests and publish alignment updates through the SDK.""" + +from collections.abc import Iterable + +from braintrust.logger import Span + +from .alignment import Alignment +from .constants import MAX_OUTPUT_CONTEXTS +from .recording import CallRecording +from .segment import AudioSegment + + +def segment_descriptor(segment: AudioSegment, span_id: str, sources: list[dict]) -> dict: + descriptor = { + "id": segment.segment_id, + "recording_group_id": "call", + "state": segment.state, + "sources": sources, + "timeline": { + "origin_unix_ms": segment.origin_unix_ms, + "recording_start_offset_ms": segment.start_ms, + "basis": "input_sample_clock_and_output_write_observation", + }, + } + if segment.state == "ready": + descriptor.update(segment.metadata) + descriptor["attachment"] = {"span_id": span_id, "ref": f"/input/audio/{segment.segment_id}"} + elif segment.state == "omitted": + descriptor["reason"] = segment.reason + descriptor["gaps"] = [ + { + "start_offset_ms": segment.start_ms, + "end_offset_ms": segment.end_ms, + "reason": segment.reason, + } + ] + return descriptor + + +class AlignmentPublisher: + """Adapt observed span ownership to the pure alignment model.""" + + def __init__(self, root: Span, recording: CallRecording): + self.root = root + self.recording = recording + self.alignment = Alignment(root.span_id) + self._spans: dict[str, Span] = {} + self._contexts: dict[str, list[Span]] = {} + + def add(self, owner: Span, ranges: list[list[int]], channel: int) -> None: + if not self.recording.reason: + self._spans[owner.span_id] = owner + self.alignment.add(owner.span_id, ranges, channel) + + def begin_output(self, context: str, owners: list[Span]) -> None: + if len(self._contexts) >= MAX_OUTPUT_CONTEXTS: + self.alignment.omitted += 1 + return + self._contexts[context] = owners + self.alignment.begin_clip(owners[0].span_id) + + def output_owners(self, context: str) -> list[Span]: + return self._contexts.get(context, []) + + def end_output(self, context: str) -> None: + self._contexts.pop(context, None) + + def add_clip_range( + self, owner: Span, start_ms: float, end_ms: float, call_start_ms: float, call_end_ms: float + ) -> None: + self.alignment.add_clip_range(owner.span_id, start_ms, end_ms, call_start_ms, call_end_ms) + + def publish_clip(self, owner: Span, descriptor: dict) -> None: + self._spans[owner.span_id] = owner + self.alignment.set_clip(owner.span_id, descriptor) + self.publish_clips() + + def _publish(self, updates: Iterable[tuple[str, dict]]) -> None: + for span_id, metadata in updates: + self._spans[span_id].log(metadata=metadata) + + def publish_clips(self) -> None: + self._publish(self.alignment.clip_updates(getattr(self.recording, "origin_unix_ms", None))) + + def publish(self) -> None: + self.publish_clips() + segments = getattr(self.recording, "completed", None) + if segments is None: + segments = ( + [{"id": "call", "start_ms": 0, "end_ms": max(self.recording.ends), "state": "ready"}] + if not self.recording.reason and self.recording.packets + else [] + ) + self._publish(self.alignment.selection_updates(segments)) + self.root.log(metadata={"braintrust.alignment.ranges_omitted": self.alignment.omitted}) diff --git a/py/src/braintrust/_audio/exporter.py b/py/src/braintrust/_audio/exporter.py new file mode 100644 index 000000000..2f415aa21 --- /dev/null +++ b/py/src/braintrust/_audio/exporter.py @@ -0,0 +1,26 @@ +"""Execute a sealed segment without deciding recording policy or publishing spans.""" + +from .attachments import UploadFailed, prepare_recording, upload_recording +from .segment import AudioSegment +from .worker import encode_in_worker + + +class SegmentExporter: + def __init__(self, audio_format: str = "ogg"): + self.audio_format = audio_format + + async def export(self, segment: AudioSegment) -> None: + def encode(): + try: + return prepare_recording(segment.segment_id, segment.source.encode, self.audio_format) + finally: + # The worker owns PCM until native encoding has finished. + segment.clear() + + try: + encoded = await encode_in_worker(encode) + segment.encoded = encoded + await upload_recording(encoded) + segment.ready(encoded) + except Exception as error: # noqa: BLE001 - recording cannot stop speech + segment.omit("upload_failed" if isinstance(error, UploadFailed) else type(error).__name__) diff --git a/py/src/braintrust/_audio/jobs.py b/py/src/braintrust/_audio/jobs.py new file mode 100644 index 000000000..291b0b656 --- /dev/null +++ b/py/src/braintrust/_audio/jobs.py @@ -0,0 +1,59 @@ +"""Own background recording tasks and release their input after work stops.""" + +import asyncio +import logging +from collections.abc import Awaitable, Callable + + +class RecordingJobs: + def __init__(self): + self._tasks: set[asyncio.Task] = set() + + def __len__(self): + return sum(not task.done() for task in self._tasks) + + def submit( + self, + operation: Callable[[], Awaitable[None]], + release: Callable[[], None], + *, + after_release: Callable[[], Awaitable[None]] | None = None, + ) -> asyncio.Task: + released = False + + def release_once(): + nonlocal released + if not released: + released = True + release() + + async def run(): + try: + await operation() + finally: + release_once() + if after_release: + await after_release() + + task = asyncio.create_task(run()) + self._tasks.add(task) + + def completed(task): + # Also runs if cancelled before run() starts. The encoding worker + # does not complete cancellation until native work has stopped. + try: + release_once() + finally: + self._tasks.discard(task) + if not task.cancelled() and (error := task.exception()) is not None: + logging.getLogger(__name__).warning( + "Recording publication failed", exc_info=(type(error), error, error.__traceback__) + ) + + task.add_done_callback(completed) + return task + + async def drain(self) -> None: + if self._tasks: + # A cancelled waiter must not release another task's encoder input. + await asyncio.shield(asyncio.gather(*tuple(self._tasks), return_exceptions=True)) diff --git a/py/src/braintrust/_audio/options.py b/py/src/braintrust/_audio/options.py new file mode 100644 index 000000000..dc1967d23 --- /dev/null +++ b/py/src/braintrust/_audio/options.py @@ -0,0 +1,24 @@ +"""Public recording configuration; no codec or framework dependencies.""" + +import math +from dataclasses import dataclass + + +@dataclass(frozen=True) +class RecordingOptions: + """Per-call limits. A rotation threshold is not a total recording limit.""" + + # A minute of dual-mono 24 kHz PCM is about 5.5 MiB; duration usually + # rotates first. Byte rotation leaves headroom while an export is in flight. + # These are conservative defaults, not throughput guarantees. + segment_duration_seconds: float = 60 + max_duration_seconds: float = 1800 + max_buffer_bytes: int = 32 * 1024 * 1024 + flush_fraction: float = 0.5 + + def __post_init__(self): + for value in (self.segment_duration_seconds, self.max_duration_seconds): + if not math.isfinite(value) or value <= 0: + raise ValueError("recording durations must be positive and finite") + if self.max_buffer_bytes <= 0 or not 0 < self.flush_fraction < 1: + raise ValueError("recording buffer must be positive and flush_fraction between zero and one") diff --git a/py/src/braintrust/_audio/recording.py b/py/src/braintrust/_audio/recording.py new file mode 100644 index 000000000..a98da2388 --- /dev/null +++ b/py/src/braintrust/_audio/recording.py @@ -0,0 +1,189 @@ +"""Bounded PCM capture; mixing and encoding are called only from worker threads.""" + +import io +import math +import time +from dataclasses import dataclass +from enum import Enum, auto + +from .budget import source_budget +from .timeline import CALL_SAMPLE_RATE, ms_to_samples, pcm_bytes_to_ms +from .types import EncodedAudio + + +@dataclass(frozen=True) +class Packet: + channel: int + start_ms: float + pcm: bytes + sample_rate: int + channels: int + + +def encode_audio(chunks: list[bytes], sample_rate: int, channels: int, audio_format: str = "ogg") -> EncodedAudio: + import numpy as np + + started = time.perf_counter() + cpu_started = time.thread_time() + pcm = b"".join(chunks) + samples = np.frombuffer(pcm, dtype=" None: + if self.state is CaptureState.OPEN: + self.state = CaptureState.SEALED + + def stop(self, reason: str) -> None: + self.state = CaptureState.STOPPED + self.reason = reason + + def omit(self, reason: str) -> None: + self.stop(reason) + self.clear() + + def clear(self): + source_budget.release(self.bytes) + self.bytes = 0 + self.packets.clear() + + def capture(self, channel, pcm, sample_rate, channels, *, observed_ns=None, observed_unix_ms=None): + if self.state is not CaptureState.OPEN: + return + if channel not in (0, 1) or channels not in (1, 2) or not 8000 <= sample_rate <= 48000: + self.omit("unsupported_audio_format") + return + if len(pcm) % (2 * channels): + self.omit("invalid_pcm_frame") + return + if not pcm: + return + observed_ns = time.monotonic_ns() if observed_ns is None else observed_ns + if self.origin_ns is None: + self.origin_ns = observed_ns + self.origin_unix_ms = time.time() * 1000 if observed_unix_ms is None else observed_unix_ms + # This transport delivers continuous input, including silent samples. + # Anchor its first packet, then use sample duration rather than arrival + # jitter. Output is intermittent, so preserve pauses between writes. + start = ( + self.ends[channel] + if channel == 0 and self.started[channel] + else max((observed_ns - self.origin_ns) / 1e6, self.ends[channel]) + ) + duration = pcm_bytes_to_ms(len(pcm), sample_rate, channels) + # Transport frames can contain seconds of audio (e.g. shutdown silence). + # Bound retained bytes and timeline duration, not the transport's packetization. + if start + duration > self.max_duration_ms: + self.omit("duration_limit") + elif self.bytes + len(pcm) > self.max_bytes: + self.omit("capture_byte_limit") + elif len(self.packets) >= self.max_packets: + self.omit("packet_limit") + elif not source_budget.reserve(len(pcm)): + self.omit("process_capture_byte_limit") + else: + self.packets.append(Packet(channel, start, pcm, sample_rate, channels)) + self.bytes += len(pcm) + self.ends[channel] = start + duration + self.started[channel] = True + # Coordinates match encode() placement, including resampling rounding. + return { + "start": ms_to_samples(start), + "end": ms_to_samples(start) + ms_to_samples(duration), + } + + def encode(self, audio_format: str = "ogg") -> EncodedAudio | None: + if self.reason or not self.packets: + return None + + import numpy as np + + started = time.perf_counter() + cpu_started = time.thread_time() + sample_rate = CALL_SAMPLE_RATE + sample_count = math.ceil(max(self.ends) * sample_rate / 1000) + samples = np.zeros((sample_count, 2), dtype=np.int16) + for packet in self.packets: + source = np.frombuffer(packet.pcm, dtype=" int: + return self.source.bytes + + def clear(self) -> None: + self.source.clear() + + def ready(self, encoded: EncodedAudio) -> None: + self.encoded = encoded + self.metadata = {key: encoded[key] for key in ("mime_type", "duration_ms", "channel_count")} + self.state = "ready" + + def omit(self, reason: str) -> None: + self.state, self.reason = "omitted", reason + self.encoded = None + self.metadata.clear() + + def interval(self) -> SegmentInterval: + return {"id": self.segment_id, "start_ms": self.start_ms, "end_ms": self.end_ms, "state": self.state} diff --git a/py/src/braintrust/_audio/segments.py b/py/src/braintrust/_audio/segments.py new file mode 100644 index 000000000..b3fd8cc76 --- /dev/null +++ b/py/src/braintrust/_audio/segments.py @@ -0,0 +1,166 @@ +"""Progressively export independently decodable segments on one sample timeline.""" + +import asyncio +from collections.abc import Awaitable, Callable + +from .budget import source_budget +from .constants import MAX_BUFFERED_PACKETS, MAX_SEGMENT_JOBS, SEGMENT_HEADROOM_MS +from .exporter import SegmentExporter +from .jobs import RecordingJobs +from .options import RecordingOptions +from .recording import CallRecording, CaptureState, Packet +from .segment import AudioSegment +from .timeline import ms_to_samples + + +class SegmentedRecording(CallRecording): + """Capture remains synchronous; detached segments belong to encoder workers. + + Input samples and output observation time must both pass a cut before it is + exported. One second of headroom permits in-flight output and packet skew. + """ + + def __init__( + self, + *, + options: RecordingOptions | None = None, + on_segment: Callable[[AudioSegment], Awaitable[None]] | None = None, + on_pending: Callable[[AudioSegment], None] | None = None, + audio_format: str = "ogg", + enabled: bool = True, + ): + self.options = options or RecordingOptions() + super().__init__( + enabled=enabled, + max_bytes=self.options.max_buffer_bytes, + max_duration_ms=self.options.max_duration_seconds * 1000, + max_packets=MAX_BUFFERED_PACKETS, + ) + self.on_segment = on_segment + self.on_pending = on_pending + self.exporter = SegmentExporter(audio_format) + self.start_ms = 0.0 + self.sequence = 0 + self.segments: list[AudioSegment] = [] + self.jobs = RecordingJobs() + self.inflight = [] + self._finish_task = None + + @property + def completed(self): + return [segment.interval() for segment in self.segments if segment.state != "pending"] + + @property + def retained_bytes(self): + return self.bytes + sum(segment.bytes for segment in self.inflight) + + def omit(self, reason): + # Keep the valid prefix. Neither an error nor a limit erases prior audio. + self.stop(reason) + + def capture(self, channel, pcm, sample_rate, channels, *, observed_ns=None, observed_unix_ms=None): + if self.state is not CaptureState.OPEN: + return None + if self.retained_bytes + len(pcm) > self.max_bytes: + self.omit("capture_byte_limit") + return None + interval = super().capture( + channel, pcm, sample_rate, channels, observed_ns=observed_ns, observed_unix_ms=observed_unix_ms + ) + if interval is None: + return None + # Never rewrite audio that was already sealed/exported. + if interval["start"] < ms_to_samples(self.start_ms): + self.omit("capture_clock_discontinuity") + packet = self.packets.pop() + self.bytes -= len(packet.pcm) + source_budget.release(len(packet.pcm)) + return None + latest = self.packets[-1].start_ms + watermark = min(self.ends[0], latest) if self.started[0] else latest + watermark = max(self.start_ms, watermark - SEGMENT_HEADROOM_MS) + cut = self.start_ms + self.options.segment_duration_seconds * 1000 + if watermark >= cut: + self._rotate(cut) + elif self.bytes >= self.max_bytes * self.options.flush_fraction and watermark > self.start_ms: + self._rotate(watermark) + return interval + + def _rotate(self, cut): + if not self.packets: + return + if len(self.jobs) >= MAX_SEGMENT_JOBS: + self.omit("recording_export_capacity") + return + snapshot = CallRecording() + segment = AudioSegment(f"call-{self.sequence:04d}", self.start_ms, cut, self.origin_unix_ms, snapshot) + later = [] + for packet in self.packets: + size = len(packet.pcm) // (2 * packet.channels) + before = min(size, max(0, round((cut - packet.start_ms) * packet.sample_rate / 1000))) + if before: + data = packet.pcm if before == size else packet.pcm[: before * 2 * packet.channels] + snapshot.packets.append( + Packet(packet.channel, packet.start_ms - self.start_ms, data, packet.sample_rate, packet.channels) + ) + snapshot.bytes += len(data) + if before < size: + data = packet.pcm if not before else packet.pcm[before * 2 * packet.channels :] + later.append( + Packet( + packet.channel, + packet.start_ms + before / packet.sample_rate * 1000, + data, + packet.sample_rate, + packet.channels, + ) + ) + snapshot.ends = [cut - self.start_ms, cut - self.start_ms] + snapshot.seal() + self.packets = later + self.bytes -= segment.bytes + self.start_ms = cut + self.sequence += 1 + self.inflight.append(segment) + self.segments.append(segment) + + def release(): + segment.clear() + self.inflight.remove(segment) + + self.jobs.submit(lambda: self._export(segment), release) + if self.on_pending: + self.on_pending(segment) + + async def _export(self, segment: AudioSegment) -> None: + try: + await self.exporter.export(segment) + if segment.state == "omitted": + self.omit("segment_export_failed") + # Publishing is independent of upload success. A logging failure + # cannot invalidate an already uploaded segment. + if self.on_segment: + await self.on_segment(segment) + finally: + segment.encoded = None + + def release_idle_buffer(self): + """Release unsubmitted audio unless a final drain still owns it.""" + if self._finish_task is None or self._finish_task.done(): + self.clear() + + async def finish(self): + if self._finish_task is None: + self._finish_task = asyncio.create_task(self._finish()) + await asyncio.shield(self._finish_task) + + async def _finish(self): + self.seal() + await self.drain_exports() + if self.packets: + self._rotate(max(self.ends)) + await self.drain_exports() + + async def drain_exports(self): + """Wait for detached segments without sealing the active recording.""" + await self.jobs.drain() diff --git a/py/src/braintrust/_audio/test_dependencies.py b/py/src/braintrust/_audio/test_dependencies.py new file mode 100644 index 000000000..b2e134fb6 --- /dev/null +++ b/py/src/braintrust/_audio/test_dependencies.py @@ -0,0 +1,40 @@ +"""The shared capture path must work without optional codecs or frameworks.""" + +import subprocess +import sys + + +def test_capture_and_disabled_encode_without_optional_packages(): + code = """ +import importlib.abc +import sys +import braintrust + +class BlockOptional(importlib.abc.MetaPathFinder): + def find_spec(self, fullname, path=None, target=None): + if fullname.split(".")[0] in {"pipecat", "numpy", "soundfile"}: + raise AssertionError("Unexpected optional import: " + fullname) + +sys.meta_path.insert(0, BlockOptional()) +from braintrust._audio.recording import CallRecording +from braintrust._audio.alignment import InputRanges +from braintrust._audio import worker, RecordingOptions +from braintrust._audio.segments import SegmentedRecording +import asyncio +disabled = SegmentedRecording(enabled=False, options=RecordingOptions()) +assert disabled.capture(0, b"\\0\\0" * 480, 24000, 1) is None +asyncio.run(disabled.finish()) +assert disabled.retained_bytes == 0 +assert disabled.completed == [] +assert worker._executor is None +recording = CallRecording() +interval = recording.capture(0, b"\\0\\0" * 480, 24000, 1, observed_ns=0) +ranges = InputRanges() +ranges.append(960, interval) +assert ranges.drain() == [[0, 480]] +recording.clear() +assert recording.encode() is None +assert CallRecording(enabled=False).encode() is None +assert worker._executor is None +""" + subprocess.run([sys.executable, "-c", code], check=True, capture_output=True, text=True) diff --git a/py/src/braintrust/_audio/test_recording.py b/py/src/braintrust/_audio/test_recording.py new file mode 100644 index 000000000..70ea6483b --- /dev/null +++ b/py/src/braintrust/_audio/test_recording.py @@ -0,0 +1,64 @@ +import io +import unittest + +import numpy as np +import soundfile as sf # pylint: disable=import-error + +from .recording import CallRecording, encode_audio + + +def frame(samples, rate=16000): + return dict( + pcm=np.asarray(samples, dtype=" bool: """Set up Braintrust tracing for Pipecat ``PipelineWorker`` instances. @@ -35,13 +39,20 @@ def setup_pipecat( When arguments are omitted, the generic environment variables ``BRAINTRUST_CAPTURE_USER_AUDIO_ATTACHMENTS`` and ``BRAINTRUST_CAPTURE_AGENT_AUDIO_ATTACHMENTS`` apply. Explicit arguments - take precedence. + take precedence. Supported voice pipelines also emit native aggregator turns, + recording descriptors, and sample selections. ``audio_format`` selects Ogg/Opus + or PCM WAV for these recordings. ``recording_options`` configures progressive + segment duration, maximum recording duration, and retained PCM limits; install ``braintrust[audio]`` to encode them. """ + if audio_format not in {"ogg", "wav"}: + raise ValueError("audio_format must be ogg or wav") set_default_observer_options( capture_audio_attachments=capture_audio_attachments, capture_user_audio_attachments=capture_user_audio_attachments, capture_agent_audio_attachments=capture_agent_audio_attachments, trace_turns=trace_turns, + audio_format=audio_format, + recording_options=recording_options, ) span = current_span() diff --git a/py/src/braintrust/integrations/pipecat/_test_audio_cassettes.py b/py/src/braintrust/integrations/pipecat/_test_audio_cassettes.py new file mode 100644 index 000000000..e379fd012 --- /dev/null +++ b/py/src/braintrust/integrations/pipecat/_test_audio_cassettes.py @@ -0,0 +1,198 @@ +"""Lossless audio sidecars for test cassettes; never used by SDK instrumentation.""" + +import base64 +import copy +import shutil +import wave +from contextlib import contextmanager +from email.parser import BytesParser +from email.policy import default +from pathlib import Path +from tempfile import TemporaryDirectory + +from vcr.persisters.filesystem import CassetteNotFoundError +from vcr.serialize import deserialize, serialize + + +def write_audio(path, data, *, pcm=False): + path.parent.mkdir(parents=True, exist_ok=True) + if pcm: + # These OpenAI speech/realtime fixtures use mono 24 kHz PCM16. + with wave.Wave_write(str(path)) as audio: + audio.setparams((1, 2, 24000, 0, "NONE", "not compressed")) + audio.writeframes(data) + else: + path.write_bytes(data) + + +def read_audio(directory, reference): + path = directory / reference["audio_file"] + with wave.open(str(path)) as audio: + if reference.get("pcm"): + assert (audio.getnchannels(), audio.getsampwidth(), audio.getframerate()) == (1, 2, 24000) + expected_bytes = audio.getnframes() * audio.getnchannels() * audio.getsampwidth() + data = audio.readframes(audio.getnframes()) + assert len(data) == expected_bytes, "Truncated WAV fixture" + if not reference.get("pcm"): + data = path.read_bytes() + start = reference.get("offset_bytes", 0) + length = reference.get("length_bytes", len(data)) + assert 0 <= start <= start + length <= len(data), "Audio reference exceeds fixture file" + return data[start : start + length] + + +def publish_cassette(staged, target): + """Replace a complete fixture set; restore old audio if manifest replacement fails. + + Re-recording is sequential. This is rollback for ordinary I/O failures, + not a filesystem transaction resilient to process termination/power loss. + """ + audio = target.with_suffix(".audio") + replacement = staged.with_suffix(".audio") + replacement.mkdir(exist_ok=True) + backup = staged.parent / "previous.audio" + had_audio = audio.exists() + if had_audio: + audio.replace(backup) + installed = False + try: + replacement.replace(audio) + installed = True + staged.replace(target) + except BaseException: + if installed: + shutil.rmtree(audio) + if had_audio: + backup.replace(audio) + raise + + +@contextmanager +def staged_cassette(target): + target.parent.mkdir(parents=True, exist_ok=True) + with TemporaryDirectory(prefix=".recording-", dir=target.parent) as directory: + staged = Path(directory) / target.name + yield staged + publish_cassette(staged, target) + + +class AudioPersister: + """Use ordinary VCR serialization after restoring exact HTTP body bytes.""" + + @staticmethod + def load_cassette(cassette_path, serializer): + path = Path(cassette_path) + if not path.is_file(): + raise CassetteNotFoundError() + data = serializer.deserialize(path.read_text()) + for interaction in data["interactions"]: + body = interaction["request"].get("body") + if isinstance(body, dict) and "parts" in body: + interaction["request"]["body"] = b"".join( + part.encode() if isinstance(part, str) else read_audio(path.parent, part) for part in body["parts"] + ) + body = interaction["response"]["body"].get("string") + if isinstance(body, dict) and "audio_file" in body: + interaction["response"]["body"]["string"] = read_audio(path.parent, body) + return deserialize(serializer.serialize(data), serializer) + + @staticmethod + def save_cassette(cassette_path, cassette_dict, serializer): + with staged_cassette(Path(cassette_path)) as staged: + AudioPersister._save(staged, cassette_dict, serializer) + + @staticmethod + def _save(path, cassette_dict, serializer): + data = serializer.deserialize(serialize(copy.deepcopy(cassette_dict), serializer)) + directory = path.with_suffix(".audio") + for index, interaction in enumerate(data["interactions"]): + request = interaction["request"] + headers = {key.lower(): value for key, value in request["headers"].items()} + content_type = headers.get("content-type", [""])[0] + if "multipart/" in content_type: + body = request["body"] + message = BytesParser(policy=default).parsebytes( + f"Content-Type: {content_type}\r\n\r\n".encode() + body + ) + parts, remaining = [], body + for number, part in enumerate(message.iter_parts()): + if not part.get_filename(): + continue + payload = part.get_payload(decode=True) + # Preserve multipart headers, boundaries and text verbatim. + prefix, found, remaining = remaining.partition(payload) + assert found, "Multipart audio payload missing" + name = directory / f"request-{index}-{number}.wav" + write_audio(name, payload) + parts.extend([prefix.decode(), {"audio_file": f"{directory.name}/{name.name}"}]) + if parts: + request["body"] = {"parts": [*parts, remaining.decode()]} + response = interaction["response"] + headers = {key.lower(): value for key, value in response["headers"].items()} + content_type = headers.get("content-type", [""])[0] + if "audio/" in content_type or "application/octet-stream" in content_type: + # The test requests PCM explicitly; do not reinterpret arbitrary codecs. + import json + + assert json.loads(request["body"])["response_format"] == "pcm" + name = directory / f"response-{index}.wav" + write_audio(name, response["body"]["string"], pcm=True) + response["body"]["string"] = {"audio_file": f"{directory.name}/{name.name}", "pcm": True} + path.parent.mkdir(parents=True, exist_ok=True) + path.write_text(serializer.serialize(data)) + + +def save_websocket(path, endpoint, events): + with staged_cassette(path) as staged: + _save_websocket(staged, endpoint, events) + + +def _save_websocket(path, endpoint, events): + """Combine chunks into playable files, retaining each chunk's exact range.""" + import json + + events = copy.deepcopy(events) + directory = path.with_suffix(".audio") + streams = {} + for event in events: + message = event["message"] + kind = message.get("type") + field = ( + "audio" + if kind == "input_audio_buffer.append" + else "delta" + if kind == "response.output_audio.delta" + else None + ) + if field is None: + continue + key = "caller" if field == "audio" else message["response_id"] + if key not in streams: + name = "caller.wav" if key == "caller" else f"agent-{sum(k != 'caller' for k in streams):03d}.wav" + streams[key] = (name, bytearray()) + name, pcm = streams[key] + chunk = base64.b64decode(message[field], validate=True) + message[field] = { + "audio_file": f"{directory.name}/{name}", + "pcm": True, + "offset_bytes": len(pcm), + "length_bytes": len(chunk), + } + pcm.extend(chunk) + for name, pcm in streams.values(): + write_audio(directory / name, pcm, pcm=True) + path.parent.mkdir(parents=True, exist_ok=True) + path.write_text(json.dumps({"endpoint": endpoint, "events": events}, indent=2) + "\n") + + +def load_websocket(path): + import json + + data = json.loads(path.read_text()) + for event in data["events"]: + message = event["message"] + for field in ("audio", "delta"): + reference = message.get(field) + if isinstance(reference, dict) and "audio_file" in reference: + message[field] = base64.b64encode(read_audio(path.parent, reference)).decode() + return data diff --git a/py/src/braintrust/integrations/pipecat/_test_websocket.py b/py/src/braintrust/integrations/pipecat/_test_websocket.py new file mode 100644 index 000000000..33d4f3848 --- /dev/null +++ b/py/src/braintrust/integrations/pipecat/_test_websocket.py @@ -0,0 +1,127 @@ +"""Test-only, ordered recording/replay at Pipecat's real WebSocket boundary. + +Only transport I/O is replaced. Requests must match; native service handlers, +frames, aggregators and instrumentation are unchanged. No headers are stored. +""" + +import asyncio +import json +from pathlib import Path + +from ._test_audio_cassettes import load_websocket, save_websocket + + +class WebSocketCassette: + def __init__(self, path: Path, record=False): + self.path, self.record = path, record + saved = {} if record else load_websocket(path) + self.events = saved.get("events", []) + self.index = 0 + self.condition = asyncio.Condition() + self.closed = False + self.socket = None + self.client_ids = {} + self.failure = None + self.endpoint = saved.get("endpoint") + + async def connect(self, real_connect, **kwargs): + if self.record: + self.endpoint = kwargs["uri"] + self.socket = await real_connect(**kwargs) + else: + assert kwargs["uri"] == self.endpoint, "WebSocket endpoint changed; re-record cassette" + return self + + def _normalize(self, message): + value = json.loads(json.dumps(message)) + value.pop("event_id", None) + item = value.get("item", {}) + if item.get("id") in self.client_ids: + item["id"] = self.client_ids[item["id"]] + return value + + async def send(self, data): + message = json.loads(data) + if self.record: + self.events.append({"direction": "send", "message": message}) + await self.socket.send(data) + return + try: + async with self.condition: + await asyncio.wait_for( + self.condition.wait_for( + lambda: ( + self.closed + or self.index == len(self.events) + or self.events[self.index]["direction"] == "send" + ) + ), + 5, + ) + assert not self.closed and self.index < len(self.events), "Unexpected WebSocket send" + expected = self.events[self.index]["message"] + # Initial conversation item IDs are client-generated. Preserve + # their correspondence in incoming echoes, not just in matching. + if message.get("type") == "conversation.item.create": + actual_id = message["item"].get("id") + expected_id = expected.get("item", {}).get("id") + if actual_id and expected_id: + self.client_ids[actual_id] = expected_id + assert self._normalize(message) == self._normalize(expected), ( + f"WebSocket request mismatch at event {self.index}: {message.get('type')}" + ) + self.index += 1 + self.condition.notify_all() + except Exception as error: + self.failure = error + raise + + def __aiter__(self): + return self + + async def __anext__(self): + if self.record: + data = await self.socket.recv() + self.events.append({"direction": "receive", "message": json.loads(data)}) + return data + async with self.condition: + await self.condition.wait_for( + lambda: ( + self.closed + or (self.index < len(self.events) and self.events[self.index]["direction"] == "receive") + ) + ) + if self.closed: + raise StopAsyncIteration + message = self.events[self.index]["message"] + self.index += 1 + self.condition.notify_all() + # Rewrite only values identical to client IDs learned from outgoing + # requests. Provider-generated IDs and all payloads are preserved. + reverse = {recorded: actual for actual, recorded in self.client_ids.items()} + + def remap(value): + if isinstance(value, dict): + return {key: remap(item) for key, item in value.items()} + if isinstance(value, list): + return [remap(item) for item in value] + return reverse.get(value, value) if isinstance(value, str) else value + + await asyncio.sleep(0) + return json.dumps(remap(message)) + + async def close(self): + self.closed = True + if self.socket: + await self.socket.close() + async with self.condition: + self.condition.notify_all() + + def verify(self): + assert self.failure is None, f"WebSocket replay failed: {self.failure}" + if not self.record: + assert self.index == len(self.events), "WebSocket cassette was not fully consumed" + + def save(self): + assert self.record and self.events + save_websocket(self.path, self.endpoint, self.events) diff --git a/py/src/braintrust/integrations/pipecat/cassettes/latest/test_cascade_voice_conversation.audio/request-0-2.wav b/py/src/braintrust/integrations/pipecat/cassettes/latest/test_cascade_voice_conversation.audio/request-0-2.wav new file mode 100644 index 000000000..6d5c2aca4 Binary files /dev/null and b/py/src/braintrust/integrations/pipecat/cassettes/latest/test_cascade_voice_conversation.audio/request-0-2.wav differ diff --git a/py/src/braintrust/integrations/pipecat/cassettes/latest/test_cascade_voice_conversation.audio/response-3.wav b/py/src/braintrust/integrations/pipecat/cassettes/latest/test_cascade_voice_conversation.audio/response-3.wav new file mode 100644 index 000000000..ccdc4923d Binary files /dev/null and b/py/src/braintrust/integrations/pipecat/cassettes/latest/test_cascade_voice_conversation.audio/response-3.wav differ diff --git a/py/src/braintrust/integrations/pipecat/cassettes/latest/test_cascade_voice_conversation.yaml b/py/src/braintrust/integrations/pipecat/cassettes/latest/test_cascade_voice_conversation.yaml new file mode 100644 index 000000000..5ea45c89e --- /dev/null +++ b/py/src/braintrust/integrations/pipecat/cassettes/latest/test_cascade_voice_conversation.yaml @@ -0,0 +1,431 @@ +interactions: +- request: + body: + parts: + - "--61f6aee3212ab98ae8a4bcfba196d0cc\r\nContent-Disposition: form-data; name=\"model\"\r\n\r\ngpt-transcribe\r\n--61f6aee3212ab98ae8a4bcfba196d0cc\r\nContent-Disposition: + form-data; name=\"language\"\r\n\r\nen\r\n--61f6aee3212ab98ae8a4bcfba196d0cc\r\nContent-Disposition: + form-data; name=\"file\"; filename=\"audio.wav\"\r\nContent-Type: audio/wav\r\n\r\n" + - audio_file: test_cascade_voice_conversation.audio/request-0-2.wav + - "\r\n--61f6aee3212ab98ae8a4bcfba196d0cc--\r\n" + headers: + accept: + - application/json + accept-encoding: + - gzip, deflate + connection: + - keep-alive + content-length: + - '95975' + content-type: + - multipart/form-data; boundary=61f6aee3212ab98ae8a4bcfba196d0cc + host: + - api.openai.com + user-agent: + - AsyncOpenAI/Python 3.22.1 + x-stainless-arch: + - arm64 + x-stainless-async: + - async:asyncio + x-stainless-lang: + - python + x-stainless-os: + - MacOS + x-stainless-package-version: + - 3.22.1 + x-stainless-read-timeout: + - '600' + x-stainless-retry-count: + - '0' + x-stainless-runtime: + - CPython + x-stainless-runtime-version: + - 3.12.14 + method: POST + uri: https://api.openai.com/v1/audio/transcriptions + response: + body: + string: '{"text":"Where is my order number 1042?","languages":[{"code":"en"}],"usage":{"type":"duration","seconds":3}}' + headers: + access-control-expose-headers: + - X-Request-ID + - CF-Ray + alt-svc: + - h3=":443"; ma=86400 + cf-cache-status: + - DYNAMIC + cf-ray: + - a45e173f48dc9991-ORD + connection: + - keep-alive + content-length: + - '109' + content-type: + - application/json + date: + - Mon, 05 Oct 2026 17:19:22 GMT + openai-processing-ms: + - '415' + openai-version: + - '2020-10-01' + server: + - cloudflare + strict-transport-security: + - max-age=31536000; includeSubDomains; preload + transfer-encoding: + - chunked + x-content-type-options: + - nosniff + x-openai-proxy-wasm: + - v0.1 + x-ratelimit-limit-requests: + - '30000' + x-ratelimit-remaining-requests: + - '29999' + x-ratelimit-reset-requests: + - 2ms + x-request-id: + - req_60ff4c35d79a4ef4ba5524b883e47c7a + status: + code: 200 + message: OK +- request: + body: '{"messages":[{"role":"system","content":"You are an English order assistant. + Always call lookup_order for the supplied order number. After its result, answer + only: Your order arrives Friday."},{"role":"user","content":"Where is my order + number 1042?"}],"model":"gpt-4.1-mini","stream":true,"stream_options":{"include_usage":true},"temperature":0,"tools":[{"type":"function","function":{"name":"lookup_order","description":"Look + up an order","parameters":{"type":"object","properties":{"order_id":{"type":"string"}},"required":["order_id"]}}}]}' + headers: + accept: + - application/json + accept-encoding: + - gzip, deflate + connection: + - keep-alive + content-length: + - '543' + content-type: + - application/json + host: + - api.openai.com + user-agent: + - AsyncOpenAI/Python 3.22.1 + x-stainless-arch: + - arm64 + x-stainless-async: + - async:asyncio + x-stainless-lang: + - python + x-stainless-os: + - MacOS + x-stainless-package-version: + - 3.22.1 + x-stainless-read-timeout: + - '600' + x-stainless-retry-count: + - '0' + x-stainless-runtime: + - CPython + x-stainless-runtime-version: + - 3.12.14 + method: POST + uri: https://api.openai.com/v1/chat/completions + response: + body: + string: 'data: {"id":"chatcmpl-EVgdOuXIzx8k2PyBeZ13DlFvyTiHs","object":"chat.completion.chunk","created":1791220762,"model":"gpt-4.1-mini-2025-04-14","service_tier":"default","system_fingerprint":"fp_717f55dfc9","choices":[{"index":0,"delta":{"role":"assistant","content":null,"tool_calls":[{"index":0,"id":"call_HNcMI5HXdcAexG9Q2LyrywSk","type":"function","function":{"name":"lookup_order","arguments":""}}],"refusal":null},"logprobs":null,"finish_reason":null}],"usage":null,"obfuscation":"0G"} + + + data: {"id":"chatcmpl-EVgdOuXIzx8k2PyBeZ13DlFvyTiHs","object":"chat.completion.chunk","created":1791220762,"model":"gpt-4.1-mini-2025-04-14","service_tier":"default","system_fingerprint":"fp_717f55dfc9","choices":[{"index":0,"delta":{"tool_calls":[{"index":0,"function":{"arguments":"{\""}}]},"logprobs":null,"finish_reason":null}],"usage":null,"obfuscation":"t0EAEPjwOE3V4"} + + + data: {"id":"chatcmpl-EVgdOuXIzx8k2PyBeZ13DlFvyTiHs","object":"chat.completion.chunk","created":1791220762,"model":"gpt-4.1-mini-2025-04-14","service_tier":"default","system_fingerprint":"fp_717f55dfc9","choices":[{"index":0,"delta":{"tool_calls":[{"index":0,"function":{"arguments":"order"}}]},"logprobs":null,"finish_reason":null}],"usage":null,"obfuscation":"hTsuGEQluaQ"} + + + data: {"id":"chatcmpl-EVgdOuXIzx8k2PyBeZ13DlFvyTiHs","object":"chat.completion.chunk","created":1791220762,"model":"gpt-4.1-mini-2025-04-14","service_tier":"default","system_fingerprint":"fp_717f55dfc9","choices":[{"index":0,"delta":{"tool_calls":[{"index":0,"function":{"arguments":"_id"}}]},"logprobs":null,"finish_reason":null}],"usage":null,"obfuscation":"RbDW9yy4p5d0B"} + + + data: {"id":"chatcmpl-EVgdOuXIzx8k2PyBeZ13DlFvyTiHs","object":"chat.completion.chunk","created":1791220762,"model":"gpt-4.1-mini-2025-04-14","service_tier":"default","system_fingerprint":"fp_717f55dfc9","choices":[{"index":0,"delta":{"tool_calls":[{"index":0,"function":{"arguments":"\":\""}}]},"logprobs":null,"finish_reason":null}],"usage":null,"obfuscation":"NQsK0o1U7Yp"} + + + data: {"id":"chatcmpl-EVgdOuXIzx8k2PyBeZ13DlFvyTiHs","object":"chat.completion.chunk","created":1791220762,"model":"gpt-4.1-mini-2025-04-14","service_tier":"default","system_fingerprint":"fp_717f55dfc9","choices":[{"index":0,"delta":{"tool_calls":[{"index":0,"function":{"arguments":"104"}}]},"logprobs":null,"finish_reason":null}],"usage":null,"obfuscation":"m129b49pxd4cL"} + + + data: {"id":"chatcmpl-EVgdOuXIzx8k2PyBeZ13DlFvyTiHs","object":"chat.completion.chunk","created":1791220762,"model":"gpt-4.1-mini-2025-04-14","service_tier":"default","system_fingerprint":"fp_717f55dfc9","choices":[{"index":0,"delta":{"tool_calls":[{"index":0,"function":{"arguments":"2"}}]},"logprobs":null,"finish_reason":null}],"usage":null,"obfuscation":"dcNLW3HIbOM5kP8"} + + + data: {"id":"chatcmpl-EVgdOuXIzx8k2PyBeZ13DlFvyTiHs","object":"chat.completion.chunk","created":1791220762,"model":"gpt-4.1-mini-2025-04-14","service_tier":"default","system_fingerprint":"fp_717f55dfc9","choices":[{"index":0,"delta":{"tool_calls":[{"index":0,"function":{"arguments":"\"}"}}]},"logprobs":null,"finish_reason":null}],"usage":null,"obfuscation":"EWvEcBb97SY20"} + + + data: {"id":"chatcmpl-EVgdOuXIzx8k2PyBeZ13DlFvyTiHs","object":"chat.completion.chunk","created":1791220762,"model":"gpt-4.1-mini-2025-04-14","service_tier":"default","system_fingerprint":"fp_717f55dfc9","choices":[{"index":0,"delta":{},"logprobs":null,"finish_reason":"tool_calls"}],"usage":null,"obfuscation":"mM84yutyoCf36X"} + + + data: {"id":"chatcmpl-EVgdOuXIzx8k2PyBeZ13DlFvyTiHs","object":"chat.completion.chunk","created":1791220762,"model":"gpt-4.1-mini-2025-04-14","service_tier":"default","system_fingerprint":"fp_717f55dfc9","choices":[],"usage":{"prompt_tokens":81,"completion_tokens":16,"total_tokens":97,"prompt_tokens_details":{"cached_tokens":0,"audio_tokens":0},"completion_tokens_details":{"reasoning_tokens":0,"audio_tokens":0,"accepted_prediction_tokens":0,"rejected_prediction_tokens":0}},"obfuscation":"h4w8R04Gi"} + + + data: [DONE] + + + ' + headers: + access-control-expose-headers: + - X-Request-ID + - CF-Ray + - CF-Ray + alt-svc: + - h3=":443"; ma=86400 + cf-cache-status: + - DYNAMIC + cf-ray: + - a45e17434956eb08-ORD + connection: + - keep-alive + content-type: + - text/event-stream; charset=utf-8 + date: + - Mon, 05 Oct 2026 17:19:22 GMT + openai-processing-ms: + - '401' + openai-version: + - '2020-10-01' + server: + - cloudflare + strict-transport-security: + - max-age=31536000; includeSubDomains; preload + transfer-encoding: + - chunked + x-content-type-options: + - nosniff + x-openai-proxy-wasm: + - v0.1 + x-ratelimit-limit-requests: + - '30000' + x-ratelimit-limit-tokens: + - '150000000' + x-ratelimit-remaining-requests: + - '29999' + x-ratelimit-remaining-tokens: + - '149999952' + x-ratelimit-reset-requests: + - 2ms + x-ratelimit-reset-tokens: + - 0s + x-request-id: + - req_abcbe4d2155743049ef5fc4d96018f97 + status: + code: 200 + message: OK +- request: + body: '{"messages":[{"role":"system","content":"You are an English order assistant. + Always call lookup_order for the supplied order number. After its result, answer + only: Your order arrives Friday."},{"role":"user","content":"Where is my order + number 1042?"},{"role":"assistant","tool_calls":[{"id":"call_HNcMI5HXdcAexG9Q2LyrywSk","function":{"name":"lookup_order","arguments":"{\"order_id\": + \"1042\"}"},"type":"function"}]},{"role":"tool","content":"{\"order_id\": \"1042\", + \"delivery\": \"Friday\"}","tool_call_id":"call_HNcMI5HXdcAexG9Q2LyrywSk"}],"model":"gpt-4.1-mini","stream":true,"stream_options":{"include_usage":true},"temperature":0,"tools":[{"type":"function","function":{"name":"lookup_order","description":"Look + up an order","parameters":{"type":"object","properties":{"order_id":{"type":"string"}},"required":["order_id"]}}}]}' + headers: + accept: + - application/json + accept-encoding: + - gzip, deflate + connection: + - keep-alive + content-length: + - '836' + content-type: + - application/json + host: + - api.openai.com + user-agent: + - AsyncOpenAI/Python 3.22.1 + x-stainless-arch: + - arm64 + x-stainless-async: + - async:asyncio + x-stainless-lang: + - python + x-stainless-os: + - MacOS + x-stainless-package-version: + - 3.22.1 + x-stainless-read-timeout: + - '600' + x-stainless-retry-count: + - '0' + x-stainless-runtime: + - CPython + x-stainless-runtime-version: + - 3.12.14 + method: POST + uri: https://api.openai.com/v1/chat/completions + response: + body: + string: 'data: {"id":"chatcmpl-EVgdPGIU8tt1JYoX26BJKEwutVjgO","object":"chat.completion.chunk","created":1791220763,"model":"gpt-4.1-mini-2025-04-14","service_tier":"default","system_fingerprint":"fp_717f55dfc9","choices":[{"index":0,"delta":{"role":"assistant","content":"","refusal":null},"logprobs":null,"finish_reason":null}],"usage":null,"obfuscation":"iy4PmEzM"} + + + data: {"id":"chatcmpl-EVgdPGIU8tt1JYoX26BJKEwutVjgO","object":"chat.completion.chunk","created":1791220763,"model":"gpt-4.1-mini-2025-04-14","service_tier":"default","system_fingerprint":"fp_717f55dfc9","choices":[{"index":0,"delta":{"content":"Your"},"logprobs":null,"finish_reason":null}],"usage":null,"obfuscation":"SJmzGt"} + + + data: {"id":"chatcmpl-EVgdPGIU8tt1JYoX26BJKEwutVjgO","object":"chat.completion.chunk","created":1791220763,"model":"gpt-4.1-mini-2025-04-14","service_tier":"default","system_fingerprint":"fp_717f55dfc9","choices":[{"index":0,"delta":{"content":" + order"},"logprobs":null,"finish_reason":null}],"usage":null,"obfuscation":"7VgR"} + + + data: {"id":"chatcmpl-EVgdPGIU8tt1JYoX26BJKEwutVjgO","object":"chat.completion.chunk","created":1791220763,"model":"gpt-4.1-mini-2025-04-14","service_tier":"default","system_fingerprint":"fp_717f55dfc9","choices":[{"index":0,"delta":{"content":" + arrives"},"logprobs":null,"finish_reason":null}],"usage":null,"obfuscation":"6t"} + + + data: {"id":"chatcmpl-EVgdPGIU8tt1JYoX26BJKEwutVjgO","object":"chat.completion.chunk","created":1791220763,"model":"gpt-4.1-mini-2025-04-14","service_tier":"default","system_fingerprint":"fp_717f55dfc9","choices":[{"index":0,"delta":{"content":" + Friday"},"logprobs":null,"finish_reason":null}],"usage":null,"obfuscation":"AQQ"} + + + data: {"id":"chatcmpl-EVgdPGIU8tt1JYoX26BJKEwutVjgO","object":"chat.completion.chunk","created":1791220763,"model":"gpt-4.1-mini-2025-04-14","service_tier":"default","system_fingerprint":"fp_717f55dfc9","choices":[{"index":0,"delta":{"content":"."},"logprobs":null,"finish_reason":null}],"usage":null,"obfuscation":"HSolHVyjx"} + + + data: {"id":"chatcmpl-EVgdPGIU8tt1JYoX26BJKEwutVjgO","object":"chat.completion.chunk","created":1791220763,"model":"gpt-4.1-mini-2025-04-14","service_tier":"default","system_fingerprint":"fp_717f55dfc9","choices":[{"index":0,"delta":{},"logprobs":null,"finish_reason":"stop"}],"usage":null,"obfuscation":"QUpY"} + + + data: {"id":"chatcmpl-EVgdPGIU8tt1JYoX26BJKEwutVjgO","object":"chat.completion.chunk","created":1791220763,"model":"gpt-4.1-mini-2025-04-14","service_tier":"default","system_fingerprint":"fp_717f55dfc9","choices":[],"usage":{"prompt_tokens":119,"completion_tokens":6,"total_tokens":125,"prompt_tokens_details":{"cached_tokens":0,"audio_tokens":0},"completion_tokens_details":{"reasoning_tokens":0,"audio_tokens":0,"accepted_prediction_tokens":0,"rejected_prediction_tokens":0}},"obfuscation":"lqUhtSpH"} + + + data: [DONE] + + + ' + headers: + access-control-expose-headers: + - X-Request-ID + - CF-Ray + - CF-Ray + alt-svc: + - h3=":443"; ma=86400 + cf-cache-status: + - DYNAMIC + cf-ray: + - a45e1747d93beb08-ORD + connection: + - keep-alive + content-type: + - text/event-stream; charset=utf-8 + date: + - Mon, 05 Oct 2026 17:19:23 GMT + openai-processing-ms: + - '342' + openai-version: + - '2020-10-01' + server: + - cloudflare + strict-transport-security: + - max-age=31536000; includeSubDomains; preload + transfer-encoding: + - chunked + x-content-type-options: + - nosniff + x-openai-proxy-wasm: + - v0.1 + x-ratelimit-limit-requests: + - '30000' + x-ratelimit-limit-tokens: + - '150000000' + x-ratelimit-remaining-requests: + - '29998' + x-ratelimit-remaining-tokens: + - '149999842' + x-ratelimit-reset-requests: + - 4ms + x-ratelimit-reset-tokens: + - 0s + x-request-id: + - req_06cb46e2f249447ea68ce13d2c440229 + status: + code: 200 + message: OK +- request: + body: '{"input":"Your order arrives Friday.","model":"gpt-4o-mini-tts","voice":"alloy","response_format":"pcm"}' + headers: + accept: + - application/octet-stream + accept-encoding: + - gzip, deflate + connection: + - keep-alive + content-length: + - '104' + content-type: + - application/json + host: + - api.openai.com + user-agent: + - AsyncOpenAI/Python 3.22.1 + x-stainless-arch: + - arm64 + x-stainless-async: + - async:asyncio + x-stainless-lang: + - python + x-stainless-os: + - MacOS + x-stainless-package-version: + - 3.22.1 + x-stainless-raw-response: + - stream + x-stainless-read-timeout: + - '600' + x-stainless-retry-count: + - '0' + x-stainless-runtime: + - CPython + x-stainless-runtime-version: + - 3.12.14 + method: POST + uri: https://api.openai.com/v1/audio/speech + response: + body: + string: + audio_file: test_cascade_voice_conversation.audio/response-3.wav + pcm: true + headers: + access-control-expose-headers: + - X-Request-ID + - CF-Ray + alt-svc: + - h3=":443"; ma=86400 + cf-cache-status: + - DYNAMIC + cf-ray: + - a45e174b9f200f96-ORD + connection: + - keep-alive + content-type: + - audio/pcm + date: + - Mon, 05 Oct 2026 17:19:23 GMT + openai-processing-ms: + - '350' + openai-version: + - '2020-10-01' + server: + - cloudflare + strict-transport-security: + - max-age=31536000; includeSubDomains; preload + transfer-encoding: + - chunked + x-content-type-options: + - nosniff + x-openai-proxy-wasm: + - v0.1 + x-ratelimit-limit-requests: + - '30000' + x-ratelimit-limit-tokens: + - '150000000' + x-ratelimit-remaining-requests: + - '29999' + x-ratelimit-remaining-tokens: + - '149999995' + x-ratelimit-reset-requests: + - 2ms + x-ratelimit-reset-tokens: + - 0s + x-request-id: + - req_747786e6e11a4ea1ad86e5112285942c + status: + code: 200 + message: OK +version: 1 diff --git a/py/src/braintrust/integrations/pipecat/cassettes/latest/test_realtime_voice_conversation.audio/agent-000.wav b/py/src/braintrust/integrations/pipecat/cassettes/latest/test_realtime_voice_conversation.audio/agent-000.wav new file mode 100644 index 000000000..b5f0b07c5 Binary files /dev/null and b/py/src/braintrust/integrations/pipecat/cassettes/latest/test_realtime_voice_conversation.audio/agent-000.wav differ diff --git a/py/src/braintrust/integrations/pipecat/cassettes/latest/test_realtime_voice_conversation.audio/agent-001.wav b/py/src/braintrust/integrations/pipecat/cassettes/latest/test_realtime_voice_conversation.audio/agent-001.wav new file mode 100644 index 000000000..702ae2668 Binary files /dev/null and b/py/src/braintrust/integrations/pipecat/cassettes/latest/test_realtime_voice_conversation.audio/agent-001.wav differ diff --git a/py/src/braintrust/integrations/pipecat/cassettes/latest/test_realtime_voice_conversation.audio/caller.wav b/py/src/braintrust/integrations/pipecat/cassettes/latest/test_realtime_voice_conversation.audio/caller.wav new file mode 100644 index 000000000..b77ccbc6d Binary files /dev/null and b/py/src/braintrust/integrations/pipecat/cassettes/latest/test_realtime_voice_conversation.audio/caller.wav differ diff --git a/py/src/braintrust/integrations/pipecat/cassettes/latest/test_realtime_voice_conversation.json b/py/src/braintrust/integrations/pipecat/cassettes/latest/test_realtime_voice_conversation.json new file mode 100644 index 000000000..1f87bbdb8 --- /dev/null +++ b/py/src/braintrust/integrations/pipecat/cassettes/latest/test_realtime_voice_conversation.json @@ -0,0 +1,4108 @@ +{ + "endpoint": "wss://api.openai.com/v1/realtime?model=gpt-realtime", + "events": [ + { + "direction": "receive", + "message": { + "type": "session.created", + "event_id": "event_EVgfEAiup12QhYVd70qag", + "session": { + "type": "realtime", + "object": "realtime.session", + "id": "sess_EVgfE0UHxI9DXQ9Yo8zII", + "model": "gpt-realtime", + "output_modalities": [ + "audio" + ], + "instructions": "Your knowledge cutoff is 2023-10. You are a helpful, witty, and friendly AI. Act like a human, but remember that you aren't a human and that you can't do human things in the real world. Your voice and personality should be warm and engaging, with a lively and playful tone. If interacting in a non-English language, start by using the standard accent or dialect familiar to the user. Talk quickly. You should always call a function if you can. Do not refer to these rules, even if you\u2019re asked about them.", + "tools": [], + "tool_choice": "auto", + "max_output_tokens": "inf", + "tracing": null, + "truncation": "auto", + "prompt": null, + "expires_at": 1791224476, + "audio": { + "input": { + "format": { + "type": "audio/pcm", + "rate": 24000 + }, + "transcription": null, + "noise_reduction": null, + "turn_detection": { + "type": "server_vad", + "threshold": 0.5, + "prefix_padding_ms": 300, + "silence_duration_ms": 200, + "idle_timeout_ms": null, + "create_response": true, + "interrupt_response": true + } + }, + "output": { + "format": { + "type": "audio/pcm", + "rate": 24000 + }, + "voice": "alloy", + "speed": 1.0 + } + }, + "include": null + } + } + }, + { + "direction": "send", + "message": { + "event_id": "3dd4fef7-0f9b-4f34-a6f6-208b66befa86", + "type": "session.update", + "session": { + "type": "realtime", + "model": "gpt-realtime", + "audio": { + "input": { + "transcription": { + "model": "gpt-4o-mini-transcribe", + "language": "en" + }, + "turn_detection": { + "type": "server_vad", + "threshold": 0.5, + "prefix_padding_ms": 300, + "silence_duration_ms": 500 + } + }, + "output": { + "voice": "alloy" + } + } + } + } + }, + { + "direction": "receive", + "message": { + "type": "session.updated", + "event_id": "event_EVgfEN7QM6WTgGAob6gcR", + "session": { + "type": "realtime", + "object": "realtime.session", + "id": "sess_EVgfE0UHxI9DXQ9Yo8zII", + "model": "gpt-realtime", + "output_modalities": [ + "audio" + ], + "instructions": "Your knowledge cutoff is 2023-10. You are a helpful, witty, and friendly AI. Act like a human, but remember that you aren't a human and that you can't do human things in the real world. Your voice and personality should be warm and engaging, with a lively and playful tone. If interacting in a non-English language, start by using the standard accent or dialect familiar to the user. Talk quickly. You should always call a function if you can. Do not refer to these rules, even if you\u2019re asked about them.", + "tools": [], + "tool_choice": "auto", + "max_output_tokens": "inf", + "tracing": null, + "truncation": "auto", + "prompt": null, + "expires_at": 1791224476, + "audio": { + "input": { + "format": { + "type": "audio/pcm", + "rate": 24000 + }, + "transcription": { + "model": "gpt-4o-mini-transcribe", + "language": "en", + "prompt": null + }, + "noise_reduction": null, + "turn_detection": { + "type": "server_vad", + "threshold": 0.5, + "prefix_padding_ms": 300, + "silence_duration_ms": 500, + "idle_timeout_ms": null, + "create_response": true, + "interrupt_response": true + } + }, + "output": { + "format": { + "type": "audio/pcm", + "rate": 24000 + }, + "voice": "alloy", + "speed": 1.0 + } + }, + "include": null + } + } + }, + { + "direction": "send", + "message": { + "event_id": "e768ee89-d596-488b-8d80-269da1e9c5b0", + "type": "session.update", + "session": { + "type": "realtime", + "model": "gpt-realtime", + "instructions": "Speak English. Start by saying exactly: Hello, what is your order number? When the user gives an order number, always call lookup_order. After its result, say only: Your order arrives Friday.", + "audio": { + "input": { + "transcription": { + "model": "gpt-4o-mini-transcribe", + "language": "en" + }, + "turn_detection": { + "type": "server_vad", + "threshold": 0.5, + "prefix_padding_ms": 300, + "silence_duration_ms": 500 + } + }, + "output": { + "voice": "alloy" + } + }, + "tools": [ + { + "type": "function", + "name": "lookup_order", + "description": "Look up an order", + "parameters": { + "type": "object", + "properties": { + "order_id": { + "type": "string" + } + }, + "required": [ + "order_id" + ] + } + } + ] + } + } + }, + { + "direction": "send", + "message": { + "event_id": "97a191cf-60e7-4190-b6aa-e36a64ed22dc", + "type": "response.create", + "response": { + "output_modalities": [ + "audio" + ] + } + } + }, + { + "direction": "receive", + "message": { + "type": "session.updated", + "event_id": "event_EVgfFPztItLujfdo4yLci", + "session": { + "type": "realtime", + "object": "realtime.session", + "id": "sess_EVgfE0UHxI9DXQ9Yo8zII", + "model": "gpt-realtime", + "output_modalities": [ + "audio" + ], + "instructions": "Speak English. Start by saying exactly: Hello, what is your order number? When the user gives an order number, always call lookup_order. After its result, say only: Your order arrives Friday.", + "tools": [ + { + "type": "function", + "name": "lookup_order", + "description": "Look up an order", + "parameters": { + "type": "object", + "properties": { + "order_id": { + "type": "string" + } + }, + "required": [ + "order_id" + ] + } + } + ], + "tool_choice": "auto", + "max_output_tokens": "inf", + "tracing": null, + "truncation": "auto", + "prompt": null, + "expires_at": 1791224476, + "audio": { + "input": { + "format": { + "type": "audio/pcm", + "rate": 24000 + }, + "transcription": { + "model": "gpt-4o-mini-transcribe", + "language": "en", + "prompt": null + }, + "noise_reduction": null, + "turn_detection": { + "type": "server_vad", + "threshold": 0.5, + "prefix_padding_ms": 300, + "silence_duration_ms": 500, + "idle_timeout_ms": null, + "create_response": true, + "interrupt_response": true + } + }, + "output": { + "format": { + "type": "audio/pcm", + "rate": 24000 + }, + "voice": "alloy", + "speed": 1.0 + } + }, + "include": null + } + } + }, + { + "direction": "receive", + "message": { + "type": "response.created", + "event_id": "event_EVgfFylg6nuToHIIbEL1c", + "response": { + "object": "realtime.response", + "id": "resp_EVgfF05xkwLTRprf4ZMGS", + "status": "in_progress", + "status_details": null, + "output": [], + "conversation_id": "conv_EVgfECJ3whhWkN1aLHqyp", + "output_modalities": [ + "audio" + ], + "max_output_tokens": "inf", + "audio": { + "output": { + "format": { + "type": "audio/pcm", + "rate": 24000 + }, + "voice": "alloy" + } + }, + "usage": null, + "metadata": null + } + } + }, + { + "direction": "receive", + "message": { + "type": "response.output_item.added", + "event_id": "event_EVgfFcvOCYJw4U3QzzAta", + "response_id": "resp_EVgfF05xkwLTRprf4ZMGS", + "output_index": 0, + "item": { + "id": "item_EVgfFMwZMFVUydi3vk3kH", + "type": "message", + "status": "in_progress", + "role": "assistant", + "content": [] + } + } + }, + { + "direction": "receive", + "message": { + "type": "conversation.item.added", + "event_id": "event_EVgfFx0bCZHjn5bmy6T0p", + "previous_item_id": null, + "item": { + "id": "item_EVgfFMwZMFVUydi3vk3kH", + "type": "message", + "status": "in_progress", + "role": "assistant", + "content": [] + } + } + }, + { + "direction": "receive", + "message": { + "type": "response.content_part.added", + "event_id": "event_EVgfFxiYWK2vHHKeTiT30", + "response_id": "resp_EVgfF05xkwLTRprf4ZMGS", + "item_id": "item_EVgfFMwZMFVUydi3vk3kH", + "output_index": 0, + "content_index": 0, + "part": { + "type": "audio", + "transcript": "" + } + } + }, + { + "direction": "receive", + "message": { + "type": "response.output_audio_transcript.delta", + "event_id": "event_EVgfF0TFL44zN97ZV3rCS", + "response_id": "resp_EVgfF05xkwLTRprf4ZMGS", + "item_id": "item_EVgfFMwZMFVUydi3vk3kH", + "output_index": 0, + "content_index": 0, + "delta": "Hello", + "obfuscation": "DeLEdh6Ytke" + } + }, + { + "direction": "receive", + "message": { + "type": "response.output_audio_transcript.delta", + "event_id": "event_EVgfFGukh1XaEPV55gHT0", + "response_id": "resp_EVgfF05xkwLTRprf4ZMGS", + "item_id": "item_EVgfFMwZMFVUydi3vk3kH", + "output_index": 0, + "content_index": 0, + "delta": ",", + "obfuscation": "6jYgU2pWZdrsrEx" + } + }, + { + "direction": "receive", + "message": { + "type": "response.output_audio_transcript.delta", + "event_id": "event_EVgfF1Dw7ZGupP2WNCLoB", + "response_id": "resp_EVgfF05xkwLTRprf4ZMGS", + "item_id": "item_EVgfFMwZMFVUydi3vk3kH", + "output_index": 0, + "content_index": 0, + "delta": " what", + "obfuscation": "nIqZ5jeLRyx" + } + }, + { + "direction": "receive", + "message": { + "type": "response.output_audio.delta", + "event_id": "event_EVgfFSake8THjRPVSasBa", + "response_id": "resp_EVgfF05xkwLTRprf4ZMGS", + "item_id": "item_EVgfFMwZMFVUydi3vk3kH", + "output_index": 0, + "content_index": 0, + "delta": { + "audio_file": "test_realtime_voice_conversation.audio/agent-000.wav", + "pcm": true, + "offset_bytes": 0, + "length_bytes": 4800 + } + } + }, + { + "direction": "receive", + "message": { + "type": "response.output_audio.delta", + "event_id": "event_EVgfFa2hjq3GWEf0xjMtU", + "response_id": "resp_EVgfF05xkwLTRprf4ZMGS", + "item_id": "item_EVgfFMwZMFVUydi3vk3kH", + "output_index": 0, + "content_index": 0, + "delta": { + "audio_file": "test_realtime_voice_conversation.audio/agent-000.wav", + "pcm": true, + "offset_bytes": 4800, + "length_bytes": 7200 + } + } + }, + { + "direction": "receive", + "message": { + "type": "response.output_audio.delta", + "event_id": "event_EVgfF99gI14pN9GbvgPbz", + "response_id": "resp_EVgfF05xkwLTRprf4ZMGS", + "item_id": "item_EVgfFMwZMFVUydi3vk3kH", + "output_index": 0, + "content_index": 0, + "delta": { + "audio_file": "test_realtime_voice_conversation.audio/agent-000.wav", + "pcm": true, + "offset_bytes": 12000, + "length_bytes": 12000 + } + } + }, + { + "direction": "receive", + "message": { + "type": "response.output_audio_transcript.delta", + "event_id": "event_EVgfFvkFSKIxozDZkmL4J", + "response_id": "resp_EVgfF05xkwLTRprf4ZMGS", + "item_id": "item_EVgfFMwZMFVUydi3vk3kH", + "output_index": 0, + "content_index": 0, + "delta": " is", + "obfuscation": "lo9K3Hly2JeLz" + } + }, + { + "direction": "receive", + "message": { + "type": "response.output_audio_transcript.delta", + "event_id": "event_EVgfFHEA9PPwyQVreUPcG", + "response_id": "resp_EVgfF05xkwLTRprf4ZMGS", + "item_id": "item_EVgfFMwZMFVUydi3vk3kH", + "output_index": 0, + "content_index": 0, + "delta": " your", + "obfuscation": "24fzDDAbbzc" + } + }, + { + "direction": "receive", + "message": { + "type": "response.output_audio_transcript.delta", + "event_id": "event_EVgfFF1nBxumTYIVgwb3w", + "response_id": "resp_EVgfF05xkwLTRprf4ZMGS", + "item_id": "item_EVgfFMwZMFVUydi3vk3kH", + "output_index": 0, + "content_index": 0, + "delta": " order", + "obfuscation": "esClxxYiFa" + } + }, + { + "direction": "receive", + "message": { + "type": "response.output_audio_transcript.delta", + "event_id": "event_EVgfFdGpkB1nwKjtXOO6c", + "response_id": "resp_EVgfF05xkwLTRprf4ZMGS", + "item_id": "item_EVgfFMwZMFVUydi3vk3kH", + "output_index": 0, + "content_index": 0, + "delta": " number", + "obfuscation": "pPGWNfMzm" + } + }, + { + "direction": "receive", + "message": { + "type": "response.output_audio_transcript.delta", + "event_id": "event_EVgfFE0ToAAUqU1mqIEpn", + "response_id": "resp_EVgfF05xkwLTRprf4ZMGS", + "item_id": "item_EVgfFMwZMFVUydi3vk3kH", + "output_index": 0, + "content_index": 0, + "delta": "?", + "obfuscation": "G2ppfEvThGrMStn" + } + }, + { + "direction": "receive", + "message": { + "type": "response.output_audio.delta", + "event_id": "event_EVgfFnGaL4zCZfL2zjKlp", + "response_id": "resp_EVgfF05xkwLTRprf4ZMGS", + "item_id": "item_EVgfFMwZMFVUydi3vk3kH", + "output_index": 0, + "content_index": 0, + "delta": { + "audio_file": "test_realtime_voice_conversation.audio/agent-000.wav", + "pcm": true, + "offset_bytes": 24000, + "length_bytes": 12000 + } + } + }, + { + "direction": "receive", + "message": { + "type": "response.output_audio.delta", + "event_id": "event_EVgfFZf2vYtFmKGgXtua6", + "response_id": "resp_EVgfF05xkwLTRprf4ZMGS", + "item_id": "item_EVgfFMwZMFVUydi3vk3kH", + "output_index": 0, + "content_index": 0, + "delta": { + "audio_file": "test_realtime_voice_conversation.audio/agent-000.wav", + "pcm": true, + "offset_bytes": 36000, + "length_bytes": 12000 + } + } + }, + { + "direction": "receive", + "message": { + "type": "response.output_audio.delta", + "event_id": "event_EVgfFbMRFE3NC5TK09gjS", + "response_id": "resp_EVgfF05xkwLTRprf4ZMGS", + "item_id": "item_EVgfFMwZMFVUydi3vk3kH", + "output_index": 0, + "content_index": 0, + "delta": { + "audio_file": "test_realtime_voice_conversation.audio/agent-000.wav", + "pcm": true, + "offset_bytes": 48000, + "length_bytes": 12000 + } + } + }, + { + "direction": "receive", + "message": { + "type": "response.output_audio.delta", + "event_id": "event_EVgfFBMt8MZL2DZkIFIYA", + "response_id": "resp_EVgfF05xkwLTRprf4ZMGS", + "item_id": "item_EVgfFMwZMFVUydi3vk3kH", + "output_index": 0, + "content_index": 0, + "delta": { + "audio_file": "test_realtime_voice_conversation.audio/agent-000.wav", + "pcm": true, + "offset_bytes": 60000, + "length_bytes": 12000 + } + } + }, + { + "direction": "receive", + "message": { + "type": "response.output_audio.delta", + "event_id": "event_EVgfFlySfzpaPGIQTaoll", + "response_id": "resp_EVgfF05xkwLTRprf4ZMGS", + "item_id": "item_EVgfFMwZMFVUydi3vk3kH", + "output_index": 0, + "content_index": 0, + "delta": { + "audio_file": "test_realtime_voice_conversation.audio/agent-000.wav", + "pcm": true, + "offset_bytes": 72000, + "length_bytes": 26400 + } + } + }, + { + "direction": "receive", + "message": { + "type": "response.output_audio.done", + "event_id": "event_EVgfFifZ6swki1doiVUx0", + "response_id": "resp_EVgfF05xkwLTRprf4ZMGS", + "item_id": "item_EVgfFMwZMFVUydi3vk3kH", + "output_index": 0, + "content_index": 0 + } + }, + { + "direction": "receive", + "message": { + "type": "response.output_audio_transcript.done", + "event_id": "event_EVgfFp9wkrPpEd8ysK94r", + "response_id": "resp_EVgfF05xkwLTRprf4ZMGS", + "item_id": "item_EVgfFMwZMFVUydi3vk3kH", + "output_index": 0, + "content_index": 0, + "transcript": "Hello, what is your order number?" + } + }, + { + "direction": "receive", + "message": { + "type": "response.content_part.done", + "event_id": "event_EVgfFXSGWi0KpOx4hGqES", + "response_id": "resp_EVgfF05xkwLTRprf4ZMGS", + "item_id": "item_EVgfFMwZMFVUydi3vk3kH", + "output_index": 0, + "content_index": 0, + "part": { + "type": "audio", + "transcript": "Hello, what is your order number?" + } + } + }, + { + "direction": "receive", + "message": { + "type": "conversation.item.done", + "event_id": "event_EVgfFyEfywh398vOxgBqm", + "previous_item_id": null, + "item": { + "id": "item_EVgfFMwZMFVUydi3vk3kH", + "type": "message", + "status": "completed", + "role": "assistant", + "content": [ + { + "type": "output_audio", + "transcript": "Hello, what is your order number?" + } + ] + } + } + }, + { + "direction": "receive", + "message": { + "type": "response.output_item.done", + "event_id": "event_EVgfFz2vGp6iGLHmDM7Jy", + "response_id": "resp_EVgfF05xkwLTRprf4ZMGS", + "output_index": 0, + "item": { + "id": "item_EVgfFMwZMFVUydi3vk3kH", + "type": "message", + "status": "completed", + "role": "assistant", + "content": [ + { + "type": "output_audio", + "transcript": "Hello, what is your order number?" + } + ] + } + } + }, + { + "direction": "receive", + "message": { + "type": "response.done", + "event_id": "event_EVgfFPmrJJi1WrEu6n9MU", + "response": { + "object": "realtime.response", + "id": "resp_EVgfF05xkwLTRprf4ZMGS", + "status": "completed", + "status_details": null, + "output": [ + { + "id": "item_EVgfFMwZMFVUydi3vk3kH", + "type": "message", + "status": "completed", + "role": "assistant", + "content": [ + { + "type": "output_audio", + "transcript": "Hello, what is your order number?" + } + ] + } + ], + "conversation_id": "conv_EVgfECJ3whhWkN1aLHqyp", + "output_modalities": [ + "audio" + ], + "max_output_tokens": "inf", + "audio": { + "output": { + "format": { + "type": "audio/pcm", + "rate": 24000 + }, + "voice": "alloy" + } + }, + "usage": { + "total_tokens": 134, + "input_tokens": 75, + "output_tokens": 59, + "input_token_details": { + "text_tokens": 75, + "audio_tokens": 0, + "image_tokens": 0, + "cached_tokens": 0, + "cached_tokens_details": { + "text_tokens": 0, + "audio_tokens": 0, + "image_tokens": 0 + } + }, + "output_token_details": { + "text_tokens": 18, + "audio_tokens": 41 + } + }, + "metadata": null + } + } + }, + { + "direction": "send", + "message": { + "event_id": "f7bfec2b-6337-49ba-84ad-cdd7cb216713", + "type": "input_audio_buffer.append", + "audio": { + "audio_file": "test_realtime_voice_conversation.audio/caller.wav", + "pcm": true, + "offset_bytes": 0, + "length_bytes": 960 + } + } + }, + { + "direction": "receive", + "message": { + "type": "rate_limits.updated", + "event_id": "event_EVgfFoOG2vC4FO7RSBKTu", + "rate_limits": [ + { + "name": "tokens", + "limit": 15000000, + "remaining": 14999295, + "reset_seconds": 0.002 + } + ] + } + }, + { + "direction": "send", + "message": { + "event_id": "a3226ad9-901c-49d0-ba5b-85901ec861c7", + "type": "input_audio_buffer.append", + "audio": { + "audio_file": "test_realtime_voice_conversation.audio/caller.wav", + "pcm": true, + "offset_bytes": 960, + "length_bytes": 960 + } + } + }, + { + "direction": "send", + "message": { + "event_id": "d17ec867-e351-4751-a463-d34852df50cf", + "type": "input_audio_buffer.append", + "audio": { + "audio_file": "test_realtime_voice_conversation.audio/caller.wav", + "pcm": true, + "offset_bytes": 1920, + "length_bytes": 960 + } + } + }, + { + "direction": "send", + "message": { + "event_id": "688940fc-c37d-46b2-beaf-df032335b090", + "type": "input_audio_buffer.append", + "audio": { + "audio_file": "test_realtime_voice_conversation.audio/caller.wav", + "pcm": true, + "offset_bytes": 2880, + "length_bytes": 960 + } + } + }, + { + "direction": "send", + "message": { + "event_id": "e30c4e33-9aa7-4ab4-89bb-8fc96b702c40", + "type": "input_audio_buffer.append", + "audio": { + "audio_file": "test_realtime_voice_conversation.audio/caller.wav", + "pcm": true, + "offset_bytes": 3840, + "length_bytes": 960 + } + } + }, + { + "direction": "send", + "message": { + "event_id": "a073f6c8-fde5-4355-bf01-08476cddb0e3", + "type": "input_audio_buffer.append", + "audio": { + "audio_file": "test_realtime_voice_conversation.audio/caller.wav", + "pcm": true, + "offset_bytes": 4800, + "length_bytes": 960 + } + } + }, + { + "direction": "receive", + "message": { + "type": "input_audio_buffer.speech_started", + "event_id": "event_EVgfFnlJEQsBTNebtdJR7", + "audio_start_ms": 0, + "item_id": "item_EVgfFVrfdinRe1ZSR1mqw" + } + }, + { + "direction": "receive", + "message": { + "type": "response.done", + "event_id": "event_EVgfF9SV1K6uGIKS8ARtK", + "response": { + "object": "realtime.response", + "id": "resp_EVgfF05xkwLTRprf4ZMGS", + "status": "cancelled", + "status_details": { + "type": "cancelled", + "reason": "turn_detected" + }, + "output": [ + { + "id": "item_EVgfFMwZMFVUydi3vk3kH", + "type": "message", + "status": "completed", + "role": "assistant", + "content": [ + { + "type": "output_audio", + "transcript": "Hello, what is your order number?" + } + ] + } + ], + "conversation_id": "conv_EVgfECJ3whhWkN1aLHqyp", + "output_modalities": [ + "audio" + ], + "max_output_tokens": "inf", + "audio": { + "output": { + "format": { + "type": "audio/pcm", + "rate": 24000 + }, + "voice": "alloy" + } + }, + "usage": { + "total_tokens": 134, + "input_tokens": 75, + "output_tokens": 59, + "input_token_details": { + "text_tokens": 75, + "audio_tokens": 0, + "image_tokens": 0, + "cached_tokens": 0, + "cached_tokens_details": { + "text_tokens": 0, + "audio_tokens": 0, + "image_tokens": 0 + } + }, + "output_token_details": { + "text_tokens": 18, + "audio_tokens": 41 + } + }, + "metadata": null + } + } + }, + { + "direction": "send", + "message": { + "event_id": "02f51c34-3e63-460f-bdb9-d1134186b710", + "type": "input_audio_buffer.append", + "audio": { + "audio_file": "test_realtime_voice_conversation.audio/caller.wav", + "pcm": true, + "offset_bytes": 5760, + "length_bytes": 960 + } + } + }, + { + "direction": "send", + "message": { + "event_id": "aa446dd7-a0a2-4df6-8eca-2da9b2cc2659", + "type": "input_audio_buffer.append", + "audio": { + "audio_file": "test_realtime_voice_conversation.audio/caller.wav", + "pcm": true, + "offset_bytes": 6720, + "length_bytes": 960 + } + } + }, + { + "direction": "send", + "message": { + "event_id": "ca46c49f-8431-40d0-bfb1-1f4e75593d93", + "type": "input_audio_buffer.append", + "audio": { + "audio_file": "test_realtime_voice_conversation.audio/caller.wav", + "pcm": true, + "offset_bytes": 7680, + "length_bytes": 960 + } + } + }, + { + "direction": "send", + "message": { + "event_id": "dae61607-b169-452e-93ca-8ddcbb05d2fd", + "type": "input_audio_buffer.append", + "audio": { + "audio_file": "test_realtime_voice_conversation.audio/caller.wav", + "pcm": true, + "offset_bytes": 8640, + "length_bytes": 960 + } + } + }, + { + "direction": "send", + "message": { + "event_id": "b50454b7-844e-4973-9023-f38390e4b22e", + "type": "input_audio_buffer.append", + "audio": { + "audio_file": "test_realtime_voice_conversation.audio/caller.wav", + "pcm": true, + "offset_bytes": 9600, + "length_bytes": 960 + } + } + }, + { + "direction": "send", + "message": { + "event_id": "b2e58dd2-2bde-4da1-a1c2-67c6a33baca0", + "type": "input_audio_buffer.append", + "audio": { + "audio_file": "test_realtime_voice_conversation.audio/caller.wav", + "pcm": true, + "offset_bytes": 10560, + "length_bytes": 960 + } + } + }, + { + "direction": "send", + "message": { + "event_id": "48c399b3-496b-44f9-a5f1-9baf2d0fc32c", + "type": "input_audio_buffer.append", + "audio": { + "audio_file": "test_realtime_voice_conversation.audio/caller.wav", + "pcm": true, + "offset_bytes": 11520, + "length_bytes": 960 + } + } + }, + { + "direction": "send", + "message": { + "event_id": "116df30c-f3d0-4f99-9a54-551657f144c1", + "type": "input_audio_buffer.append", + "audio": { + "audio_file": "test_realtime_voice_conversation.audio/caller.wav", + "pcm": true, + "offset_bytes": 12480, + "length_bytes": 960 + } + } + }, + { + "direction": "send", + "message": { + "event_id": "68ca726c-1aa5-4322-bbfc-c36f0e086873", + "type": "input_audio_buffer.append", + "audio": { + "audio_file": "test_realtime_voice_conversation.audio/caller.wav", + "pcm": true, + "offset_bytes": 13440, + "length_bytes": 960 + } + } + }, + { + "direction": "send", + "message": { + "event_id": "a67b7195-1555-4196-9e11-7bac06acfbe5", + "type": "input_audio_buffer.append", + "audio": { + "audio_file": "test_realtime_voice_conversation.audio/caller.wav", + "pcm": true, + "offset_bytes": 14400, + "length_bytes": 960 + } + } + }, + { + "direction": "send", + "message": { + "event_id": "b261659e-ae51-498e-88af-363a207dd722", + "type": "input_audio_buffer.append", + "audio": { + "audio_file": "test_realtime_voice_conversation.audio/caller.wav", + "pcm": true, + "offset_bytes": 15360, + "length_bytes": 960 + } + } + }, + { + "direction": "send", + "message": { + "event_id": "6cfd996c-25f2-493a-99c8-ac835739b5be", + "type": "input_audio_buffer.append", + "audio": { + "audio_file": "test_realtime_voice_conversation.audio/caller.wav", + "pcm": true, + "offset_bytes": 16320, + "length_bytes": 960 + } + } + }, + { + "direction": "send", + "message": { + "event_id": "ffa43c26-db0b-4e9f-971c-65e4fdca8fc6", + "type": "input_audio_buffer.append", + "audio": { + "audio_file": "test_realtime_voice_conversation.audio/caller.wav", + "pcm": true, + "offset_bytes": 17280, + "length_bytes": 960 + } + } + }, + { + "direction": "send", + "message": { + "event_id": "bb90a923-c79e-4e6e-9bb9-94ccaf74db91", + "type": "input_audio_buffer.append", + "audio": { + "audio_file": "test_realtime_voice_conversation.audio/caller.wav", + "pcm": true, + "offset_bytes": 18240, + "length_bytes": 960 + } + } + }, + { + "direction": "send", + "message": { + "event_id": "1d603fcf-0134-4ad0-9f31-bdff0b4465de", + "type": "input_audio_buffer.append", + "audio": { + "audio_file": "test_realtime_voice_conversation.audio/caller.wav", + "pcm": true, + "offset_bytes": 19200, + "length_bytes": 960 + } + } + }, + { + "direction": "send", + "message": { + "event_id": "d0aef808-4c4f-4e39-adef-3d53847660b4", + "type": "input_audio_buffer.append", + "audio": { + "audio_file": "test_realtime_voice_conversation.audio/caller.wav", + "pcm": true, + "offset_bytes": 20160, + "length_bytes": 960 + } + } + }, + { + "direction": "send", + "message": { + "event_id": "ab889c3c-7617-4bc9-b900-3d4e77920384", + "type": "input_audio_buffer.append", + "audio": { + "audio_file": "test_realtime_voice_conversation.audio/caller.wav", + "pcm": true, + "offset_bytes": 21120, + "length_bytes": 960 + } + } + }, + { + "direction": "send", + "message": { + "event_id": "18de6373-e9fa-4c13-ae23-1f4eae985b96", + "type": "input_audio_buffer.append", + "audio": { + "audio_file": "test_realtime_voice_conversation.audio/caller.wav", + "pcm": true, + "offset_bytes": 22080, + "length_bytes": 960 + } + } + }, + { + "direction": "send", + "message": { + "event_id": "87724c6e-c1d3-4f63-b160-d8d5e74f4655", + "type": "input_audio_buffer.append", + "audio": { + "audio_file": "test_realtime_voice_conversation.audio/caller.wav", + "pcm": true, + "offset_bytes": 23040, + "length_bytes": 960 + } + } + }, + { + "direction": "send", + "message": { + "event_id": "d5a1b5cb-ac38-4e73-8523-8202c9190ac7", + "type": "input_audio_buffer.append", + "audio": { + "audio_file": "test_realtime_voice_conversation.audio/caller.wav", + "pcm": true, + "offset_bytes": 24000, + "length_bytes": 960 + } + } + }, + { + "direction": "send", + "message": { + "event_id": "c58e7e59-c0c9-4dfb-80fe-46fcf435ac30", + "type": "input_audio_buffer.append", + "audio": { + "audio_file": "test_realtime_voice_conversation.audio/caller.wav", + "pcm": true, + "offset_bytes": 24960, + "length_bytes": 960 + } + } + }, + { + "direction": "send", + "message": { + "event_id": "61cfe423-8034-446b-9363-1bece3f05667", + "type": "input_audio_buffer.append", + "audio": { + "audio_file": "test_realtime_voice_conversation.audio/caller.wav", + "pcm": true, + "offset_bytes": 25920, + "length_bytes": 960 + } + } + }, + { + "direction": "send", + "message": { + "event_id": "4e83dc99-6495-4b24-b5ee-5386b5579908", + "type": "input_audio_buffer.append", + "audio": { + "audio_file": "test_realtime_voice_conversation.audio/caller.wav", + "pcm": true, + "offset_bytes": 26880, + "length_bytes": 960 + } + } + }, + { + "direction": "send", + "message": { + "event_id": "3cb52bc2-6f59-47ec-9a7f-e1a85d5d3132", + "type": "input_audio_buffer.append", + "audio": { + "audio_file": "test_realtime_voice_conversation.audio/caller.wav", + "pcm": true, + "offset_bytes": 27840, + "length_bytes": 960 + } + } + }, + { + "direction": "send", + "message": { + "event_id": "563b6b6a-0348-4de7-8c2d-6784d051868f", + "type": "input_audio_buffer.append", + "audio": { + "audio_file": "test_realtime_voice_conversation.audio/caller.wav", + "pcm": true, + "offset_bytes": 28800, + "length_bytes": 960 + } + } + }, + { + "direction": "send", + "message": { + "event_id": "a032a0bc-6b56-4adf-aef3-1ca158b0b5b4", + "type": "input_audio_buffer.append", + "audio": { + "audio_file": "test_realtime_voice_conversation.audio/caller.wav", + "pcm": true, + "offset_bytes": 29760, + "length_bytes": 960 + } + } + }, + { + "direction": "send", + "message": { + "event_id": "6f25b8e1-d8ba-40af-89d8-f3f098636fa2", + "type": "input_audio_buffer.append", + "audio": { + "audio_file": "test_realtime_voice_conversation.audio/caller.wav", + "pcm": true, + "offset_bytes": 30720, + "length_bytes": 960 + } + } + }, + { + "direction": "send", + "message": { + "event_id": "105a642c-5289-4796-beab-5dd48c21bc3f", + "type": "input_audio_buffer.append", + "audio": { + "audio_file": "test_realtime_voice_conversation.audio/caller.wav", + "pcm": true, + "offset_bytes": 31680, + "length_bytes": 960 + } + } + }, + { + "direction": "send", + "message": { + "event_id": "9fdba103-9e5b-47db-8cdc-e062ad975628", + "type": "input_audio_buffer.append", + "audio": { + "audio_file": "test_realtime_voice_conversation.audio/caller.wav", + "pcm": true, + "offset_bytes": 32640, + "length_bytes": 960 + } + } + }, + { + "direction": "send", + "message": { + "event_id": "32966952-a4b3-4332-9556-55c64acf4a5a", + "type": "input_audio_buffer.append", + "audio": { + "audio_file": "test_realtime_voice_conversation.audio/caller.wav", + "pcm": true, + "offset_bytes": 33600, + "length_bytes": 960 + } + } + }, + { + "direction": "send", + "message": { + "event_id": "15b3d18a-c39e-4260-a103-32f94a040254", + "type": "input_audio_buffer.append", + "audio": { + "audio_file": "test_realtime_voice_conversation.audio/caller.wav", + "pcm": true, + "offset_bytes": 34560, + "length_bytes": 960 + } + } + }, + { + "direction": "send", + "message": { + "event_id": "cd0ea8b0-c0d1-4f8b-b97a-7b9b6a31460e", + "type": "input_audio_buffer.append", + "audio": { + "audio_file": "test_realtime_voice_conversation.audio/caller.wav", + "pcm": true, + "offset_bytes": 35520, + "length_bytes": 960 + } + } + }, + { + "direction": "send", + "message": { + "event_id": "d5f54928-f425-46f4-b147-4c4dcd47de20", + "type": "input_audio_buffer.append", + "audio": { + "audio_file": "test_realtime_voice_conversation.audio/caller.wav", + "pcm": true, + "offset_bytes": 36480, + "length_bytes": 960 + } + } + }, + { + "direction": "send", + "message": { + "event_id": "479c61f4-43b0-4582-96ba-4388140c30da", + "type": "input_audio_buffer.append", + "audio": { + "audio_file": "test_realtime_voice_conversation.audio/caller.wav", + "pcm": true, + "offset_bytes": 37440, + "length_bytes": 960 + } + } + }, + { + "direction": "send", + "message": { + "event_id": "9b937a7c-1649-4abd-a162-e145e308be5d", + "type": "input_audio_buffer.append", + "audio": { + "audio_file": "test_realtime_voice_conversation.audio/caller.wav", + "pcm": true, + "offset_bytes": 38400, + "length_bytes": 960 + } + } + }, + { + "direction": "send", + "message": { + "event_id": "10765989-e813-4373-b31a-603265f797a8", + "type": "input_audio_buffer.append", + "audio": { + "audio_file": "test_realtime_voice_conversation.audio/caller.wav", + "pcm": true, + "offset_bytes": 39360, + "length_bytes": 960 + } + } + }, + { + "direction": "send", + "message": { + "event_id": "17f9a3ff-d3dc-44d2-98ff-91f9c8924639", + "type": "input_audio_buffer.append", + "audio": { + "audio_file": "test_realtime_voice_conversation.audio/caller.wav", + "pcm": true, + "offset_bytes": 40320, + "length_bytes": 960 + } + } + }, + { + "direction": "send", + "message": { + "event_id": "210cd4ca-7046-45f3-b096-27fd0123fce2", + "type": "input_audio_buffer.append", + "audio": { + "audio_file": "test_realtime_voice_conversation.audio/caller.wav", + "pcm": true, + "offset_bytes": 41280, + "length_bytes": 960 + } + } + }, + { + "direction": "send", + "message": { + "event_id": "7e76ebc7-78a9-41e7-87e3-9440c6aaf84d", + "type": "input_audio_buffer.append", + "audio": { + "audio_file": "test_realtime_voice_conversation.audio/caller.wav", + "pcm": true, + "offset_bytes": 42240, + "length_bytes": 960 + } + } + }, + { + "direction": "send", + "message": { + "event_id": "28bb66af-98cf-430e-ab09-6f5114a0e241", + "type": "input_audio_buffer.append", + "audio": { + "audio_file": "test_realtime_voice_conversation.audio/caller.wav", + "pcm": true, + "offset_bytes": 43200, + "length_bytes": 960 + } + } + }, + { + "direction": "send", + "message": { + "event_id": "436ac059-e477-441a-8006-de79444f2331", + "type": "input_audio_buffer.append", + "audio": { + "audio_file": "test_realtime_voice_conversation.audio/caller.wav", + "pcm": true, + "offset_bytes": 44160, + "length_bytes": 960 + } + } + }, + { + "direction": "send", + "message": { + "event_id": "f8450ed2-fe8a-402d-a1c2-29076611569e", + "type": "input_audio_buffer.append", + "audio": { + "audio_file": "test_realtime_voice_conversation.audio/caller.wav", + "pcm": true, + "offset_bytes": 45120, + "length_bytes": 960 + } + } + }, + { + "direction": "send", + "message": { + "event_id": "493ebafe-5f6f-4951-8507-1c8183e0c9fa", + "type": "input_audio_buffer.append", + "audio": { + "audio_file": "test_realtime_voice_conversation.audio/caller.wav", + "pcm": true, + "offset_bytes": 46080, + "length_bytes": 960 + } + } + }, + { + "direction": "send", + "message": { + "event_id": "47c146d0-4e31-4749-886f-f1045e048e6c", + "type": "input_audio_buffer.append", + "audio": { + "audio_file": "test_realtime_voice_conversation.audio/caller.wav", + "pcm": true, + "offset_bytes": 47040, + "length_bytes": 960 + } + } + }, + { + "direction": "send", + "message": { + "event_id": "929d26fa-1421-446a-9ee4-b2a97e776bc2", + "type": "input_audio_buffer.append", + "audio": { + "audio_file": "test_realtime_voice_conversation.audio/caller.wav", + "pcm": true, + "offset_bytes": 48000, + "length_bytes": 960 + } + } + }, + { + "direction": "send", + "message": { + "event_id": "d45d3cff-c768-4957-96e2-b7bd2267f520", + "type": "input_audio_buffer.append", + "audio": { + "audio_file": "test_realtime_voice_conversation.audio/caller.wav", + "pcm": true, + "offset_bytes": 48960, + "length_bytes": 960 + } + } + }, + { + "direction": "send", + "message": { + "event_id": "173a79d6-5bdb-4054-b6be-87ea572cfc34", + "type": "input_audio_buffer.append", + "audio": { + "audio_file": "test_realtime_voice_conversation.audio/caller.wav", + "pcm": true, + "offset_bytes": 49920, + "length_bytes": 960 + } + } + }, + { + "direction": "send", + "message": { + "event_id": "5ea459ab-def3-4303-bc7d-84709eea1039", + "type": "input_audio_buffer.append", + "audio": { + "audio_file": "test_realtime_voice_conversation.audio/caller.wav", + "pcm": true, + "offset_bytes": 50880, + "length_bytes": 960 + } + } + }, + { + "direction": "send", + "message": { + "event_id": "5fa28be3-ab00-40ff-8c87-3df275e5092c", + "type": "input_audio_buffer.append", + "audio": { + "audio_file": "test_realtime_voice_conversation.audio/caller.wav", + "pcm": true, + "offset_bytes": 51840, + "length_bytes": 960 + } + } + }, + { + "direction": "send", + "message": { + "event_id": "98771546-703d-40e6-8807-e987fcf437f5", + "type": "input_audio_buffer.append", + "audio": { + "audio_file": "test_realtime_voice_conversation.audio/caller.wav", + "pcm": true, + "offset_bytes": 52800, + "length_bytes": 960 + } + } + }, + { + "direction": "send", + "message": { + "event_id": "76e0c0cf-8bb4-4e6e-acb1-ed92b9a1686f", + "type": "input_audio_buffer.append", + "audio": { + "audio_file": "test_realtime_voice_conversation.audio/caller.wav", + "pcm": true, + "offset_bytes": 53760, + "length_bytes": 960 + } + } + }, + { + "direction": "send", + "message": { + "event_id": "fb929e4a-afee-42d6-a2ad-43c2793047fc", + "type": "input_audio_buffer.append", + "audio": { + "audio_file": "test_realtime_voice_conversation.audio/caller.wav", + "pcm": true, + "offset_bytes": 54720, + "length_bytes": 960 + } + } + }, + { + "direction": "send", + "message": { + "event_id": "51975e9d-670c-45fa-9072-9f568a45449d", + "type": "input_audio_buffer.append", + "audio": { + "audio_file": "test_realtime_voice_conversation.audio/caller.wav", + "pcm": true, + "offset_bytes": 55680, + "length_bytes": 960 + } + } + }, + { + "direction": "send", + "message": { + "event_id": "83a01c70-740e-43ae-adb6-eab20f2e66e7", + "type": "input_audio_buffer.append", + "audio": { + "audio_file": "test_realtime_voice_conversation.audio/caller.wav", + "pcm": true, + "offset_bytes": 56640, + "length_bytes": 960 + } + } + }, + { + "direction": "send", + "message": { + "event_id": "c0fe50b9-b7e9-4221-bd73-5c2d600920d4", + "type": "input_audio_buffer.append", + "audio": { + "audio_file": "test_realtime_voice_conversation.audio/caller.wav", + "pcm": true, + "offset_bytes": 57600, + "length_bytes": 960 + } + } + }, + { + "direction": "send", + "message": { + "event_id": "69128bdf-e775-4bff-8786-29c181202991", + "type": "input_audio_buffer.append", + "audio": { + "audio_file": "test_realtime_voice_conversation.audio/caller.wav", + "pcm": true, + "offset_bytes": 58560, + "length_bytes": 960 + } + } + }, + { + "direction": "send", + "message": { + "event_id": "f61c7afa-8f20-481a-99d5-5edc0c8c5435", + "type": "input_audio_buffer.append", + "audio": { + "audio_file": "test_realtime_voice_conversation.audio/caller.wav", + "pcm": true, + "offset_bytes": 59520, + "length_bytes": 960 + } + } + }, + { + "direction": "send", + "message": { + "event_id": "55ba71b5-b898-4bd3-943c-c155a5983472", + "type": "input_audio_buffer.append", + "audio": { + "audio_file": "test_realtime_voice_conversation.audio/caller.wav", + "pcm": true, + "offset_bytes": 60480, + "length_bytes": 960 + } + } + }, + { + "direction": "send", + "message": { + "event_id": "6a52f532-dd47-45a7-929b-8bd85f3dbe53", + "type": "input_audio_buffer.append", + "audio": { + "audio_file": "test_realtime_voice_conversation.audio/caller.wav", + "pcm": true, + "offset_bytes": 61440, + "length_bytes": 960 + } + } + }, + { + "direction": "send", + "message": { + "event_id": "409c85de-761e-4878-822e-6c9608444c20", + "type": "input_audio_buffer.append", + "audio": { + "audio_file": "test_realtime_voice_conversation.audio/caller.wav", + "pcm": true, + "offset_bytes": 62400, + "length_bytes": 960 + } + } + }, + { + "direction": "send", + "message": { + "event_id": "d6d1eaae-930d-49f6-a9f1-047b9699f370", + "type": "input_audio_buffer.append", + "audio": { + "audio_file": "test_realtime_voice_conversation.audio/caller.wav", + "pcm": true, + "offset_bytes": 63360, + "length_bytes": 960 + } + } + }, + { + "direction": "send", + "message": { + "event_id": "33482a0a-187a-41ee-9f92-66b72ae503c3", + "type": "input_audio_buffer.append", + "audio": { + "audio_file": "test_realtime_voice_conversation.audio/caller.wav", + "pcm": true, + "offset_bytes": 64320, + "length_bytes": 960 + } + } + }, + { + "direction": "send", + "message": { + "event_id": "58897aef-5362-4a74-9151-293f22f2c3a1", + "type": "input_audio_buffer.append", + "audio": { + "audio_file": "test_realtime_voice_conversation.audio/caller.wav", + "pcm": true, + "offset_bytes": 65280, + "length_bytes": 960 + } + } + }, + { + "direction": "send", + "message": { + "event_id": "cf2c4ca1-fa37-42b2-a613-748c03af1cc0", + "type": "input_audio_buffer.append", + "audio": { + "audio_file": "test_realtime_voice_conversation.audio/caller.wav", + "pcm": true, + "offset_bytes": 66240, + "length_bytes": 960 + } + } + }, + { + "direction": "send", + "message": { + "event_id": "a2c7b986-6f1a-4b20-92ea-c651692f6e26", + "type": "input_audio_buffer.append", + "audio": { + "audio_file": "test_realtime_voice_conversation.audio/caller.wav", + "pcm": true, + "offset_bytes": 67200, + "length_bytes": 960 + } + } + }, + { + "direction": "send", + "message": { + "event_id": "473b7c86-3a14-4ef9-9010-75e0d827a828", + "type": "input_audio_buffer.append", + "audio": { + "audio_file": "test_realtime_voice_conversation.audio/caller.wav", + "pcm": true, + "offset_bytes": 68160, + "length_bytes": 960 + } + } + }, + { + "direction": "send", + "message": { + "event_id": "47938415-a4bc-421b-967e-e073872ce6f2", + "type": "input_audio_buffer.append", + "audio": { + "audio_file": "test_realtime_voice_conversation.audio/caller.wav", + "pcm": true, + "offset_bytes": 69120, + "length_bytes": 960 + } + } + }, + { + "direction": "send", + "message": { + "event_id": "b9c5ffab-d406-43ad-876e-3b5065b08282", + "type": "input_audio_buffer.append", + "audio": { + "audio_file": "test_realtime_voice_conversation.audio/caller.wav", + "pcm": true, + "offset_bytes": 70080, + "length_bytes": 960 + } + } + }, + { + "direction": "send", + "message": { + "event_id": "cee8c170-4506-419b-b001-d3d8ac1607d7", + "type": "input_audio_buffer.append", + "audio": { + "audio_file": "test_realtime_voice_conversation.audio/caller.wav", + "pcm": true, + "offset_bytes": 71040, + "length_bytes": 960 + } + } + }, + { + "direction": "send", + "message": { + "event_id": "e799c112-03e2-4901-a8a8-bab1ed5bfea8", + "type": "input_audio_buffer.append", + "audio": { + "audio_file": "test_realtime_voice_conversation.audio/caller.wav", + "pcm": true, + "offset_bytes": 72000, + "length_bytes": 960 + } + } + }, + { + "direction": "send", + "message": { + "event_id": "33e2c2ef-81bd-4ee5-8e5e-3180b9c23c5a", + "type": "input_audio_buffer.append", + "audio": { + "audio_file": "test_realtime_voice_conversation.audio/caller.wav", + "pcm": true, + "offset_bytes": 72960, + "length_bytes": 960 + } + } + }, + { + "direction": "send", + "message": { + "event_id": "109cc7bd-38ef-40ea-9bd9-df54352dea8e", + "type": "input_audio_buffer.append", + "audio": { + "audio_file": "test_realtime_voice_conversation.audio/caller.wav", + "pcm": true, + "offset_bytes": 73920, + "length_bytes": 960 + } + } + }, + { + "direction": "send", + "message": { + "event_id": "db800133-e793-487f-82f6-5e1ca6e148ff", + "type": "input_audio_buffer.append", + "audio": { + "audio_file": "test_realtime_voice_conversation.audio/caller.wav", + "pcm": true, + "offset_bytes": 74880, + "length_bytes": 960 + } + } + }, + { + "direction": "send", + "message": { + "event_id": "7add7dcd-f373-43f2-b60b-c2ec1d5f5d28", + "type": "input_audio_buffer.append", + "audio": { + "audio_file": "test_realtime_voice_conversation.audio/caller.wav", + "pcm": true, + "offset_bytes": 75840, + "length_bytes": 960 + } + } + }, + { + "direction": "send", + "message": { + "event_id": "dd6f1c49-aad1-44a3-b804-ee6f9f40edd1", + "type": "input_audio_buffer.append", + "audio": { + "audio_file": "test_realtime_voice_conversation.audio/caller.wav", + "pcm": true, + "offset_bytes": 76800, + "length_bytes": 960 + } + } + }, + { + "direction": "send", + "message": { + "event_id": "ee4e621e-1444-4aaf-9d70-de8ee90b46d8", + "type": "input_audio_buffer.append", + "audio": { + "audio_file": "test_realtime_voice_conversation.audio/caller.wav", + "pcm": true, + "offset_bytes": 77760, + "length_bytes": 960 + } + } + }, + { + "direction": "send", + "message": { + "event_id": "501a95e7-f776-45de-a65d-7c8a7f6791fa", + "type": "input_audio_buffer.append", + "audio": { + "audio_file": "test_realtime_voice_conversation.audio/caller.wav", + "pcm": true, + "offset_bytes": 78720, + "length_bytes": 960 + } + } + }, + { + "direction": "send", + "message": { + "event_id": "19f5bcef-e170-46f9-869c-07890c9e62b9", + "type": "input_audio_buffer.append", + "audio": { + "audio_file": "test_realtime_voice_conversation.audio/caller.wav", + "pcm": true, + "offset_bytes": 79680, + "length_bytes": 960 + } + } + }, + { + "direction": "send", + "message": { + "event_id": "784b8432-c9bd-40b5-8fe9-ea95fd5c740f", + "type": "input_audio_buffer.append", + "audio": { + "audio_file": "test_realtime_voice_conversation.audio/caller.wav", + "pcm": true, + "offset_bytes": 80640, + "length_bytes": 960 + } + } + }, + { + "direction": "send", + "message": { + "event_id": "1e827666-9fe3-48dd-940b-7d1322ab3ce0", + "type": "input_audio_buffer.append", + "audio": { + "audio_file": "test_realtime_voice_conversation.audio/caller.wav", + "pcm": true, + "offset_bytes": 81600, + "length_bytes": 960 + } + } + }, + { + "direction": "send", + "message": { + "event_id": "c1e38745-37d1-41c6-a421-03253f140824", + "type": "input_audio_buffer.append", + "audio": { + "audio_file": "test_realtime_voice_conversation.audio/caller.wav", + "pcm": true, + "offset_bytes": 82560, + "length_bytes": 960 + } + } + }, + { + "direction": "send", + "message": { + "event_id": "30820b77-2929-4ac5-aa8d-f6dd3e2efebd", + "type": "input_audio_buffer.append", + "audio": { + "audio_file": "test_realtime_voice_conversation.audio/caller.wav", + "pcm": true, + "offset_bytes": 83520, + "length_bytes": 960 + } + } + }, + { + "direction": "send", + "message": { + "event_id": "a8f3e4b2-9f62-48a2-8a0f-d57ef52e246d", + "type": "input_audio_buffer.append", + "audio": { + "audio_file": "test_realtime_voice_conversation.audio/caller.wav", + "pcm": true, + "offset_bytes": 84480, + "length_bytes": 960 + } + } + }, + { + "direction": "send", + "message": { + "event_id": "e8cd74e6-007c-46ca-aca9-c8624ec6a785", + "type": "input_audio_buffer.append", + "audio": { + "audio_file": "test_realtime_voice_conversation.audio/caller.wav", + "pcm": true, + "offset_bytes": 85440, + "length_bytes": 960 + } + } + }, + { + "direction": "send", + "message": { + "event_id": "fd0ebeb3-c3aa-44ec-bac6-88defd19572d", + "type": "input_audio_buffer.append", + "audio": { + "audio_file": "test_realtime_voice_conversation.audio/caller.wav", + "pcm": true, + "offset_bytes": 86400, + "length_bytes": 960 + } + } + }, + { + "direction": "send", + "message": { + "event_id": "9d4b0138-16fd-4ffc-bfb5-718509091e9c", + "type": "input_audio_buffer.append", + "audio": { + "audio_file": "test_realtime_voice_conversation.audio/caller.wav", + "pcm": true, + "offset_bytes": 87360, + "length_bytes": 960 + } + } + }, + { + "direction": "send", + "message": { + "event_id": "45abb5c4-4e19-4617-9467-358679752c2f", + "type": "input_audio_buffer.append", + "audio": { + "audio_file": "test_realtime_voice_conversation.audio/caller.wav", + "pcm": true, + "offset_bytes": 88320, + "length_bytes": 960 + } + } + }, + { + "direction": "send", + "message": { + "event_id": "3d713e3a-5168-4867-9581-381bb672f08c", + "type": "input_audio_buffer.append", + "audio": { + "audio_file": "test_realtime_voice_conversation.audio/caller.wav", + "pcm": true, + "offset_bytes": 89280, + "length_bytes": 960 + } + } + }, + { + "direction": "send", + "message": { + "event_id": "3f147f5a-8be3-48e6-ac96-524408176c5e", + "type": "input_audio_buffer.append", + "audio": { + "audio_file": "test_realtime_voice_conversation.audio/caller.wav", + "pcm": true, + "offset_bytes": 90240, + "length_bytes": 960 + } + } + }, + { + "direction": "send", + "message": { + "event_id": "20bdddcb-3d4d-484f-954f-a4855b9cfd2c", + "type": "input_audio_buffer.append", + "audio": { + "audio_file": "test_realtime_voice_conversation.audio/caller.wav", + "pcm": true, + "offset_bytes": 91200, + "length_bytes": 960 + } + } + }, + { + "direction": "send", + "message": { + "event_id": "4cbef2fc-3d46-431a-9cde-2dc9440d442d", + "type": "input_audio_buffer.append", + "audio": { + "audio_file": "test_realtime_voice_conversation.audio/caller.wav", + "pcm": true, + "offset_bytes": 92160, + "length_bytes": 960 + } + } + }, + { + "direction": "send", + "message": { + "event_id": "95813d97-8134-48c6-a97c-c669bba688f7", + "type": "input_audio_buffer.append", + "audio": { + "audio_file": "test_realtime_voice_conversation.audio/caller.wav", + "pcm": true, + "offset_bytes": 93120, + "length_bytes": 960 + } + } + }, + { + "direction": "send", + "message": { + "event_id": "69b22e4a-d8cc-4cad-8127-04615471e4c5", + "type": "input_audio_buffer.append", + "audio": { + "audio_file": "test_realtime_voice_conversation.audio/caller.wav", + "pcm": true, + "offset_bytes": 94080, + "length_bytes": 960 + } + } + }, + { + "direction": "send", + "message": { + "event_id": "1fecc163-47b3-40f4-a8c3-0f518a18cab2", + "type": "input_audio_buffer.append", + "audio": { + "audio_file": "test_realtime_voice_conversation.audio/caller.wav", + "pcm": true, + "offset_bytes": 95040, + "length_bytes": 960 + } + } + }, + { + "direction": "send", + "message": { + "event_id": "47d942b6-b93c-4688-9911-2e20c3c93acd", + "type": "input_audio_buffer.append", + "audio": { + "audio_file": "test_realtime_voice_conversation.audio/caller.wav", + "pcm": true, + "offset_bytes": 96000, + "length_bytes": 960 + } + } + }, + { + "direction": "send", + "message": { + "event_id": "381702ab-f3d5-41d4-a63e-72c5b141e0a0", + "type": "input_audio_buffer.append", + "audio": { + "audio_file": "test_realtime_voice_conversation.audio/caller.wav", + "pcm": true, + "offset_bytes": 96960, + "length_bytes": 960 + } + } + }, + { + "direction": "send", + "message": { + "event_id": "4baa6457-dd12-44f6-ad76-1957fc3d2cc2", + "type": "input_audio_buffer.append", + "audio": { + "audio_file": "test_realtime_voice_conversation.audio/caller.wav", + "pcm": true, + "offset_bytes": 97920, + "length_bytes": 960 + } + } + }, + { + "direction": "send", + "message": { + "event_id": "bfbb2d54-6a0b-4ac6-a37b-6e34725ac218", + "type": "input_audio_buffer.append", + "audio": { + "audio_file": "test_realtime_voice_conversation.audio/caller.wav", + "pcm": true, + "offset_bytes": 98880, + "length_bytes": 960 + } + } + }, + { + "direction": "send", + "message": { + "event_id": "012fdf16-26d6-4496-80cd-506bb91ac503", + "type": "input_audio_buffer.append", + "audio": { + "audio_file": "test_realtime_voice_conversation.audio/caller.wav", + "pcm": true, + "offset_bytes": 99840, + "length_bytes": 960 + } + } + }, + { + "direction": "send", + "message": { + "event_id": "d370cd2b-2062-4198-9b6e-2a9ea4d32013", + "type": "input_audio_buffer.append", + "audio": { + "audio_file": "test_realtime_voice_conversation.audio/caller.wav", + "pcm": true, + "offset_bytes": 100800, + "length_bytes": 960 + } + } + }, + { + "direction": "send", + "message": { + "event_id": "ab131834-dfc9-4cb5-a888-162117269a4e", + "type": "input_audio_buffer.append", + "audio": { + "audio_file": "test_realtime_voice_conversation.audio/caller.wav", + "pcm": true, + "offset_bytes": 101760, + "length_bytes": 960 + } + } + }, + { + "direction": "send", + "message": { + "event_id": "abad5e75-e88c-49b0-9252-26580cbf984e", + "type": "input_audio_buffer.append", + "audio": { + "audio_file": "test_realtime_voice_conversation.audio/caller.wav", + "pcm": true, + "offset_bytes": 102720, + "length_bytes": 960 + } + } + }, + { + "direction": "send", + "message": { + "event_id": "a3636f27-f1f5-4640-ba16-5c4c2bc75006", + "type": "input_audio_buffer.append", + "audio": { + "audio_file": "test_realtime_voice_conversation.audio/caller.wav", + "pcm": true, + "offset_bytes": 103680, + "length_bytes": 960 + } + } + }, + { + "direction": "send", + "message": { + "event_id": "c2383ec2-dc15-4a13-beae-bb63d01d5e35", + "type": "input_audio_buffer.append", + "audio": { + "audio_file": "test_realtime_voice_conversation.audio/caller.wav", + "pcm": true, + "offset_bytes": 104640, + "length_bytes": 960 + } + } + }, + { + "direction": "send", + "message": { + "event_id": "e4f55448-8051-4791-83d1-3ba9f092127f", + "type": "input_audio_buffer.append", + "audio": { + "audio_file": "test_realtime_voice_conversation.audio/caller.wav", + "pcm": true, + "offset_bytes": 105600, + "length_bytes": 960 + } + } + }, + { + "direction": "send", + "message": { + "event_id": "0e72d30e-da7a-4c4a-a2cb-fff1932378b9", + "type": "input_audio_buffer.append", + "audio": { + "audio_file": "test_realtime_voice_conversation.audio/caller.wav", + "pcm": true, + "offset_bytes": 106560, + "length_bytes": 960 + } + } + }, + { + "direction": "send", + "message": { + "event_id": "72a0a6c4-589d-4a4a-8549-095b7a41431a", + "type": "input_audio_buffer.append", + "audio": { + "audio_file": "test_realtime_voice_conversation.audio/caller.wav", + "pcm": true, + "offset_bytes": 107520, + "length_bytes": 960 + } + } + }, + { + "direction": "send", + "message": { + "event_id": "d87f507d-0f4a-4fb3-93b9-acfbdc9a6186", + "type": "input_audio_buffer.append", + "audio": { + "audio_file": "test_realtime_voice_conversation.audio/caller.wav", + "pcm": true, + "offset_bytes": 108480, + "length_bytes": 960 + } + } + }, + { + "direction": "send", + "message": { + "event_id": "b90f5da7-b503-4ade-8c11-62bf30c05498", + "type": "input_audio_buffer.append", + "audio": { + "audio_file": "test_realtime_voice_conversation.audio/caller.wav", + "pcm": true, + "offset_bytes": 109440, + "length_bytes": 960 + } + } + }, + { + "direction": "send", + "message": { + "event_id": "ce78e385-3124-4a55-9e75-22cfd24ef286", + "type": "input_audio_buffer.append", + "audio": { + "audio_file": "test_realtime_voice_conversation.audio/caller.wav", + "pcm": true, + "offset_bytes": 110400, + "length_bytes": 960 + } + } + }, + { + "direction": "send", + "message": { + "event_id": "ff873656-0410-4d6e-bf30-6aa64b0c9d08", + "type": "input_audio_buffer.append", + "audio": { + "audio_file": "test_realtime_voice_conversation.audio/caller.wav", + "pcm": true, + "offset_bytes": 111360, + "length_bytes": 960 + } + } + }, + { + "direction": "send", + "message": { + "event_id": "c4b85220-ed28-48c7-88fd-d0e3f9df0ffe", + "type": "input_audio_buffer.append", + "audio": { + "audio_file": "test_realtime_voice_conversation.audio/caller.wav", + "pcm": true, + "offset_bytes": 112320, + "length_bytes": 960 + } + } + }, + { + "direction": "send", + "message": { + "event_id": "91652b74-c891-49ea-8bb6-dbd9998d02e4", + "type": "input_audio_buffer.append", + "audio": { + "audio_file": "test_realtime_voice_conversation.audio/caller.wav", + "pcm": true, + "offset_bytes": 113280, + "length_bytes": 960 + } + } + }, + { + "direction": "send", + "message": { + "event_id": "e4305a2f-e0df-46b5-b851-92105d5c3e0b", + "type": "input_audio_buffer.append", + "audio": { + "audio_file": "test_realtime_voice_conversation.audio/caller.wav", + "pcm": true, + "offset_bytes": 114240, + "length_bytes": 960 + } + } + }, + { + "direction": "send", + "message": { + "event_id": "b9807c6e-c129-4caa-86de-50c3b5597d86", + "type": "input_audio_buffer.append", + "audio": { + "audio_file": "test_realtime_voice_conversation.audio/caller.wav", + "pcm": true, + "offset_bytes": 115200, + "length_bytes": 960 + } + } + }, + { + "direction": "send", + "message": { + "event_id": "3ce37cf8-5330-404f-8608-166e99dd10ea", + "type": "input_audio_buffer.append", + "audio": { + "audio_file": "test_realtime_voice_conversation.audio/caller.wav", + "pcm": true, + "offset_bytes": 116160, + "length_bytes": 960 + } + } + }, + { + "direction": "send", + "message": { + "event_id": "ddb7ed48-ab3b-4f5f-9fed-760b44520c48", + "type": "input_audio_buffer.append", + "audio": { + "audio_file": "test_realtime_voice_conversation.audio/caller.wav", + "pcm": true, + "offset_bytes": 117120, + "length_bytes": 960 + } + } + }, + { + "direction": "send", + "message": { + "event_id": "c355a218-258b-4e7c-bac2-41bc0bb95865", + "type": "input_audio_buffer.append", + "audio": { + "audio_file": "test_realtime_voice_conversation.audio/caller.wav", + "pcm": true, + "offset_bytes": 118080, + "length_bytes": 960 + } + } + }, + { + "direction": "send", + "message": { + "event_id": "8a091b74-016a-4722-9c6c-a401285ab1d9", + "type": "input_audio_buffer.append", + "audio": { + "audio_file": "test_realtime_voice_conversation.audio/caller.wav", + "pcm": true, + "offset_bytes": 119040, + "length_bytes": 960 + } + } + }, + { + "direction": "send", + "message": { + "event_id": "f197850d-76fe-4803-b222-39d0a617bd0f", + "type": "input_audio_buffer.append", + "audio": { + "audio_file": "test_realtime_voice_conversation.audio/caller.wav", + "pcm": true, + "offset_bytes": 120000, + "length_bytes": 960 + } + } + }, + { + "direction": "send", + "message": { + "event_id": "5cb90ef0-5eab-4588-917d-333f719ae17e", + "type": "input_audio_buffer.append", + "audio": { + "audio_file": "test_realtime_voice_conversation.audio/caller.wav", + "pcm": true, + "offset_bytes": 120960, + "length_bytes": 960 + } + } + }, + { + "direction": "send", + "message": { + "event_id": "db26ef05-a2a2-43cb-aafb-219c99fa81dc", + "type": "input_audio_buffer.append", + "audio": { + "audio_file": "test_realtime_voice_conversation.audio/caller.wav", + "pcm": true, + "offset_bytes": 121920, + "length_bytes": 960 + } + } + }, + { + "direction": "send", + "message": { + "event_id": "f3ea7ef2-f13e-4313-a4c9-2552a445c624", + "type": "input_audio_buffer.append", + "audio": { + "audio_file": "test_realtime_voice_conversation.audio/caller.wav", + "pcm": true, + "offset_bytes": 122880, + "length_bytes": 960 + } + } + }, + { + "direction": "send", + "message": { + "event_id": "5e01c98d-de14-467d-94c1-8ddd11a5683b", + "type": "input_audio_buffer.append", + "audio": { + "audio_file": "test_realtime_voice_conversation.audio/caller.wav", + "pcm": true, + "offset_bytes": 123840, + "length_bytes": 960 + } + } + }, + { + "direction": "send", + "message": { + "event_id": "a83af831-3a30-4c1c-a178-103db8db5b28", + "type": "input_audio_buffer.append", + "audio": { + "audio_file": "test_realtime_voice_conversation.audio/caller.wav", + "pcm": true, + "offset_bytes": 124800, + "length_bytes": 960 + } + } + }, + { + "direction": "send", + "message": { + "event_id": "dcce341c-5058-4e07-bec5-1df33cd1177a", + "type": "input_audio_buffer.append", + "audio": { + "audio_file": "test_realtime_voice_conversation.audio/caller.wav", + "pcm": true, + "offset_bytes": 125760, + "length_bytes": 960 + } + } + }, + { + "direction": "send", + "message": { + "event_id": "a9846961-5086-4090-9dba-cf2c5ff52629", + "type": "input_audio_buffer.append", + "audio": { + "audio_file": "test_realtime_voice_conversation.audio/caller.wav", + "pcm": true, + "offset_bytes": 126720, + "length_bytes": 960 + } + } + }, + { + "direction": "send", + "message": { + "event_id": "cf774499-6190-4176-8f64-95cf20bcb045", + "type": "input_audio_buffer.append", + "audio": { + "audio_file": "test_realtime_voice_conversation.audio/caller.wav", + "pcm": true, + "offset_bytes": 127680, + "length_bytes": 960 + } + } + }, + { + "direction": "send", + "message": { + "event_id": "babedae6-8a9c-476c-9fcd-192500600f23", + "type": "input_audio_buffer.append", + "audio": { + "audio_file": "test_realtime_voice_conversation.audio/caller.wav", + "pcm": true, + "offset_bytes": 128640, + "length_bytes": 960 + } + } + }, + { + "direction": "send", + "message": { + "event_id": "ed9c2c70-651a-4964-beda-2cee914ce508", + "type": "input_audio_buffer.append", + "audio": { + "audio_file": "test_realtime_voice_conversation.audio/caller.wav", + "pcm": true, + "offset_bytes": 129600, + "length_bytes": 960 + } + } + }, + { + "direction": "send", + "message": { + "event_id": "18f8cf47-f59c-4bb8-84f5-09e4604610bc", + "type": "input_audio_buffer.append", + "audio": { + "audio_file": "test_realtime_voice_conversation.audio/caller.wav", + "pcm": true, + "offset_bytes": 130560, + "length_bytes": 960 + } + } + }, + { + "direction": "send", + "message": { + "event_id": "9c1bdf4a-0591-4b24-b16e-395a74d94496", + "type": "input_audio_buffer.append", + "audio": { + "audio_file": "test_realtime_voice_conversation.audio/caller.wav", + "pcm": true, + "offset_bytes": 131520, + "length_bytes": 960 + } + } + }, + { + "direction": "send", + "message": { + "event_id": "948173dc-1c5c-4ee9-880a-a36a6164a7f6", + "type": "input_audio_buffer.append", + "audio": { + "audio_file": "test_realtime_voice_conversation.audio/caller.wav", + "pcm": true, + "offset_bytes": 132480, + "length_bytes": 960 + } + } + }, + { + "direction": "send", + "message": { + "event_id": "e7247f24-c9cf-4041-8558-4e7cc5f782af", + "type": "input_audio_buffer.append", + "audio": { + "audio_file": "test_realtime_voice_conversation.audio/caller.wav", + "pcm": true, + "offset_bytes": 133440, + "length_bytes": 960 + } + } + }, + { + "direction": "send", + "message": { + "event_id": "4878be01-3422-4883-94a7-e04ebf12c43d", + "type": "input_audio_buffer.append", + "audio": { + "audio_file": "test_realtime_voice_conversation.audio/caller.wav", + "pcm": true, + "offset_bytes": 134400, + "length_bytes": 960 + } + } + }, + { + "direction": "send", + "message": { + "event_id": "d0304734-5c19-4f21-92f3-608190f0a18e", + "type": "input_audio_buffer.append", + "audio": { + "audio_file": "test_realtime_voice_conversation.audio/caller.wav", + "pcm": true, + "offset_bytes": 135360, + "length_bytes": 960 + } + } + }, + { + "direction": "send", + "message": { + "event_id": "a01aef44-0e00-4c81-bf3b-cf5ba1a6ec79", + "type": "input_audio_buffer.append", + "audio": { + "audio_file": "test_realtime_voice_conversation.audio/caller.wav", + "pcm": true, + "offset_bytes": 136320, + "length_bytes": 960 + } + } + }, + { + "direction": "send", + "message": { + "event_id": "9ea16082-1a57-4c6d-823b-18408875259a", + "type": "input_audio_buffer.append", + "audio": { + "audio_file": "test_realtime_voice_conversation.audio/caller.wav", + "pcm": true, + "offset_bytes": 137280, + "length_bytes": 960 + } + } + }, + { + "direction": "send", + "message": { + "event_id": "0c6e0fbb-8884-4175-a27a-aaeb74c9e52b", + "type": "input_audio_buffer.append", + "audio": { + "audio_file": "test_realtime_voice_conversation.audio/caller.wav", + "pcm": true, + "offset_bytes": 138240, + "length_bytes": 960 + } + } + }, + { + "direction": "send", + "message": { + "event_id": "5f403c14-0035-4d60-bc12-a2ba5862506c", + "type": "input_audio_buffer.append", + "audio": { + "audio_file": "test_realtime_voice_conversation.audio/caller.wav", + "pcm": true, + "offset_bytes": 139200, + "length_bytes": 960 + } + } + }, + { + "direction": "send", + "message": { + "event_id": "f2dda14d-6c7a-4b01-9d5a-629af9ad4e13", + "type": "input_audio_buffer.append", + "audio": { + "audio_file": "test_realtime_voice_conversation.audio/caller.wav", + "pcm": true, + "offset_bytes": 140160, + "length_bytes": 960 + } + } + }, + { + "direction": "send", + "message": { + "event_id": "e7daff3f-5cbb-42bf-b536-6848df39d8eb", + "type": "input_audio_buffer.append", + "audio": { + "audio_file": "test_realtime_voice_conversation.audio/caller.wav", + "pcm": true, + "offset_bytes": 141120, + "length_bytes": 960 + } + } + }, + { + "direction": "send", + "message": { + "event_id": "aea15b99-0f4c-43d5-8b7a-cb78fc410c5d", + "type": "input_audio_buffer.append", + "audio": { + "audio_file": "test_realtime_voice_conversation.audio/caller.wav", + "pcm": true, + "offset_bytes": 142080, + "length_bytes": 960 + } + } + }, + { + "direction": "send", + "message": { + "event_id": "c2786b1c-cdd0-4c39-b5c2-52c87a41c6c1", + "type": "input_audio_buffer.append", + "audio": { + "audio_file": "test_realtime_voice_conversation.audio/caller.wav", + "pcm": true, + "offset_bytes": 143040, + "length_bytes": 960 + } + } + }, + { + "direction": "send", + "message": { + "event_id": "b363dbf7-c7ae-4a4c-b655-d357cbc98541", + "type": "input_audio_buffer.append", + "audio": { + "audio_file": "test_realtime_voice_conversation.audio/caller.wav", + "pcm": true, + "offset_bytes": 144000, + "length_bytes": 960 + } + } + }, + { + "direction": "send", + "message": { + "event_id": "30f5ea5c-c7e4-497c-afcc-134c0a594a1f", + "type": "input_audio_buffer.append", + "audio": { + "audio_file": "test_realtime_voice_conversation.audio/caller.wav", + "pcm": true, + "offset_bytes": 144960, + "length_bytes": 960 + } + } + }, + { + "direction": "send", + "message": { + "event_id": "a0e10d8c-d1f4-46e5-b1af-fbeb390db66a", + "type": "input_audio_buffer.append", + "audio": { + "audio_file": "test_realtime_voice_conversation.audio/caller.wav", + "pcm": true, + "offset_bytes": 145920, + "length_bytes": 960 + } + } + }, + { + "direction": "send", + "message": { + "event_id": "6aeb3adb-f3a4-425f-a69c-476c338dd449", + "type": "input_audio_buffer.append", + "audio": { + "audio_file": "test_realtime_voice_conversation.audio/caller.wav", + "pcm": true, + "offset_bytes": 146880, + "length_bytes": 960 + } + } + }, + { + "direction": "send", + "message": { + "event_id": "0a546860-b9e8-44bc-8447-3e61d5fd217a", + "type": "input_audio_buffer.append", + "audio": { + "audio_file": "test_realtime_voice_conversation.audio/caller.wav", + "pcm": true, + "offset_bytes": 147840, + "length_bytes": 960 + } + } + }, + { + "direction": "send", + "message": { + "event_id": "583070cf-f786-4569-8353-17e0cac701f7", + "type": "input_audio_buffer.append", + "audio": { + "audio_file": "test_realtime_voice_conversation.audio/caller.wav", + "pcm": true, + "offset_bytes": 148800, + "length_bytes": 960 + } + } + }, + { + "direction": "send", + "message": { + "event_id": "794f489f-d673-41c6-a610-a5d55a8cb81b", + "type": "input_audio_buffer.append", + "audio": { + "audio_file": "test_realtime_voice_conversation.audio/caller.wav", + "pcm": true, + "offset_bytes": 149760, + "length_bytes": 960 + } + } + }, + { + "direction": "send", + "message": { + "event_id": "2381dbdf-b836-487b-b1c4-aaace016f0ca", + "type": "input_audio_buffer.append", + "audio": { + "audio_file": "test_realtime_voice_conversation.audio/caller.wav", + "pcm": true, + "offset_bytes": 150720, + "length_bytes": 960 + } + } + }, + { + "direction": "send", + "message": { + "event_id": "8e608d2b-e8c9-4d64-ae03-7e62352f012a", + "type": "input_audio_buffer.append", + "audio": { + "audio_file": "test_realtime_voice_conversation.audio/caller.wav", + "pcm": true, + "offset_bytes": 151680, + "length_bytes": 960 + } + } + }, + { + "direction": "send", + "message": { + "event_id": "ce21055b-03af-4ba0-b606-351c83815d6c", + "type": "input_audio_buffer.append", + "audio": { + "audio_file": "test_realtime_voice_conversation.audio/caller.wav", + "pcm": true, + "offset_bytes": 152640, + "length_bytes": 960 + } + } + }, + { + "direction": "send", + "message": { + "event_id": "20fe83e2-55da-44ca-8445-587712652a1a", + "type": "input_audio_buffer.append", + "audio": { + "audio_file": "test_realtime_voice_conversation.audio/caller.wav", + "pcm": true, + "offset_bytes": 153600, + "length_bytes": 960 + } + } + }, + { + "direction": "send", + "message": { + "event_id": "a5654288-9500-4660-9c3e-2e5a885d3524", + "type": "input_audio_buffer.append", + "audio": { + "audio_file": "test_realtime_voice_conversation.audio/caller.wav", + "pcm": true, + "offset_bytes": 154560, + "length_bytes": 960 + } + } + }, + { + "direction": "send", + "message": { + "event_id": "6b9b17c4-1dc2-4d9c-bb5b-ef2fcd470cd1", + "type": "input_audio_buffer.append", + "audio": { + "audio_file": "test_realtime_voice_conversation.audio/caller.wav", + "pcm": true, + "offset_bytes": 155520, + "length_bytes": 960 + } + } + }, + { + "direction": "receive", + "message": { + "type": "input_audio_buffer.speech_stopped", + "event_id": "event_EVgfJ0RAxlAOvXlsmyw3Q", + "audio_end_ms": 3040, + "item_id": "item_EVgfFVrfdinRe1ZSR1mqw" + } + }, + { + "direction": "receive", + "message": { + "type": "input_audio_buffer.committed", + "event_id": "event_EVgfJJ35RmgC9mLmHdMWM", + "previous_item_id": "item_EVgfFMwZMFVUydi3vk3kH", + "item_id": "item_EVgfFVrfdinRe1ZSR1mqw" + } + }, + { + "direction": "send", + "message": { + "event_id": "23a878c1-ca2b-4487-90fe-057f36dd762e", + "type": "input_audio_buffer.append", + "audio": { + "audio_file": "test_realtime_voice_conversation.audio/caller.wav", + "pcm": true, + "offset_bytes": 156480, + "length_bytes": 960 + } + } + }, + { + "direction": "receive", + "message": { + "type": "conversation.item.added", + "event_id": "event_EVgfJfuq4eQVvcy8NzAOf", + "previous_item_id": "item_EVgfFMwZMFVUydi3vk3kH", + "item": { + "id": "item_EVgfFVrfdinRe1ZSR1mqw", + "type": "message", + "status": "completed", + "role": "user", + "content": [ + { + "type": "input_audio", + "transcript": null + } + ] + } + } + }, + { + "direction": "receive", + "message": { + "type": "conversation.item.done", + "event_id": "event_EVgfJlWefyODH8aYPL6Tr", + "previous_item_id": "item_EVgfFMwZMFVUydi3vk3kH", + "item": { + "id": "item_EVgfFVrfdinRe1ZSR1mqw", + "type": "message", + "status": "completed", + "role": "user", + "content": [ + { + "type": "input_audio", + "transcript": null + } + ] + } + } + }, + { + "direction": "receive", + "message": { + "type": "response.created", + "event_id": "event_EVgfJFQeCBsURKxtKhXT9", + "response": { + "object": "realtime.response", + "id": "resp_EVgfJxiXrXnWkWXUmzhkL", + "status": "in_progress", + "status_details": null, + "output": [], + "conversation_id": "conv_EVgfECJ3whhWkN1aLHqyp", + "output_modalities": [ + "audio" + ], + "max_output_tokens": "inf", + "audio": { + "output": { + "format": { + "type": "audio/pcm", + "rate": 24000 + }, + "voice": "alloy" + } + }, + "usage": null, + "metadata": null + } + } + }, + { + "direction": "send", + "message": { + "event_id": "c55d667e-ead1-4703-949e-255143548f46", + "type": "input_audio_buffer.append", + "audio": { + "audio_file": "test_realtime_voice_conversation.audio/caller.wav", + "pcm": true, + "offset_bytes": 157440, + "length_bytes": 960 + } + } + }, + { + "direction": "send", + "message": { + "event_id": "bcfd8860-3617-48ef-ac20-463ee5f59556", + "type": "input_audio_buffer.append", + "audio": { + "audio_file": "test_realtime_voice_conversation.audio/caller.wav", + "pcm": true, + "offset_bytes": 158400, + "length_bytes": 960 + } + } + }, + { + "direction": "send", + "message": { + "event_id": "0224cc3e-eb25-4ffd-a45b-93094228d6bc", + "type": "input_audio_buffer.append", + "audio": { + "audio_file": "test_realtime_voice_conversation.audio/caller.wav", + "pcm": true, + "offset_bytes": 159360, + "length_bytes": 960 + } + } + }, + { + "direction": "send", + "message": { + "event_id": "a6a339ca-6fea-4d61-a4ee-cd0e6168d98d", + "type": "input_audio_buffer.append", + "audio": { + "audio_file": "test_realtime_voice_conversation.audio/caller.wav", + "pcm": true, + "offset_bytes": 160320, + "length_bytes": 960 + } + } + }, + { + "direction": "send", + "message": { + "event_id": "ba0daa90-a533-49b2-a608-5dc72d47b251", + "type": "input_audio_buffer.append", + "audio": { + "audio_file": "test_realtime_voice_conversation.audio/caller.wav", + "pcm": true, + "offset_bytes": 161280, + "length_bytes": 960 + } + } + }, + { + "direction": "send", + "message": { + "event_id": "031729e8-3f26-4ffb-abdb-eceabea65127", + "type": "input_audio_buffer.append", + "audio": { + "audio_file": "test_realtime_voice_conversation.audio/caller.wav", + "pcm": true, + "offset_bytes": 162240, + "length_bytes": 960 + } + } + }, + { + "direction": "send", + "message": { + "event_id": "b783491d-6f9c-4910-870f-f8a66c0fddb9", + "type": "input_audio_buffer.append", + "audio": { + "audio_file": "test_realtime_voice_conversation.audio/caller.wav", + "pcm": true, + "offset_bytes": 163200, + "length_bytes": 960 + } + } + }, + { + "direction": "send", + "message": { + "event_id": "c26555bc-ccd1-4414-824a-5112e8faa56e", + "type": "input_audio_buffer.append", + "audio": { + "audio_file": "test_realtime_voice_conversation.audio/caller.wav", + "pcm": true, + "offset_bytes": 164160, + "length_bytes": 960 + } + } + }, + { + "direction": "send", + "message": { + "event_id": "29a0b42e-f600-4440-a7c9-0582c68b33ce", + "type": "input_audio_buffer.append", + "audio": { + "audio_file": "test_realtime_voice_conversation.audio/caller.wav", + "pcm": true, + "offset_bytes": 165120, + "length_bytes": 960 + } + } + }, + { + "direction": "send", + "message": { + "event_id": "96bd5f3c-f0dd-4f9a-89f5-4cb6dde21558", + "type": "input_audio_buffer.append", + "audio": { + "audio_file": "test_realtime_voice_conversation.audio/caller.wav", + "pcm": true, + "offset_bytes": 166080, + "length_bytes": 720 + } + } + }, + { + "direction": "receive", + "message": { + "type": "conversation.item.input_audio_transcription.delta", + "event_id": "event_EVgfJL85UN6Y4jawZbmF5", + "item_id": "item_EVgfFVrfdinRe1ZSR1mqw", + "content_index": 0, + "delta": "Where", + "obfuscation": "8BA0mrCENCA" + } + }, + { + "direction": "receive", + "message": { + "type": "conversation.item.input_audio_transcription.delta", + "event_id": "event_EVgfJxMyhv7pyrOiLmPRM", + "item_id": "item_EVgfFVrfdinRe1ZSR1mqw", + "content_index": 0, + "delta": " is", + "obfuscation": "Nf3Fl2dZ4xMKw" + } + }, + { + "direction": "receive", + "message": { + "type": "conversation.item.input_audio_transcription.delta", + "event_id": "event_EVgfJ8F7LTB7KPGmfJ4z5", + "item_id": "item_EVgfFVrfdinRe1ZSR1mqw", + "content_index": 0, + "delta": " my", + "obfuscation": "aMXI2Ux65ystd" + } + }, + { + "direction": "receive", + "message": { + "type": "conversation.item.input_audio_transcription.delta", + "event_id": "event_EVgfJ5YvUyTW03mjBd6Mo", + "item_id": "item_EVgfFVrfdinRe1ZSR1mqw", + "content_index": 0, + "delta": " order", + "obfuscation": "UV1BWl0qVl" + } + }, + { + "direction": "receive", + "message": { + "type": "conversation.item.input_audio_transcription.delta", + "event_id": "event_EVgfJf77pKBFMCA8V8wHr", + "item_id": "item_EVgfFVrfdinRe1ZSR1mqw", + "content_index": 0, + "delta": " number", + "obfuscation": "nnPk5Kpxf" + } + }, + { + "direction": "receive", + "message": { + "type": "conversation.item.input_audio_transcription.delta", + "event_id": "event_EVgfJN9l9XiownhQdSMVR", + "item_id": "item_EVgfFVrfdinRe1ZSR1mqw", + "content_index": 0, + "delta": " ", + "obfuscation": "TNQcYMhDOYHT0Vp" + } + }, + { + "direction": "receive", + "message": { + "type": "conversation.item.input_audio_transcription.delta", + "event_id": "event_EVgfJ3CSzYcFLzJ2K3F8n", + "item_id": "item_EVgfFVrfdinRe1ZSR1mqw", + "content_index": 0, + "delta": "104", + "obfuscation": "u0uHvdx2YxMVr" + } + }, + { + "direction": "receive", + "message": { + "type": "conversation.item.input_audio_transcription.delta", + "event_id": "event_EVgfJioIgc7GATT0AfrDM", + "item_id": "item_EVgfFVrfdinRe1ZSR1mqw", + "content_index": 0, + "delta": "2", + "obfuscation": "pk8Nf8rRFb7qVC8" + } + }, + { + "direction": "receive", + "message": { + "type": "conversation.item.input_audio_transcription.delta", + "event_id": "event_EVgfJn493wYXVatsBzciB", + "item_id": "item_EVgfFVrfdinRe1ZSR1mqw", + "content_index": 0, + "delta": "?", + "obfuscation": "wTmotNh8GwuOMib" + } + }, + { + "direction": "receive", + "message": { + "type": "response.output_item.added", + "event_id": "event_EVgfJah6n7758dfF0r0hk", + "response_id": "resp_EVgfJxiXrXnWkWXUmzhkL", + "output_index": 0, + "item": { + "id": "item_EVgfJp1sqRU5rN3ftYQF9", + "type": "function_call", + "status": "in_progress", + "name": "lookup_order", + "call_id": "call_RDQqTdChlVfCtYwS", + "arguments": "" + } + } + }, + { + "direction": "receive", + "message": { + "type": "conversation.item.added", + "event_id": "event_EVgfJYVzZcEAakXMsWPBI", + "previous_item_id": "item_EVgfFVrfdinRe1ZSR1mqw", + "item": { + "id": "item_EVgfJp1sqRU5rN3ftYQF9", + "type": "function_call", + "status": "in_progress", + "name": "lookup_order", + "call_id": "call_RDQqTdChlVfCtYwS", + "arguments": "" + } + } + }, + { + "direction": "receive", + "message": { + "type": "response.function_call_arguments.delta", + "event_id": "event_EVgfJqYg0jm5GR604Twyz", + "response_id": "resp_EVgfJxiXrXnWkWXUmzhkL", + "item_id": "item_EVgfJp1sqRU5rN3ftYQF9", + "output_index": 0, + "call_id": "call_RDQqTdChlVfCtYwS", + "delta": "{", + "obfuscation": "7UAIFEo3WZWubWT" + } + }, + { + "direction": "receive", + "message": { + "type": "response.function_call_arguments.delta", + "event_id": "event_EVgfJ6ngaN9NV07lDyyy2", + "response_id": "resp_EVgfJxiXrXnWkWXUmzhkL", + "item_id": "item_EVgfJp1sqRU5rN3ftYQF9", + "output_index": 0, + "call_id": "call_RDQqTdChlVfCtYwS", + "delta": " \n", + "obfuscation": "za4VALmTWQbpB" + } + }, + { + "direction": "receive", + "message": { + "type": "response.function_call_arguments.delta", + "event_id": "event_EVgfJGhVTb1gCY53Q0udi", + "response_id": "resp_EVgfJxiXrXnWkWXUmzhkL", + "item_id": "item_EVgfJp1sqRU5rN3ftYQF9", + "output_index": 0, + "call_id": "call_RDQqTdChlVfCtYwS", + "delta": " ", + "obfuscation": "sqxNdTVVuld3jqK" + } + }, + { + "direction": "receive", + "message": { + "type": "response.function_call_arguments.delta", + "event_id": "event_EVgfJ9f5Wum9TQOU2itiU", + "response_id": "resp_EVgfJxiXrXnWkWXUmzhkL", + "item_id": "item_EVgfJp1sqRU5rN3ftYQF9", + "output_index": 0, + "call_id": "call_RDQqTdChlVfCtYwS", + "delta": " \"", + "obfuscation": "SoYFWDqxk2IAYB" + } + }, + { + "direction": "receive", + "message": { + "type": "response.function_call_arguments.delta", + "event_id": "event_EVgfJXdPlA4Unk4hVhcNu", + "response_id": "resp_EVgfJxiXrXnWkWXUmzhkL", + "item_id": "item_EVgfJp1sqRU5rN3ftYQF9", + "output_index": 0, + "call_id": "call_RDQqTdChlVfCtYwS", + "delta": "order", + "obfuscation": "JSZMenVTIJg" + } + }, + { + "direction": "receive", + "message": { + "type": "response.function_call_arguments.delta", + "event_id": "event_EVgfJUNiVDVToJuOAvCK8", + "response_id": "resp_EVgfJxiXrXnWkWXUmzhkL", + "item_id": "item_EVgfJp1sqRU5rN3ftYQF9", + "output_index": 0, + "call_id": "call_RDQqTdChlVfCtYwS", + "delta": "_id", + "obfuscation": "LIKaiSrjaJLgK" + } + }, + { + "direction": "receive", + "message": { + "type": "response.function_call_arguments.delta", + "event_id": "event_EVgfJyN9g0JyPdwzTh7rC", + "response_id": "resp_EVgfJxiXrXnWkWXUmzhkL", + "item_id": "item_EVgfJp1sqRU5rN3ftYQF9", + "output_index": 0, + "call_id": "call_RDQqTdChlVfCtYwS", + "delta": "\":", + "obfuscation": "9D49h1GEqB4EPz" + } + }, + { + "direction": "receive", + "message": { + "type": "response.function_call_arguments.delta", + "event_id": "event_EVgfJQfa4UPUO99qo6k9C", + "response_id": "resp_EVgfJxiXrXnWkWXUmzhkL", + "item_id": "item_EVgfJp1sqRU5rN3ftYQF9", + "output_index": 0, + "call_id": "call_RDQqTdChlVfCtYwS", + "delta": " \"", + "obfuscation": "jjqT1ZXP6uS7sA" + } + }, + { + "direction": "receive", + "message": { + "type": "response.function_call_arguments.delta", + "event_id": "event_EVgfJ18WZPqXWgIrc7iqn", + "response_id": "resp_EVgfJxiXrXnWkWXUmzhkL", + "item_id": "item_EVgfJp1sqRU5rN3ftYQF9", + "output_index": 0, + "call_id": "call_RDQqTdChlVfCtYwS", + "delta": "104", + "obfuscation": "UQXCeKwH6zcEm" + } + }, + { + "direction": "receive", + "message": { + "type": "response.function_call_arguments.delta", + "event_id": "event_EVgfJ4KJqx2q56QSuEkOY", + "response_id": "resp_EVgfJxiXrXnWkWXUmzhkL", + "item_id": "item_EVgfJp1sqRU5rN3ftYQF9", + "output_index": 0, + "call_id": "call_RDQqTdChlVfCtYwS", + "delta": "2", + "obfuscation": "tbjYHfKL8aD9Ypi" + } + }, + { + "direction": "receive", + "message": { + "type": "response.function_call_arguments.delta", + "event_id": "event_EVgfJ60f5tVJGzFfbDnIK", + "response_id": "resp_EVgfJxiXrXnWkWXUmzhkL", + "item_id": "item_EVgfJp1sqRU5rN3ftYQF9", + "output_index": 0, + "call_id": "call_RDQqTdChlVfCtYwS", + "delta": "\"", + "obfuscation": "2NEpQkIWJnjPql0" + } + }, + { + "direction": "receive", + "message": { + "type": "response.function_call_arguments.delta", + "event_id": "event_EVgfJffrprSpCueCzBbKz", + "response_id": "resp_EVgfJxiXrXnWkWXUmzhkL", + "item_id": "item_EVgfJp1sqRU5rN3ftYQF9", + "output_index": 0, + "call_id": "call_RDQqTdChlVfCtYwS", + "delta": " \n", + "obfuscation": "Mqe23ITSjag48R" + } + }, + { + "direction": "receive", + "message": { + "type": "response.function_call_arguments.delta", + "event_id": "event_EVgfJE8BYvC7qhKVLFiZG", + "response_id": "resp_EVgfJxiXrXnWkWXUmzhkL", + "item_id": "item_EVgfJp1sqRU5rN3ftYQF9", + "output_index": 0, + "call_id": "call_RDQqTdChlVfCtYwS", + "delta": "}", + "obfuscation": "OcCJTNNZebjZlpi" + } + }, + { + "direction": "receive", + "message": { + "type": "response.function_call_arguments.delta", + "event_id": "event_EVgfJQkXKjEjgP9Fb4Ugp", + "response_id": "resp_EVgfJxiXrXnWkWXUmzhkL", + "item_id": "item_EVgfJp1sqRU5rN3ftYQF9", + "output_index": 0, + "call_id": "call_RDQqTdChlVfCtYwS", + "delta": " \n", + "obfuscation": "soDkUni7FDyhc" + } + }, + { + "direction": "receive", + "message": { + "type": "response.function_call_arguments.done", + "event_id": "event_EVgfJZs1meBUul8V3wFuj", + "response_id": "resp_EVgfJxiXrXnWkWXUmzhkL", + "item_id": "item_EVgfJp1sqRU5rN3ftYQF9", + "output_index": 0, + "call_id": "call_RDQqTdChlVfCtYwS", + "name": "lookup_order", + "arguments": "{ \n \"order_id\": \"1042\" \n} \n" + } + }, + { + "direction": "receive", + "message": { + "type": "conversation.item.done", + "event_id": "event_EVgfJTwsIayNzKVsPIJ0w", + "previous_item_id": "item_EVgfFVrfdinRe1ZSR1mqw", + "item": { + "id": "item_EVgfJp1sqRU5rN3ftYQF9", + "type": "function_call", + "status": "completed", + "name": "lookup_order", + "call_id": "call_RDQqTdChlVfCtYwS", + "arguments": "{ \n \"order_id\": \"1042\" \n} \n" + } + } + }, + { + "direction": "receive", + "message": { + "type": "response.output_item.done", + "event_id": "event_EVgfJpcFux6RBUJA1wZGI", + "response_id": "resp_EVgfJxiXrXnWkWXUmzhkL", + "output_index": 0, + "item": { + "id": "item_EVgfJp1sqRU5rN3ftYQF9", + "type": "function_call", + "status": "completed", + "name": "lookup_order", + "call_id": "call_RDQqTdChlVfCtYwS", + "arguments": "{ \n \"order_id\": \"1042\" \n} \n" + } + } + }, + { + "direction": "send", + "message": { + "event_id": "47353ced-f561-431f-86df-507715d74b36", + "type": "conversation.item.create", + "item": { + "id": "88c379374e4843268e84a9a687ed7246", + "type": "function_call_output", + "call_id": "call_RDQqTdChlVfCtYwS", + "output": "{\"order_id\": \"1042\", \"delivery\": \"Friday\"}" + } + } + }, + { + "direction": "send", + "message": { + "event_id": "e5bc84a4-bb91-46aa-88a3-adae5d81489f", + "type": "response.create", + "response": { + "output_modalities": [ + "audio" + ] + } + } + }, + { + "direction": "receive", + "message": { + "type": "response.done", + "event_id": "event_EVgfJ5yJoB3vVSlefKGd0", + "response": { + "object": "realtime.response", + "id": "resp_EVgfJxiXrXnWkWXUmzhkL", + "status": "completed", + "status_details": null, + "output": [ + { + "id": "item_EVgfJp1sqRU5rN3ftYQF9", + "type": "function_call", + "status": "completed", + "name": "lookup_order", + "call_id": "call_RDQqTdChlVfCtYwS", + "arguments": "{ \n \"order_id\": \"1042\" \n} \n" + } + ], + "conversation_id": "conv_EVgfECJ3whhWkN1aLHqyp", + "output_modalities": [ + "audio" + ], + "max_output_tokens": "inf", + "audio": { + "output": { + "format": { + "type": "audio/pcm", + "rate": 24000 + }, + "voice": "alloy" + } + }, + "usage": { + "total_tokens": 198, + "input_tokens": 174, + "output_tokens": 24, + "input_token_details": { + "text_tokens": 103, + "audio_tokens": 71, + "image_tokens": 0, + "cached_tokens": 0, + "cached_tokens_details": { + "text_tokens": 0, + "audio_tokens": 0, + "image_tokens": 0 + } + }, + "output_token_details": { + "text_tokens": 24, + "audio_tokens": 0 + } + }, + "metadata": null + } + } + }, + { + "direction": "receive", + "message": { + "type": "rate_limits.updated", + "event_id": "event_EVgfJSNdBzq4oCvrrbz6N", + "rate_limits": [ + { + "name": "tokens", + "limit": 15000000, + "remaining": 14999232, + "reset_seconds": 0.003 + } + ] + } + }, + { + "direction": "receive", + "message": { + "type": "conversation.item.added", + "event_id": "event_EVgfJ49Ac7LItPVTslxoH", + "previous_item_id": "item_EVgfJp1sqRU5rN3ftYQF9", + "item": { + "id": "88c379374e4843268e84a9a687ed7246", + "type": "function_call_output", + "call_id": "call_RDQqTdChlVfCtYwS", + "output": "{\"order_id\": \"1042\", \"delivery\": \"Friday\"}" + } + } + }, + { + "direction": "receive", + "message": { + "type": "conversation.item.done", + "event_id": "event_EVgfJFdCQXf8W3YeA1xPJ", + "previous_item_id": "item_EVgfJp1sqRU5rN3ftYQF9", + "item": { + "id": "88c379374e4843268e84a9a687ed7246", + "type": "function_call_output", + "call_id": "call_RDQqTdChlVfCtYwS", + "output": "{\"order_id\": \"1042\", \"delivery\": \"Friday\"}" + } + } + }, + { + "direction": "receive", + "message": { + "type": "response.created", + "event_id": "event_EVgfJjDq39AEhcwjs4Trk", + "response": { + "object": "realtime.response", + "id": "resp_EVgfJLWseCWlHxaBAAei7", + "status": "in_progress", + "status_details": null, + "output": [], + "conversation_id": "conv_EVgfECJ3whhWkN1aLHqyp", + "output_modalities": [ + "audio" + ], + "max_output_tokens": "inf", + "audio": { + "output": { + "format": { + "type": "audio/pcm", + "rate": 24000 + }, + "voice": "alloy" + } + }, + "usage": null, + "metadata": null + } + } + }, + { + "direction": "receive", + "message": { + "type": "conversation.item.input_audio_transcription.completed", + "event_id": "event_EVgfJ2HRVr5zp0DQeuvUL", + "item_id": "item_EVgfFVrfdinRe1ZSR1mqw", + "content_index": 0, + "transcript": "Where is my order number 1042?", + "usage": { + "type": "tokens", + "total_tokens": 42, + "input_tokens": 31, + "input_token_details": { + "text_tokens": 1, + "audio_tokens": 30 + }, + "output_tokens": 11 + } + } + }, + { + "direction": "receive", + "message": { + "type": "response.output_item.added", + "event_id": "event_EVgfKC2mI7grtiAkK6sgE", + "response_id": "resp_EVgfJLWseCWlHxaBAAei7", + "output_index": 0, + "item": { + "id": "item_EVgfJgz3hZNReB0kPe0AW", + "type": "message", + "status": "in_progress", + "role": "assistant", + "content": [] + } + } + }, + { + "direction": "receive", + "message": { + "type": "conversation.item.added", + "event_id": "event_EVgfKx4DM50ZLUUODGTjE", + "previous_item_id": "88c379374e4843268e84a9a687ed7246", + "item": { + "id": "item_EVgfJgz3hZNReB0kPe0AW", + "type": "message", + "status": "in_progress", + "role": "assistant", + "content": [] + } + } + }, + { + "direction": "receive", + "message": { + "type": "response.content_part.added", + "event_id": "event_EVgfKWpIJJSj7iVT01PL3", + "response_id": "resp_EVgfJLWseCWlHxaBAAei7", + "item_id": "item_EVgfJgz3hZNReB0kPe0AW", + "output_index": 0, + "content_index": 0, + "part": { + "type": "audio", + "transcript": "" + } + } + }, + { + "direction": "receive", + "message": { + "type": "response.output_audio_transcript.delta", + "event_id": "event_EVgfKA8tlWH2Y8Q15QRWI", + "response_id": "resp_EVgfJLWseCWlHxaBAAei7", + "item_id": "item_EVgfJgz3hZNReB0kPe0AW", + "output_index": 0, + "content_index": 0, + "delta": "Your", + "obfuscation": "hQHwXeEcUnTz" + } + }, + { + "direction": "receive", + "message": { + "type": "response.output_audio_transcript.delta", + "event_id": "event_EVgfKCRWDJBm8gxZkln2k", + "response_id": "resp_EVgfJLWseCWlHxaBAAei7", + "item_id": "item_EVgfJgz3hZNReB0kPe0AW", + "output_index": 0, + "content_index": 0, + "delta": " order", + "obfuscation": "lSf7HS3clv" + } + }, + { + "direction": "receive", + "message": { + "type": "response.output_audio_transcript.delta", + "event_id": "event_EVgfKk3TdCTFlWpXHkAfP", + "response_id": "resp_EVgfJLWseCWlHxaBAAei7", + "item_id": "item_EVgfJgz3hZNReB0kPe0AW", + "output_index": 0, + "content_index": 0, + "delta": " arrives", + "obfuscation": "Md2BPoRb" + } + }, + { + "direction": "receive", + "message": { + "type": "response.output_audio.delta", + "event_id": "event_EVgfKN7bA2ItyMieRsMxj", + "response_id": "resp_EVgfJLWseCWlHxaBAAei7", + "item_id": "item_EVgfJgz3hZNReB0kPe0AW", + "output_index": 0, + "content_index": 0, + "delta": { + "audio_file": "test_realtime_voice_conversation.audio/agent-001.wav", + "pcm": true, + "offset_bytes": 0, + "length_bytes": 4800 + } + } + }, + { + "direction": "receive", + "message": { + "type": "response.output_audio.delta", + "event_id": "event_EVgfKDf20sAHbW4zB8adH", + "response_id": "resp_EVgfJLWseCWlHxaBAAei7", + "item_id": "item_EVgfJgz3hZNReB0kPe0AW", + "output_index": 0, + "content_index": 0, + "delta": { + "audio_file": "test_realtime_voice_conversation.audio/agent-001.wav", + "pcm": true, + "offset_bytes": 4800, + "length_bytes": 7200 + } + } + }, + { + "direction": "receive", + "message": { + "type": "response.output_audio_transcript.delta", + "event_id": "event_EVgfKX0tcR3rC2CWLso8g", + "response_id": "resp_EVgfJLWseCWlHxaBAAei7", + "item_id": "item_EVgfJgz3hZNReB0kPe0AW", + "output_index": 0, + "content_index": 0, + "delta": " Friday", + "obfuscation": "b001URnYL" + } + }, + { + "direction": "receive", + "message": { + "type": "response.output_audio.delta", + "event_id": "event_EVgfK1PA0bzoOVOPRS62f", + "response_id": "resp_EVgfJLWseCWlHxaBAAei7", + "item_id": "item_EVgfJgz3hZNReB0kPe0AW", + "output_index": 0, + "content_index": 0, + "delta": { + "audio_file": "test_realtime_voice_conversation.audio/agent-001.wav", + "pcm": true, + "offset_bytes": 12000, + "length_bytes": 12000 + } + } + }, + { + "direction": "receive", + "message": { + "type": "response.output_audio_transcript.delta", + "event_id": "event_EVgfKueVpBTFfuDMNLYiG", + "response_id": "resp_EVgfJLWseCWlHxaBAAei7", + "item_id": "item_EVgfJgz3hZNReB0kPe0AW", + "output_index": 0, + "content_index": 0, + "delta": ".", + "obfuscation": "p37bWaBJobEfqoH" + } + }, + { + "direction": "receive", + "message": { + "type": "response.output_audio.delta", + "event_id": "event_EVgfKoKMvSKJfyTn3WnaX", + "response_id": "resp_EVgfJLWseCWlHxaBAAei7", + "item_id": "item_EVgfJgz3hZNReB0kPe0AW", + "output_index": 0, + "content_index": 0, + "delta": { + "audio_file": "test_realtime_voice_conversation.audio/agent-001.wav", + "pcm": true, + "offset_bytes": 24000, + "length_bytes": 12000 + } + } + }, + { + "direction": "receive", + "message": { + "type": "response.output_audio.delta", + "event_id": "event_EVgfK10mVVvrRNGMvNbej", + "response_id": "resp_EVgfJLWseCWlHxaBAAei7", + "item_id": "item_EVgfJgz3hZNReB0kPe0AW", + "output_index": 0, + "content_index": 0, + "delta": { + "audio_file": "test_realtime_voice_conversation.audio/agent-001.wav", + "pcm": true, + "offset_bytes": 36000, + "length_bytes": 12000 + } + } + }, + { + "direction": "receive", + "message": { + "type": "response.output_audio.delta", + "event_id": "event_EVgfKTFo2KawLetWH3ms6", + "response_id": "resp_EVgfJLWseCWlHxaBAAei7", + "item_id": "item_EVgfJgz3hZNReB0kPe0AW", + "output_index": 0, + "content_index": 0, + "delta": { + "audio_file": "test_realtime_voice_conversation.audio/agent-001.wav", + "pcm": true, + "offset_bytes": 48000, + "length_bytes": 12000 + } + } + }, + { + "direction": "receive", + "message": { + "type": "response.output_audio.delta", + "event_id": "event_EVgfKNMOW3bG6hxS9Dctt", + "response_id": "resp_EVgfJLWseCWlHxaBAAei7", + "item_id": "item_EVgfJgz3hZNReB0kPe0AW", + "output_index": 0, + "content_index": 0, + "delta": { + "audio_file": "test_realtime_voice_conversation.audio/agent-001.wav", + "pcm": true, + "offset_bytes": 60000, + "length_bytes": 12000 + } + } + }, + { + "direction": "receive", + "message": { + "type": "response.output_audio.delta", + "event_id": "event_EVgfKK3zOyrcTFYYNz2BJ", + "response_id": "resp_EVgfJLWseCWlHxaBAAei7", + "item_id": "item_EVgfJgz3hZNReB0kPe0AW", + "output_index": 0, + "content_index": 0, + "delta": { + "audio_file": "test_realtime_voice_conversation.audio/agent-001.wav", + "pcm": true, + "offset_bytes": 72000, + "length_bytes": 33600 + } + } + }, + { + "direction": "receive", + "message": { + "type": "response.output_audio.done", + "event_id": "event_EVgfK67Q5JKqF7kkhgb3b", + "response_id": "resp_EVgfJLWseCWlHxaBAAei7", + "item_id": "item_EVgfJgz3hZNReB0kPe0AW", + "output_index": 0, + "content_index": 0 + } + }, + { + "direction": "receive", + "message": { + "type": "response.output_audio_transcript.done", + "event_id": "event_EVgfKUg38NaDv2LWNV6ew", + "response_id": "resp_EVgfJLWseCWlHxaBAAei7", + "item_id": "item_EVgfJgz3hZNReB0kPe0AW", + "output_index": 0, + "content_index": 0, + "transcript": "Your order arrives Friday." + } + }, + { + "direction": "receive", + "message": { + "type": "response.content_part.done", + "event_id": "event_EVgfKQuk5s1Le7nZefyCl", + "response_id": "resp_EVgfJLWseCWlHxaBAAei7", + "item_id": "item_EVgfJgz3hZNReB0kPe0AW", + "output_index": 0, + "content_index": 0, + "part": { + "type": "audio", + "transcript": "Your order arrives Friday." + } + } + }, + { + "direction": "receive", + "message": { + "type": "conversation.item.done", + "event_id": "event_EVgfKhvjXr5zzDG4mPix7", + "previous_item_id": "88c379374e4843268e84a9a687ed7246", + "item": { + "id": "item_EVgfJgz3hZNReB0kPe0AW", + "type": "message", + "status": "completed", + "role": "assistant", + "content": [ + { + "type": "output_audio", + "transcript": "Your order arrives Friday." + } + ] + } + } + }, + { + "direction": "receive", + "message": { + "type": "response.output_item.done", + "event_id": "event_EVgfKuh3rIIUJsno1Co0o", + "response_id": "resp_EVgfJLWseCWlHxaBAAei7", + "output_index": 0, + "item": { + "id": "item_EVgfJgz3hZNReB0kPe0AW", + "type": "message", + "status": "completed", + "role": "assistant", + "content": [ + { + "type": "output_audio", + "transcript": "Your order arrives Friday." + } + ] + } + } + }, + { + "direction": "receive", + "message": { + "type": "response.done", + "event_id": "event_EVgfK4wcO7KeYIxU5DjCm", + "response": { + "object": "realtime.response", + "id": "resp_EVgfJLWseCWlHxaBAAei7", + "status": "completed", + "status_details": null, + "output": [ + { + "id": "item_EVgfJgz3hZNReB0kPe0AW", + "type": "message", + "status": "completed", + "role": "assistant", + "content": [ + { + "type": "output_audio", + "transcript": "Your order arrives Friday." + } + ] + } + ], + "conversation_id": "conv_EVgfECJ3whhWkN1aLHqyp", + "output_modalities": [ + "audio" + ], + "max_output_tokens": "inf", + "audio": { + "output": { + "format": { + "type": "audio/pcm", + "rate": 24000 + }, + "voice": "alloy" + } + }, + "usage": { + "total_tokens": 220, + "input_tokens": 163, + "output_tokens": 57, + "input_token_details": { + "text_tokens": 133, + "audio_tokens": 30, + "image_tokens": 0, + "cached_tokens": 64, + "cached_tokens_details": { + "text_tokens": 64, + "audio_tokens": 0, + "image_tokens": 0 + } + }, + "output_token_details": { + "text_tokens": 13, + "audio_tokens": 44 + } + }, + "metadata": null + } + } + }, + { + "direction": "receive", + "message": { + "type": "rate_limits.updated", + "event_id": "event_EVgfKxCXzL588xkpiYfJz", + "rate_limits": [ + { + "name": "tokens", + "limit": 15000000, + "remaining": 14998442, + "reset_seconds": 0.006 + } + ] + } + } + ] +} diff --git a/py/src/braintrust/integrations/pipecat/conftest.py b/py/src/braintrust/integrations/pipecat/conftest.py new file mode 100644 index 000000000..d564d65e2 --- /dev/null +++ b/py/src/braintrust/integrations/pipecat/conftest.py @@ -0,0 +1,3 @@ +"""Pipecat tests use in-memory trace and attachment export; provider calls use cassettes.""" + +from braintrust._audio.conftest import local_attachment_uploads as local_attachment_uploads diff --git a/py/src/braintrust/integrations/pipecat/llm_metrics.py b/py/src/braintrust/integrations/pipecat/llm_metrics.py new file mode 100644 index 000000000..9e75d1108 --- /dev/null +++ b/py/src/braintrust/integrations/pipecat/llm_metrics.py @@ -0,0 +1,73 @@ +"""Shared conversion of native Pipecat usage into Braintrust LLM metrics.""" + +from typing import Any + + +def _metadata_from_processor(processor: Any) -> dict[str, Any]: + metadata: dict[str, Any] = {} + settings = getattr(processor, "_settings", None) + model = getattr(settings, "model", None) or getattr(processor, "model", None) + if isinstance(model, str): + metadata["model"] = model + provider = _provider_from_processor(processor) + if provider: + metadata["provider"] = provider + return metadata + + +def _metadata_from_metric(metric: Any) -> dict[str, Any]: + metadata: dict[str, Any] = {} + model = getattr(metric, "model", None) + if isinstance(model, str): + metadata["model"] = model + processor = getattr(metric, "processor", None) + if isinstance(processor, str): + provider = _provider_from_name(processor) + if provider: + metadata["provider"] = provider + return metadata + + +def _provider_from_processor(processor: Any) -> str | None: + module = getattr(type(processor), "__module__", "") + return _provider_from_name(module) + + +def _provider_from_name(name: str) -> str | None: + lowered = name.lower() + providers = { + "openai": "openai", + "anthropic": "anthropic", + "google": "google", + "gemini": "google", + "mistral": "mistral", + "cohere": "cohere", + "bedrock": "bedrock", + "aws": "bedrock", + "openrouter": "openrouter", + } + for needle, provider in providers.items(): + if needle in lowered: + return provider + return None + + +def _llm_usage_metrics(usage: Any) -> dict[str, Any]: + metrics: dict[str, Any] = {} + prompt_tokens = getattr(usage, "prompt_tokens", None) + completion_tokens = getattr(usage, "completion_tokens", None) + total_tokens = getattr(usage, "total_tokens", None) + cache_read = getattr(usage, "cache_read_input_tokens", None) + cache_creation = getattr(usage, "cache_creation_input_tokens", None) + reasoning_tokens = getattr(usage, "reasoning_tokens", None) + for key, value in ( + ("prompt_tokens", prompt_tokens), + ("completion_tokens", completion_tokens), + ("tokens", total_tokens), + ("cache_read_input_tokens", cache_read), + ("cache_creation_input_tokens", cache_creation), + ("reasoning_tokens", reasoning_tokens), + ): + if isinstance(value, (int, float)) and not isinstance(value, bool): + metrics[key] = value + return metrics diff --git a/py/src/braintrust/integrations/pipecat/patchers.py b/py/src/braintrust/integrations/pipecat/patchers.py index 929cae964..85e08f1d0 100644 --- a/py/src/braintrust/integrations/pipecat/patchers.py +++ b/py/src/braintrust/integrations/pipecat/patchers.py @@ -1,5 +1,6 @@ """Pipecat integration patchers.""" +import logging from typing import Any from braintrust.integrations.base import FunctionWrapperPatcher @@ -12,6 +13,7 @@ "capture_user_audio_attachments": None, "capture_agent_audio_attachments": None, "trace_turns": True, + "audio_format": "ogg", } @@ -35,7 +37,15 @@ def traced_pipeline_worker_init(wrapped: Any, _instance: Any, args: tuple[Any, . """Inject the Braintrust Pipecat observer into PipelineWorker construction.""" kwargs = dict(kwargs) kwargs["observers"] = _with_braintrust_observer(kwargs.get("observers")) - return wrapped(*args, **kwargs) + result = wrapped(*args, **kwargs) + pipeline = kwargs.get("pipeline", args[0] if args else None) + for observer in kwargs["observers"]: + if isinstance(observer, BraintrustPipecatObserver): + try: + observer._bind_pipeline(pipeline) + except Exception: # noqa: BLE001 - optional discovery cannot break worker construction + logging.getLogger(__name__).warning("Pipecat voice discovery unavailable; using frame tracing") + return result class PipelineWorkerInitPatcher(FunctionWrapperPatcher): diff --git a/py/src/braintrust/integrations/pipecat/test_audio_cassettes.py b/py/src/braintrust/integrations/pipecat/test_audio_cassettes.py new file mode 100644 index 000000000..9bf6a411b --- /dev/null +++ b/py/src/braintrust/integrations/pipecat/test_audio_cassettes.py @@ -0,0 +1,89 @@ +"""Exercise persistence with actual recorded traffic, without recontacting providers.""" + +from pathlib import Path + +import pytest +from vcr.serializers import yamlserializer + +from ._test_audio_cassettes import AudioPersister, load_websocket, save_websocket + + +@pytest.mark.parametrize("kind", ["http", "websocket"]) +@pytest.mark.parametrize("failure_at", ["audio", "manifest"]) +def test_failed_save_preserves_previous_fixture(tmp_path, vcr_cassette_dir, monkeypatch, kind, failure_at): + from . import _test_audio_cassettes as storage + + source = Path(vcr_cassette_dir) + if kind == "http": + path = tmp_path / "test_cascade_voice_conversation.yaml" + requests, responses = AudioPersister.load_cassette(source / path.name, yamlserializer) + + def save(): + AudioPersister.save_cassette(path, {"requests": requests, "responses": responses}, yamlserializer) + else: + path = tmp_path / "test_realtime_voice_conversation.json" + data = load_websocket(source / path.name) + + def save(): + save_websocket(path, data["endpoint"], data["events"]) + + save() + before = {p.relative_to(tmp_path): p.read_bytes() for p in tmp_path.rglob("*") if p.is_file()} + original_write = storage.write_audio + + def failed_write(destination, data, **kwargs): + original_write(destination, data, **kwargs) + destination.write_bytes(b"partial write") + raise OSError("disk write failed") + + original_replace = Path.replace + + def failed_replace(source, target): + if Path(target) == path: + raise OSError("disk write failed") + return original_replace(source, target) + + with monkeypatch.context() as patch: + if failure_at == "audio": + patch.setattr(storage, "write_audio", failed_write) + else: + patch.setattr(Path, "replace", failed_replace) + with pytest.raises(OSError, match="disk write failed"): + save() + assert {p.relative_to(tmp_path): p.read_bytes() for p in tmp_path.rglob("*") if p.is_file()} == before + + +@pytest.mark.parametrize("validated", [False, True]) +def test_http_rerecord_promotes_only_validated_conversation(tmp_path, vcr_cassette_dir, validated): + from types import SimpleNamespace + + import vcr + + from .test_voice_pipeline import vcr_cassette_name + + name = "test_cascade_voice_conversation" + target = tmp_path / f"{name}.yaml" + requests, responses = AudioPersister.load_cassette(Path(vcr_cassette_dir) / target.name, yamlserializer) + AudioPersister.save_cassette(target, {"requests": requests, "responses": responses}, yamlserializer) + target.write_text("# previous recording\n" + target.read_text()) + before = {p.relative_to(tmp_path): p.read_bytes() for p in tmp_path.rglob("*") if p.is_file()} + recorder = vcr.VCR(record_mode="all") + recorder.register_persister(AudioPersister) + request = SimpleNamespace(node=SimpleNamespace(originalname=name)) + fixture = vcr_cassette_name.__wrapped__(request, str(tmp_path), recorder) + with recorder.use_cassette(next(fixture)) as cassette: + assert not cassette.requests, "re-recording must not append old interactions" + for req, response in zip(requests, responses): + cassette.append(req, response) + request.node.voice_recording_validated = validated + with pytest.raises(StopIteration): + next(fixture) + after = {p.relative_to(tmp_path): p.read_bytes() for p in tmp_path.rglob("*") if p.is_file()} + if validated: + assert not target.read_text().startswith("# previous recording") + loaded_requests, loaded_responses = AudioPersister.load_cassette(target, yamlserializer) + assert [r._to_dict() for r in loaded_requests] == [r._to_dict() for r in requests] + assert "!!binary" not in target.read_text() + assert loaded_responses == responses + else: + assert after == before diff --git a/py/src/braintrust/integrations/pipecat/test_pipecat.py b/py/src/braintrust/integrations/pipecat/test_pipecat.py index ab8511857..4883e0e36 100644 --- a/py/src/braintrust/integrations/pipecat/test_pipecat.py +++ b/py/src/braintrust/integrations/pipecat/test_pipecat.py @@ -2,7 +2,6 @@ import asyncio import importlib -import inspect import os import tempfile from pathlib import Path @@ -13,7 +12,6 @@ from braintrust.integrations.pipecat import ( BraintrustPipecatObserver, setup_pipecat, - wrap_pipeline_worker, ) from braintrust.integrations.test_utils import verify_autoinstrument_script from braintrust.integrations.versioning import detect_module_version, version_satisfies @@ -28,6 +26,11 @@ def memory_logger(): yield bgl +@pytest.fixture +def vcr_cassette_name(request): + return request.node.originalname or request.node.name + + def _ensure_nltk_punkt_tab(): data_dir = Path(tempfile.gettempdir()) / "braintrust-pipecat-nltk-data" punkt_tab = data_dir / "tokenizers" / "punkt_tab" @@ -59,33 +62,6 @@ def _single_span(logs, name): return matches[0] -def test_pipecat_observer_filters_metrics_from_other_processors(): - LLMTokenUsage = _import("pipecat.metrics.metrics.LLMTokenUsage") - LLMUsageMetricsData = _import("pipecat.metrics.metrics.LLMUsageMetricsData") - MetricsFrame = _import("pipecat.frames.frames.MetricsFrame") - - observer = BraintrustPipecatObserver() - processor = SimpleNamespace(name="OpenAILLMService#1") - frame = MetricsFrame( - data=[ - LLMUsageMetricsData( - processor=processor.name, - model="gpt-4o-mini", - value=LLMTokenUsage(prompt_tokens=10, completion_tokens=5, total_tokens=15), - ), - LLMUsageMetricsData( - processor="JevClassifier#1", - model="gpt-4o-mini", - value=LLMTokenUsage(prompt_tokens=100, completion_tokens=20, total_tokens=120), - ), - ] - ) - - observer._capture_metrics(frame, processor) - - assert observer._llm_metrics == {"prompt_tokens": 10, "completion_tokens": 5, "tokens": 15} - - @pytest.mark.asyncio async def test_span_customizer_redacts_incremental_tts_input(memory_logger): TTSStartedFrame = _import("pipecat.frames.frames.TTSStartedFrame") @@ -125,145 +101,62 @@ def on_span_export(self, data): set_span_customizers(None) -def _pipeline_worker_kwargs(**overrides): - PipelineWorker = _import("pipecat.pipeline.worker.PipelineWorker") - signature = inspect.signature(PipelineWorker) - kwargs = {"idle_timeout_secs": None} - for name, value in { - "enable_turn_tracking": False, - "enable_rtvi": False, - "check_dangling_tasks": False, - }.items(): - if name in signature.parameters: - kwargs[name] = value - kwargs.update(overrides) - return kwargs - - def _make_worker(pipeline, **overrides): PipelineWorker = _import("pipecat.pipeline.worker.PipelineWorker") - return PipelineWorker(pipeline, **_pipeline_worker_kwargs(**overrides)) + return PipelineWorker(pipeline, **overrides) def _worker_runner_kwargs(**overrides): - WorkerRunner = _import("pipecat.workers.runner.WorkerRunner") - signature = inspect.signature(WorkerRunner) - kwargs = {"handle_sigint": False} - if "check_dangling_tasks" in signature.parameters: - kwargs["check_dangling_tasks"] = False - kwargs.update(overrides) - return kwargs - - -@pytest.mark.asyncio -async def test_pipecat_observer_capture_audio_attachments_adds_tts_and_user_audio(memory_logger): - TTSStartedFrame = _import("pipecat.frames.frames.TTSStartedFrame") - TTSTextFrame = _import("pipecat.frames.frames.TTSTextFrame") - TTSAudioRawFrame = _import("pipecat.frames.frames.TTSAudioRawFrame") - TTSStoppedFrame = _import("pipecat.frames.frames.TTSStoppedFrame") - UserStartedSpeakingFrame = _import("pipecat.frames.frames.UserStartedSpeakingFrame") - UserAudioRawFrame = _import("pipecat.frames.frames.UserAudioRawFrame") - UserStoppedSpeakingFrame = _import("pipecat.frames.frames.UserStoppedSpeakingFrame") - - observer = BraintrustPipecatObserver(capture_audio_attachments=True) - first_chunk = b"\x00\x00\x01\x00" * 20 - second_chunk = b"\x02\x00\x03\x00" * 10 - - await observer.on_pipeline_started() - await observer._handle_frame(TTSStartedFrame(context_id="ctx")) - await observer._handle_frame(TTSTextFrame("hello from tts", aggregated_by="sentence", context_id="ctx")) - await observer._handle_frame(TTSAudioRawFrame(first_chunk, sample_rate=16000, num_channels=1, context_id="ctx")) - await observer._handle_frame(TTSAudioRawFrame(second_chunk, sample_rate=16000, num_channels=1, context_id="ctx")) - await observer._handle_frame(TTSStoppedFrame(context_id="ctx")) - await observer._handle_frame(UserStartedSpeakingFrame()) - await observer._handle_frame(UserAudioRawFrame(first_chunk, sample_rate=16000, num_channels=1, user_id="user-1")) - await observer._handle_frame(UserAudioRawFrame(second_chunk, sample_rate=16000, num_channels=1, user_id="user-1")) - await observer._handle_frame(UserStoppedSpeakingFrame()) - await observer.cleanup() - - logs = memory_logger.pop() - tts_span = _single_span(logs, "tts_response") - tts_audio = tts_span["output"]["audio"] - assert isinstance(tts_audio, Attachment) - assert tts_audio.reference["content_type"] == "audio/wav" - assert tts_audio.data.startswith(b"RIFF") - assert tts_span["output"]["audio_size_bytes"] == len(first_chunk) + len(second_chunk) - assert tts_span["output"]["sample_rate"] == 16000 - assert tts_span["output"]["num_channels"] == 1 - assert tts_span["output"]["num_frames"] == 60 - - user_span = _single_span(logs, "user_speaking") - user_audio = user_span["input"]["audio"] - assert isinstance(user_audio, Attachment) - assert user_audio.reference["content_type"] == "audio/wav" - assert user_audio.data.startswith(b"RIFF") - assert user_span["input"]["audio_size_bytes"] == len(first_chunk) + len(second_chunk) - assert user_span["input"]["num_frames"] == 60 - assert user_span["input"]["user_id"] == "user-1" - - observer_without_start = BraintrustPipecatObserver(capture_audio_attachments=True) - await observer_without_start.on_pipeline_started() - await observer_without_start._handle_frame( - UserAudioRawFrame(first_chunk, sample_rate=16000, num_channels=1, user_id="user-1") - ) - await observer_without_start._handle_frame( - UserAudioRawFrame(second_chunk, sample_rate=16000, num_channels=1, user_id="user-1") - ) - await observer_without_start._handle_frame(UserStoppedSpeakingFrame()) - await observer_without_start.cleanup() - - logs = memory_logger.pop() - auto_started_user_span = _single_span(logs, "user_speaking") - assert auto_started_user_span["input"]["num_frames"] == 60 + # pytest owns process signal handling; retain all pipeline/lifecycle defaults. + return {"handle_sigint": False, **overrides} -@pytest.mark.parametrize( - ("capture_user", "capture_agent"), - [(True, False), (False, True)], - ids=["user-only", "agent-only"], -) +@pytest.mark.parametrize("capture_user,capture_agent", [(False, False), (True, False), (False, True), (True, True)]) @pytest.mark.asyncio -async def test_pipecat_observer_audio_capture_env_vars_are_independent( - monkeypatch, memory_logger, capture_user, capture_agent -): - TTSStartedFrame = _import("pipecat.frames.frames.TTSStartedFrame") - TTSAudioRawFrame = _import("pipecat.frames.frames.TTSAudioRawFrame") - TTSStoppedFrame = _import("pipecat.frames.frames.TTSStoppedFrame") - UserStartedSpeakingFrame = _import("pipecat.frames.frames.UserStartedSpeakingFrame") - UserAudioRawFrame = _import("pipecat.frames.frames.UserAudioRawFrame") - UserStoppedSpeakingFrame = _import("pipecat.frames.frames.UserStoppedSpeakingFrame") - audio = b"\x00\x00\x01\x00" * 20 +async def test_legacy_audio_capture_policy(monkeypatch, memory_logger, capture_user, capture_agent): + # Local PCM handling and opt-in policy have no HTTP behavior to record. + import io + import wave + frames = importlib.import_module("pipecat.frames.frames") monkeypatch.setenv("BRAINTRUST_CAPTURE_USER_AUDIO_ATTACHMENTS", str(capture_user).lower()) monkeypatch.setenv("BRAINTRUST_CAPTURE_AGENT_AUDIO_ATTACHMENTS", str(capture_agent).lower()) - observer = BraintrustPipecatObserver() - assert observer.capture_user_audio_attachments is capture_user - assert observer.capture_agent_audio_attachments is capture_agent - + observer = BraintrustPipecatObserver(trace_turns=False) + chunks = [b"\x01\x00" * 40, b"\x02\x00" * 20] await observer.on_pipeline_started() - await observer._handle_frame(TTSStartedFrame(context_id="ctx")) - await observer._handle_frame(TTSAudioRawFrame(audio, sample_rate=16000, num_channels=1, context_id="ctx")) - await observer._handle_frame(TTSStoppedFrame(context_id="ctx")) - await observer._handle_frame(UserStartedSpeakingFrame()) - await observer._handle_frame(UserAudioRawFrame(audio, sample_rate=16000, num_channels=1, user_id="user-1")) - await observer._handle_frame(UserStoppedSpeakingFrame()) + await observer._handle_frame(frames.TTSStartedFrame(context_id="ctx")) + for chunk in chunks: + await observer._handle_frame(frames.TTSAudioRawFrame(chunk, 16000, 1, context_id="ctx")) + await observer._handle_frame(frames.TTSStoppedFrame(context_id="ctx")) + # Cover both explicit speech start and the implicit start on first audio. + if not capture_agent: + await observer._handle_frame(frames.UserStartedSpeakingFrame()) + for chunk in chunks: + await observer._handle_frame(frames.UserAudioRawFrame(chunk, 16000, 1, user_id="user-1")) + await observer._handle_frame(frames.UserStoppedSpeakingFrame()) await observer.cleanup() - logs = memory_logger.pop() - tts_output = _single_span(logs, "tts_response").get("output", {}) - if capture_agent: - assert isinstance(tts_output["audio"], Attachment) - else: - assert "audio" not in tts_output - if capture_user: - assert isinstance(_single_span(logs, "user_speaking")["input"]["audio"], Attachment) - else: - assert not _spans_named(logs, "user_speaking") + tts = _single_span(logs, "tts_response").get("output", {}) + user = _single_span(logs, "user_speaking")["input"] if capture_user else {} + assert ("audio" in tts) is capture_agent + assert bool(_spans_named(logs, "user_speaking")) is capture_user + for payload in (tts, user): + if "audio" not in payload: + continue + assert isinstance(payload["audio"], Attachment) + assert payload["audio"].reference["content_type"] == "audio/wav" + with wave.open(io.BytesIO(payload["audio"].data)) as audio: + assert audio.getframerate() == 16000 + assert audio.getnchannels() == 1 + assert audio.readframes(audio.getnframes()) == b"".join(chunks) + assert payload["audio_size_bytes"] == 120 + assert payload["num_frames"] == 60 @pytest.mark.vcr +@pytest.mark.parametrize("native", [False, True], ids=["legacy", "native"]) @pytest.mark.asyncio -async def test_setup_pipecat_traces_real_pipeline_frames(memory_logger): +async def test_setup_pipecat_traces_real_pipeline_frames(memory_logger, native): EndFrame = _import("pipecat.frames.frames.EndFrame") LLMContextFrame = _import("pipecat.frames.frames.LLMContextFrame") Pipeline = _import("pipecat.pipeline.pipeline.Pipeline") @@ -272,6 +165,10 @@ async def test_setup_pipecat_traces_real_pipeline_frames(memory_logger): WorkerRunner = _import("pipecat.workers.runner.WorkerRunner") PipelineParams = _import("pipecat.pipeline.worker.PipelineParams") + if native and not version_satisfies( + detect_module_version(importlib.import_module("pipecat"), ("pipecat",)), ">=1.12.0" + ): + pytest.skip("Native voice hooks require Pipecat 1.12") assert setup_pipecat(project_name="test-project-pipecat-py-tracing") init_test_logger("test-project-pipecat-py-tracing") llm = OpenAILLMService( @@ -282,8 +179,20 @@ async def test_setup_pipecat_traces_real_pipeline_frames(memory_logger): max_completion_tokens=20, ), ) + processors = [llm] + if native: + pair = _import("pipecat.processors.aggregators.llm_response_universal.LLMContextAggregatorPair")(LLMContext()) + params = _import("pipecat.transports.base_transport.TransportParams")() + processors = [ + _import("pipecat.transports.base_input.BaseInputTransport")(params), + _import("pipecat.services.openai.stt.OpenAISTTService")(api_key="unused"), + pair.user(), + llm, + _import("pipecat.transports.base_output.BaseOutputTransport")(params), + pair.assistant(), + ] worker = _make_worker( - Pipeline([llm]), + Pipeline(processors), name="bt-pipecat-test-worker", params=PipelineParams(enable_metrics=True, enable_usage_metrics=True), ) @@ -303,13 +212,26 @@ async def on_pipeline_started(_worker, _frame): await asyncio.wait_for(runner.run(), timeout=20) observer = next(o for o in getattr(worker, "_observer")._observers if isinstance(o, BraintrustPipecatObserver)) - if version_satisfies(detect_module_version(importlib.import_module("pipecat"), ("pipecat",)), ">=1.12.0"): - assert observer.observe_every_push is False - assert not hasattr(observer, "_seen_frame_ids") - else: - assert observer._seen_frame_ids + assert (observer._voice is not None) is native logs = memory_logger.pop() + if native: + pipeline_span = _single_span(logs, "pipecat.pipeline") + turn = _single_span(logs, "assistant_turn") + llm_span = _single_span(logs, "llm_response") + assert turn["span_parents"] == [pipeline_span["span_id"]] + assert llm_span["span_parents"] == [turn["span_id"]] + assert llm_span["metadata"]["turn.id"] == turn["span_id"] + assert llm_span["span_attributes"]["type"] == "llm" + assert "braintrust pipecat integration" in llm_span["output"][0]["content"].lower() + assert llm_span["metadata"]["contrib.pipecat.usage"]["value"]["total_tokens"] == llm_span["metrics"]["tokens"] + assert llm_span["metadata"]["model"] == "gpt-4o-mini" + assert llm_span["metadata"]["provider"] == "openai" + assert llm_span["metrics"]["prompt_tokens"] > 0 + assert llm_span["metrics"]["completion_tokens"] > 0 + assert llm_span["metrics"]["time_to_first_token"] >= 0 + assert llm_span["metadata"]["contrib.pipecat.ttfb"] + return pipeline_span = _single_span(logs, "pipecat_pipeline") assert _span_type(pipeline_span) == "task" assert pipeline_span.get("metrics", {}).get("end") is not None @@ -339,34 +261,136 @@ async def on_pipeline_started(_worker, _frame): assert llm_span["metrics"]["tokens"] >= llm_span["metrics"]["completion_tokens"] -def test_setup_and_wrap_pipeline_worker_are_idempotent(): - Pipeline = _import("pipecat.pipeline.pipeline.Pipeline") - PipelineWorker = _import("pipecat.pipeline.worker.PipelineWorker") - IdentityFilter = _import("pipecat.processors.filters.identity_filter.IdentityFilter") - - assert setup_pipecat(project_name="test-project-pipecat-py-tracing") - assert setup_pipecat(project_name="test-project-pipecat-py-tracing") - assert setup_pipecat(project_name="test-project-pipecat-py-tracing", capture_audio_attachments=True) - assert wrap_pipeline_worker(PipelineWorker) is PipelineWorker - - capturing_worker = _make_worker(Pipeline([IdentityFilter()])) - capturing_observers = getattr(getattr(capturing_worker, "_observer"), "_observers") - capturing_bt_observer = next( - observer for observer in capturing_observers if isinstance(observer, BraintrustPipecatObserver) - ) - assert capturing_bt_observer.capture_audio_attachments is True - - assert setup_pipecat(project_name="test-project-pipecat-py-tracing", capture_audio_attachments=False) - explicit_observer = BraintrustPipecatObserver() - worker = _make_worker(Pipeline([IdentityFilter()]), observers=[explicit_observer]) - worker_observer = getattr(worker, "_observer") - observers = getattr(worker_observer, "_observers") - braintrust_observers = [observer for observer in observers if isinstance(observer, BraintrustPipecatObserver)] - assert braintrust_observers == [explicit_observer] - - @pytest.mark.vcr @pytest.mark.skipif(__import__("sys").version_info < (3, 11), reason="Pipecat AI 1.x requires Python 3.11+") def test_auto_instrument_pipecat_subprocess(): pytest.importorskip("pipecat") verify_autoinstrument_script("test_auto_pipecat.py") + + +@pytest.mark.parametrize("metric_name,capture_audio", [("TurnMetricsData", False), ("SmartTurnMetricsData", True)]) +@pytest.mark.asyncio +async def test_legacy_turn_metrics_follow_speech_or_pipeline(memory_logger, metric_name, capture_audio): + frames = importlib.import_module("pipecat.frames.frames") + metric = _import(f"pipecat.metrics.metrics.{metric_name}")( + processor="BaseSmartTurn", is_complete=True, probability=0.97, e2e_processing_time_ms=82.4 + ) + observer = BraintrustPipecatObserver(trace_turns=False, capture_audio_attachments=capture_audio) + await observer.on_pipeline_started() + await observer._handle_frame(frames.UserStartedSpeakingFrame()) + await observer._handle_frame(frames.MetricsFrame(data=[metric])) + await observer.cleanup() + span = _single_span(memory_logger.pop(), "user_speaking" if capture_audio else "pipecat_pipeline") + assert span["metadata"]["contrib.pipecat.turn_metrics"] == [ + dict( + type=metric_name, + processor="BaseSmartTurn", + is_complete=True, + probability=0.97, + e2e_processing_time_ms=82.4, + ) + ] + + +@pytest.mark.asyncio +async def test_ttfb_routes_by_processor_and_retains_unmatched(memory_logger): + observer = BraintrustPipecatObserver(trace_turns=False, capture_audio_attachments=False) + frames = importlib.import_module("pipecat.frames.frames") + metric_class = _import("pipecat.metrics.metrics.TTFBMetricsData") + processor_class = _import("pipecat.processors.frame_processor.FrameProcessor") + llm, tts, stt = [processor_class(name=name) for name in ("llm", "tts", "stt")] + await observer._handle_frame(frames.LLMFullResponseStartFrame(), processor=llm) + await observer._handle_frame(frames.TTSStartedFrame(), processor=tts) + for processor, value in ((tts, 0.09), (stt, 0.12), (llm, 0.24)): + await observer._handle_frame( + frames.MetricsFrame(data=[metric_class(processor=processor.name, value=value)]), processor=processor + ) + usage_class = _import("pipecat.metrics.metrics.LLMUsageMetricsData") + tokens_class = _import("pipecat.metrics.metrics.LLMTokenUsage") + await observer._handle_frame( + frames.MetricsFrame( + data=[ + usage_class( + processor="llm", value=tokens_class(prompt_tokens=10, completion_tokens=5, total_tokens=15) + ), + usage_class( + processor="unrelated-classifier", + value=tokens_class(prompt_tokens=100, completion_tokens=20, total_tokens=120), + ), + ] + ), + processor=llm, + ) + await observer._handle_frame(frames.TTSStoppedFrame(), processor=tts) + observer._close_all_open_spans() + logs = memory_logger.pop() + assert _single_span(logs, "tts_response")["metadata"]["contrib.pipecat.ttfb"][0]["value"] == 0.09 + assert _single_span(logs, "pipecat_llm_response")["metrics"]["time_to_first_token"] == 0.24 + assert _single_span(logs, "pipecat_llm_response")["metrics"]["tokens"] == 15 + assert _single_span(logs, "pipecat_pipeline")["metadata"]["contrib.pipecat.ttfb"][0]["processor"] == "stt" + + +def test_ttfb_router_bounds_and_rejects_mismatched_sources(): + from braintrust.integrations.pipecat.ttfb import TTFBRouter + + metric_class = _import("pipecat.metrics.metrics.TTFBMetricsData") + root, operation = [], [] + router = TTFBRouter(lambda **row: root.append(row)) + source = SimpleNamespace(name="tts") + router.start("a", source, lambda **row: operation.append(row)) + metric = metric_class(processor="tts", value=0.1) + router.capture(metric, SimpleNamespace(name="tts")) + assert not operation # Same name is not the same processor instance. + for _ in range(35): + router.capture(metric, source) + assert operation[-1]["metadata"]["braintrust.ttfb.omitted"] == 3 + measurement = _import("pipecat.metrics.metrics.ProcessingMetricsData")(processor="tts", value=0.5) + router.capture_measurement(measurement, SimpleNamespace(name="tts")) + assert root[-1]["metadata"]["contrib.pipecat.measurements"][0]["type"] == "ProcessingMetricsData" + for _ in range(35): + router.capture_measurement(measurement, source) + assert operation[-1]["metadata"]["braintrust.measurements.omitted"] == 3 + retained = [ + row["metadata"]["contrib.pipecat.measurements"] + for row in operation + if "contrib.pipecat.measurements" in row["metadata"] + ] + assert len(retained[-1]) == 32 + router.start("a", source, lambda **row: operation.append(row)) + assert router.capture(metric, source) is None + router.clear() + assert not router.active + + +def test_request_metrics_keep_identity_across_overlap_shutdown_and_limits(): + from braintrust.integrations.pipecat.ttfb import TTFBRouter + + usage = _import("pipecat.metrics.metrics.TTSUsageMetricsData") + root, first, second = [], [], [] + source = SimpleNamespace(name="tts") + router = TTFBRouter(lambda **row: root.append(row)) + a, b = ("tts", "a"), ("tts", "b") + router.start(b, source, lambda **row: second.append(row)) + router.capture_request(usage(processor="tts", value=11), source, a) + assert not root and not second # An active sibling is not the request's owner. + router.start(a, source, lambda **row: first.append(row)) + router.capture_request(usage(processor="tts", value=22), source, b) + router.end(a) + router.capture_request(usage(processor="tts", value=33), source, a) + assert [m["value"] for m in first[-1]["metadata"]["contrib.pipecat.measurements"]] == [11, 33] + assert [m["value"] for m in second[-1]["metadata"]["contrib.pipecat.measurements"]] == [22] + router.capture_request(usage(processor="tts", value=44), SimpleNamespace(name="tts"), b) + assert root[-1]["metadata"]["contrib.pipecat.measurements"][0]["value"] == 44 + # Requests that never produce audio, including cancellation, cannot grow + # retention without bound or be silently dropped at shutdown. + for index in range(300): + router.capture_request(usage(processor="tts", value=index), source, ("tts", str(index))) + assert len(router.pending) == 256 + for index in range(100): + key = ("tts", f"completed-{index}") + router.start(key, source, lambda **row: None) + router.end(key) + assert len(router.completed) == 64 + router.clear() + assert not router.pending and not router.completed and not router.active + assert root[-1]["metadata"]["braintrust.measurements.omitted"] == 269 diff --git a/py/src/braintrust/integrations/pipecat/test_voice_pipeline.py b/py/src/braintrust/integrations/pipecat/test_voice_pipeline.py new file mode 100644 index 000000000..62c60645b --- /dev/null +++ b/py/src/braintrust/integrations/pipecat/test_voice_pipeline.py @@ -0,0 +1,448 @@ +"""Provider-backed voice contract: HTTP cassettes, real Pipecat pipeline and export.""" + +# pylint: disable=import-error +import asyncio +import io +import json +import os +import wave +from email.parser import BytesParser +from email.policy import default +from pathlib import Path +from tempfile import TemporaryDirectory + +import numpy as np +import pytest +from braintrust._audio import RecordingOptions +from braintrust.integrations.pipecat import setup_pipecat +from braintrust.integrations.pipecat.test_pipecat import ( + _make_worker, + _single_span, + _worker_runner_kwargs, +) +from braintrust.integrations.pipecat.test_pipecat import memory_logger as memory_logger +from braintrust.test_helpers import init_test_logger +from pipecat.adapters.schemas.function_schema import FunctionSchema +from pipecat.adapters.schemas.tools_schema import ToolsSchema +from pipecat.frames.frames import InputAudioRawFrame, VADUserStartedSpeakingFrame, VADUserStoppedSpeakingFrame +from pipecat.pipeline.pipeline import Pipeline +from pipecat.pipeline.worker import PipelineParams +from pipecat.processors.aggregators.llm_context import LLMContext +from pipecat.processors.aggregators.llm_response_universal import LLMContextAggregatorPair +from pipecat.services.openai.llm import OpenAILLMService +from pipecat.services.openai.stt import OpenAISTTService +from pipecat.services.openai.tts import OpenAITTSService +from pipecat.transports.base_input import BaseInputTransport +from pipecat.transports.base_output import BaseOutputTransport +from pipecat.transports.base_transport import TransportParams +from pipecat.workers.runner import WorkerRunner + + +def _request_payload(request): + content_type = request.headers.get("Content-Type", "") + body = request.body + if "multipart/form-data" in content_type: + message = BytesParser(policy=default).parsebytes(f"Content-Type: {content_type}\r\n\r\n".encode() + body) + return [(part.get("Content-Disposition"), part.get_payload(decode=True)) for part in message.iter_parts()] + return json.loads(body) + + +@pytest.fixture(scope="module") +def vcr(vcr): + # Compare audio bytes, prompts, model/settings and tool results. Multipart + # boundaries are random HTTP framing, not part of the request's meaning. + def payload_matches(actual, expected): + assert _request_payload(actual) == _request_payload(expected) + + from braintrust.integrations.pipecat._test_audio_cassettes import AudioPersister + + vcr.register_persister(AudioPersister) + vcr.register_matcher("voice_payload", payload_matches) + return vcr + + +@pytest.fixture +def vcr_cassette_name(request, vcr_cassette_dir, vcr): + name = request.node.originalname or request.node.name + if vcr.record_mode != "all": + yield name + return + from braintrust.integrations.pipecat._test_audio_cassettes import publish_cassette + + # A fresh temporary path avoids VCR appending old interactions. The + # dependent VCR fixture saves here before this fixture's teardown runs. + target = Path(vcr_cassette_dir) / f"{name}.yaml" + target.parent.mkdir(parents=True, exist_ok=True) + with TemporaryDirectory(prefix=".recording-", dir=target.parent) as directory: + staged = Path(directory) / target.name + yield str(staged) + if getattr(request.node, "voice_recording_validated", False): + publish_cassette(staged, target) + + +class MemoryInput(BaseInputTransport): + async def start(self, frame): + await super().start(frame) + await self.set_transport_ready(frame) + + +class MemoryOutput(BaseOutputTransport): + """Local application transport: accept output without a telephone or playback delay.""" + + async def start(self, frame): + await super().start(frame) + await self.set_transport_ready(frame) + + async def write_audio_frame(self, frame): + self.last_audio_frame = frame + return True + + +async def _run_conversation(worker, finished): + runner = WorkerRunner(**_worker_runner_kwargs()) + task = asyncio.create_task(runner.run(worker)) + try: + await asyncio.wait_for(finished.wait(), 45) + await worker.stop_when_done() + await asyncio.wait_for(task, 10) + finally: + if not task.done(): + await worker.cancel() + await task + + +def _decode_recordings(rows): + spans = {row["span_id"]: row for row in rows} + recordings = {} + for owner in rows: + for descriptor in owner.get("metadata", {}).get("audio.recordings", []): + assert descriptor["state"] == "ready", descriptor + attachment = descriptor["attachment"] + value = spans[attachment["span_id"]] + for part in attachment["ref"].strip("/").split("/"): + part = part.replace("~1", "/").replace("~0", "~") + value = value[int(part)] if isinstance(value, list) else value[part] + with wave.open(io.BytesIO(value.data)) as audio: + rate = audio.getframerate() + channels = audio.getnchannels() + samples = np.frombuffer(audio.readframes(audio.getnframes()), dtype=" rate * 2 + assert np.any(samples[: -rate * 2, 1]), "agent speech must survive shutdown" + assert not np.any(samples[-rate * 2 :, 1]), "closing silence must be retained" + + +@pytest.mark.vcr(match_on=["method", "uri", "voice_payload"]) +@pytest.mark.asyncio +@pytest.mark.parametrize("early_metrics", [False, True]) +async def test_cascade_voice_conversation(memory_logger, request, monkeypatch, early_metrics): + setup_pipecat(capture_audio_attachments=True, audio_format="wav", recording_options=RecordingOptions()) + init_test_logger("test-project-pipecat-py-tracing") + tool_calls = [] + + async def lookup_order(params): + tool_calls.append(params.arguments) + await params.result_callback({"order_id": params.arguments["order_id"], "delivery": "Friday"}) + + context = LLMContext( + messages=[ + { + "role": "system", + "content": "You are an English order assistant. Always call lookup_order for the supplied order number. " + "After its result, answer only: Your order arrives Friday.", + } + ], + tools=ToolsSchema( + standard_tools=[ + FunctionSchema( + name="lookup_order", + description="Look up an order", + properties={"order_id": {"type": "string"}}, + required=["order_id"], + handler=lookup_order, + ) + ] + ), + ) + pair = LLMContextAggregatorPair(context) + params = TransportParams(audio_in_enabled=True, audio_out_enabled=True) + source, output = MemoryInput(params), MemoryOutput(params) + key = os.environ["OPENAI_API_KEY"] + stt = OpenAISTTService(api_key=key) + llm = OpenAILLMService(api_key=key, settings=OpenAILLMService.Settings(model="gpt-4.1-mini", temperature=0)) + tts = OpenAITTSService( + api_key=key, + settings=OpenAITTSService.Settings(model="gpt-4o-mini-tts", voice="alloy"), + ) + if early_metrics: + # Exercise native audio-queue scheduling without replacing the service, + # metric emission, or provider responses. + usage_emitted = asyncio.Event() + original_push = tts.push_frame + + async def delayed_start(frame, *args, **kwargs): + if type(frame).__name__ == "TTSStartedFrame": + await usage_emitted.wait() # Covered by the conversation's overall deadline. + await asyncio.sleep(0) + result = await original_push(frame, *args, **kwargs) + if type(frame).__name__ == "MetricsFrame" and any( + type(metric).__name__ == "TTSUsageMetricsData" for metric in frame.data + ): + usage_emitted.set() + return result + + monkeypatch.setattr(tts, "push_frame", delayed_start) + worker = _make_worker( + Pipeline([source, stt, pair.user(), llm, tts, output, pair.assistant()]), + params=PipelineParams( + audio_in_sample_rate=16000, audio_out_sample_rate=24000, enable_metrics=True, enable_usage_metrics=True + ), + ) + finished = asyncio.Event() + + @pair.assistant().event_handler("on_assistant_turn_stopped") + async def turn_finished(aggregator, message): + if "friday" in message.content.lower(): + finished.set() + + @worker.event_handler("on_pipeline_started") + async def feed(worker, frame): + # A fixed recording and speech boundaries are the test input. Recognition, + # model/tool behavior, synthesis and turn callbacks all run through Pipecat. + with wave.open(str(Path(__file__).parent / "voice/fixtures/order.wav")) as audio: + pcm = audio.readframes(audio.getnframes()) + await source.push_frame(VADUserStartedSpeakingFrame()) + for offset in range(0, len(pcm), 640): + await source.push_audio_frame(InputAudioRawFrame(pcm[offset : offset + 640], 16000, 1)) + await asyncio.sleep(0.02) + await source.push_frame(VADUserStoppedSpeakingFrame()) + + await _run_conversation(worker, finished) + rows = memory_logger.pop() + assert not [r for r in rows if r.get("error")] + assert len(tool_calls) == 1 + assert "1042" in tool_calls[0]["order_id"].replace(" ", "") + root = _single_span(rows, "pipecat.pipeline") + recordings = _decode_recordings(rows) + _assert_shutdown_audio(output, root, recordings) + user = _single_span(rows, "user_turn") + recognition = _single_span(rows, "stt") + tool = _single_span(rows, "lookup_order") + synthesis = _single_span(rows, "tts") + assert synthesis["input"]["text"] + # Request identity must survive early metrics, without root duplicates. + tts_measurements = [ + (row, measurement) + for row in rows + for measurement in row.get("metadata", {}).get("contrib.pipecat.measurements", []) + if measurement["processor"].startswith("OpenAITTSService") + ] + for metric_type in ("TTSUsageMetricsData", "ProcessingMetricsData"): + matching = [(row, metric) for row, metric in tts_measurements if metric["type"] == metric_type] + assert len(matching) == 1 + owner, metric = matching[0] + assert owner["span_id"] == synthesis["span_id"] + assert metric["value"] > 0 + assert all( + metric["processor"].startswith("OpenAITTSService") + for metric in synthesis.get("metadata", {}).get("contrib.pipecat.measurements", []) + ) + for row in rows: + assert not any(key.startswith("pipecat.") for key in row.get("metadata", {})) + events = row.get("metadata", {}).get("contrib.pipecat.events", []) + assert not any(e["contrib.pipecat.frame.type"] in {"StartFrame", "EndFrame", "MetricsFrame"} for e in events) + models = [r for r in rows if r.get("span_attributes", {}).get("name") == "llm_response"] + assert models + assert recognition["span_parents"] == [user["span_id"]] + assert any(tool["span_parents"] == [r["span_id"]] for r in models) + assert any( + r["span_id"] not in tool["span_parents"] and "friday" in json.dumps(r["output"]).lower() for r in models + ), "a model response must deliver the tool result to the caller" + assert "friday" in synthesis["metadata"]["contrib.pipecat.text"].lower() + for model in models: + assert model["metadata"]["model"] == "gpt-4.1-mini" + assert model["metrics"]["tokens"] > 0 + _assert_selections(user, root, recordings, 0) + _assert_selections(synthesis, root, recordings, 1) + request.node.voice_recording_validated = True + + +@pytest.mark.asyncio +async def test_realtime_voice_conversation(memory_logger, monkeypatch, request, vcr_cassette_dir): + from braintrust.integrations.pipecat._test_websocket import WebSocketCassette + from pipecat.frames.frames import LLMRunFrame + from pipecat.services.openai.realtime import events + from pipecat.services.openai.realtime import llm as realtime_module + + cassette = WebSocketCassette( + Path(vcr_cassette_dir) / "test_realtime_voice_conversation.json", + record=request.config.getoption("--vcr-record") == "all", + ) + original_connect = realtime_module.websocket_connect + + async def connect(**kwargs): + return await cassette.connect(original_connect, **kwargs) + + monkeypatch.setattr(realtime_module, "websocket_connect", connect) + setup_pipecat(capture_audio_attachments=True, audio_format="wav") + init_test_logger("test-project-pipecat-py-tracing") + tool_calls = [] + + async def lookup_order(params): + tool_calls.append(params.arguments) + await params.result_callback({"order_id": params.arguments["order_id"], "delivery": "Friday"}) + + context = LLMContext( + messages=[ + { + "role": "system", + "content": "Speak English. Start by saying exactly: Hello, what is your order number? " + "When the user gives an order number, always call lookup_order. " + "After its result, say only: Your order arrives Friday.", + } + ], + tools=ToolsSchema( + standard_tools=[ + FunctionSchema( + name="lookup_order", + description="Look up an order", + properties={"order_id": {"type": "string"}}, + required=["order_id"], + handler=lookup_order, + ) + ] + ), + ) + pair = LLMContextAggregatorPair(context) + params = TransportParams(audio_in_enabled=True, audio_out_enabled=True) + source, output = MemoryInput(params), MemoryOutput(params) + service = realtime_module.OpenAIRealtimeLLMService( + api_key=os.environ["OPENAI_API_KEY"], + settings=realtime_module.OpenAIRealtimeLLMService.Settings( + model="gpt-realtime", + session_properties=events.SessionProperties( + audio=events.AudioConfiguration( + input=events.AudioInput( + transcription=events.InputAudioTranscription(model="gpt-4o-mini-transcribe", language="en"), + turn_detection=events.TurnDetection(), + ), + output=events.AudioOutput(voice="alloy"), + ) + ), + ), + ) + worker = _make_worker( + Pipeline([source, pair.user(), service, output, pair.assistant()]), + params=PipelineParams( + audio_in_sample_rate=24000, audio_out_sample_rate=24000, enable_metrics=True, enable_usage_metrics=True + ), + ) + greeted, finished = asyncio.Event(), asyncio.Event() + + @pair.assistant().event_handler("on_assistant_turn_stopped") + async def turn_finished(aggregator, message): + if "friday" in message.content.lower(): + finished.set() + else: + greeted.set() + + @worker.event_handler("on_pipeline_started") + async def feed(worker, frame): + await worker.queue_frame(LLMRunFrame()) + await asyncio.wait_for(greeted.wait(), 20) + # Fixed wire-rate input keeps strict cassette matching independent of + # platform-specific resampler rounding. Recording uses this same input. + with wave.open(str(Path(__file__).parent / "voice/fixtures/order-24khz.wav")) as audio: + assert (audio.getframerate(), audio.getnchannels(), audio.getsampwidth()) == (24000, 1, 2) + pcm = audio.readframes(audio.getnframes()) + pcm += b"\0" * 48000 # Server VAD detects the end of the spoken request. + for offset in range(0, len(pcm), 960): + await source.push_audio_frame(InputAudioRawFrame(pcm[offset : offset + 960], 24000, 1)) + await asyncio.sleep(0.02) + + await _run_conversation(worker, finished) + cassette.verify() + rows = memory_logger.pop() + assert not [r for r in rows if r.get("error")] + assert len(tool_calls) == 1 + assert "1042" in tool_calls[0]["order_id"].replace(" ", "") + root = _single_span(rows, "pipecat.pipeline") + recordings = _decode_recordings(rows) + _assert_shutdown_audio(output, root, recordings) + user = _single_span(rows, "user_turn") + tool = _single_span(rows, "lookup_order") + models = [r for r in rows if r.get("span_attributes", {}).get("name") == "llm_response"] + assert models + assert any(tool["span_parents"] == [r["span_id"]] for r in models) + assert any( + r["span_id"] not in tool["span_parents"] and "friday" in json.dumps(r["output"]).lower() for r in models + ), "a model response must deliver the tool result to the caller" + _assert_selections(user, root, recordings, 0) + for model in models: + assert model["metadata"]["provider"] == "openai" + assert model["metadata"]["model"].startswith("gpt-realtime") + assert model["metrics"]["tokens"] > 0 + assert model["metadata"].get("contrib.pipecat.text") != "" + assert model["output"] + assert not [r for r in rows if r.get("span_attributes", {}).get("name") in ("stt", "tts")] + audio_spans = [r for r in rows if r.get("span_attributes", {}).get("name") == "pipecat.audio_output"] + assert audio_spans + for audio_span in audio_spans: + assert audio_span["metadata"]["openai.response.id"] + _assert_selections(audio_span, root, recordings, 1) + if cassette.record: + cassette.save() diff --git a/py/src/braintrust/integrations/pipecat/test_websocket_cassette.py b/py/src/braintrust/integrations/pipecat/test_websocket_cassette.py new file mode 100644 index 000000000..c900310f3 --- /dev/null +++ b/py/src/braintrust/integrations/pipecat/test_websocket_cassette.py @@ -0,0 +1,31 @@ +"""Guard the recorder itself: changed requests must never silently replay.""" + +import json + +import pytest + +from ._test_websocket import WebSocketCassette + + +@pytest.mark.asyncio +async def test_websocket_cassette_rejects_changed_payload(tmp_path): + path = tmp_path / "socket.json" + path.write_text( + json.dumps( + { + "endpoint": "wss://example.test", + "events": [ + { + "direction": "send", + "message": {"type": "response.create", "response": {"modalities": ["audio"]}}, + }, + ], + } + ) + ) + cassette = WebSocketCassette(path) + with pytest.raises(AssertionError, match="request mismatch"): + await cassette.send(json.dumps({"type": "response.create", "response": {"modalities": ["text"]}})) + with pytest.raises(AssertionError, match="replay failed"): + cassette.verify() + await cassette.close() diff --git a/py/src/braintrust/integrations/pipecat/tracing.py b/py/src/braintrust/integrations/pipecat/tracing.py index 554c70048..fbf697cde 100644 --- a/py/src/braintrust/integrations/pipecat/tracing.py +++ b/py/src/braintrust/integrations/pipecat/tracing.py @@ -1,8 +1,10 @@ """Observer-based tracing for Pipecat pipelines.""" import json +import logging from typing import Any +from braintrust._audio import RecordingOptions from braintrust.integrations.utils import ( _is_not_given, _normalize_chat_messages, @@ -13,6 +15,10 @@ from braintrust.logger import NOOP_SPAN, Attachment, SpanTypeAttribute, current_span from braintrust.logger import start_span as _bt_start_span +from .llm_metrics import _llm_usage_metrics, _metadata_from_metric, _metadata_from_processor +from .ttfb import TTFBRouter +from .turn_metrics import TURN_METRIC_TYPES, log_turn_metric + _INSTRUMENTATION = "pipecat-auto" @@ -78,6 +84,8 @@ def __init__( capture_user_audio_attachments: bool | None = None, capture_agent_audio_attachments: bool | None = None, trace_turns: bool = True, + audio_format: str = "ogg", + recording_options: RecordingOptions | None = None, **kwargs: Any, ) -> None: self._uses_native_frame_deduplication = _USES_NATIVE_FRAME_DEDUPLICATION @@ -93,6 +101,11 @@ def __init__( capture_agent_audio_attachments=capture_agent_audio_attachments, ) self.capture_audio_attachments = self.capture_user_audio_attachments or self.capture_agent_audio_attachments + if audio_format not in {"ogg", "wav"}: + raise ValueError("audio_format must be ogg or wav") + self.audio_format = audio_format + self.recording_options = recording_options + self._voice = None self.trace_turns = trace_turns self._parent = _current_parent_export() self._pipeline_span: Any | None = None @@ -104,6 +117,7 @@ def __init__( self._llm_span: Any | None = None self._llm_parent: str | None = None self._llm_text_parts: list[str] = [] + self._ttfb = TTFBRouter(self._log_unmatched_ttfb) self._llm_metrics: dict[str, Any] = {} self._llm_metadata: dict[str, Any] = {} self._llm_tool_calls: list[dict[str, Any]] = [] @@ -113,14 +127,50 @@ def __init__( self._tts_default_span: Any | None = None self._tts_audio: dict[str, bytearray] = {} self._tts_audio_metadata: dict[str, dict[str, Any]] = {} + self._turn_metric_state: dict[str, Any] = {} + self._unassociated_turn_metric_state: dict[str, Any] = {} self._user_audio_span: Any | None = None self._user_audio: bytearray | None = None self._user_audio_metadata: dict[str, Any] = {} + def _bind_pipeline(self, pipeline: Any) -> None: + if self._voice is not None or not self.trace_turns: + return + import importlib.metadata + + if importlib.metadata.version("pipecat-ai") != "1.12.0": + return + from .voice.discovery import discover + from .voice.instrumentation import NativeObserver + + configuration = discover(pipeline) + if configuration is None: + return + root = start_span(name="pipecat.pipeline", type="task", set_current=False, parent=self._parent) + voice = NativeObserver( + root, + root=root, + capture_user_audio=self.capture_user_audio_attachments, + capture_agent_audio=self.capture_agent_audio_attachments, + audio_format=self.audio_format, + recording_options=self.recording_options, + ) + try: + voice.bind(**configuration) + except Exception: # noqa: BLE001 - unsupported hooks preserve the existing observer + voice.hooks.close() + root.end() + logging.getLogger(__name__).warning("Pipecat voice hooks unavailable; using frame tracing") + return + self._voice = voice + async def on_pipeline_started(self) -> None: - self._ensure_pipeline_span() + if self._voice is None: + self._ensure_pipeline_span() async def on_process_frame(self, data: Any) -> None: + if self._voice is not None: + return frame = getattr(data, "frame", None) processor = getattr(data, "processor", None) is_terminal_at_sink = type(frame).__name__ in _TERMINAL_FRAME_TYPES and _is_pipeline_sink_processor(processor) @@ -129,9 +179,20 @@ async def on_process_frame(self, data: Any) -> None: await self._handle_frame(frame, processor=processor) async def on_push_frame(self, data: Any) -> None: + if self._voice is not None: + try: + await self._voice.on_push_frame(data) + except Exception: # noqa: BLE001 - tracing cannot stop frame delivery + logging.getLogger(__name__).warning("Pipecat frame observation failed") + return await self._handle_frame(getattr(data, "frame", None), processor=getattr(data, "source", None)) async def cleanup(self) -> None: + if self._voice is not None: + try: + await self._voice.finish() + except Exception: # noqa: BLE001 - export failure cannot stop pipeline cleanup + logging.getLogger(__name__).warning("Pipecat voice finalization failed") self._close_all_open_spans() await super().cleanup() @@ -175,7 +236,7 @@ async def _handle_frame(self, frame: Any, *, processor: Any = None) -> None: elif frame_type == "TranscriptionFrame": self._log_transcription(frame) elif frame_type == "TTSStartedFrame": - self._start_tts_span(frame) + self._start_tts_span(frame, processor) elif frame_type == "TTSTextFrame": self._append_tts_text(frame) elif frame_type == "TTSAudioRawFrame": @@ -231,6 +292,7 @@ def _capture_llm_context(self, frame: Any, processor: Any) -> None: def _start_llm_span(self, processor: Any) -> None: self._ensure_pipeline_span(processor=processor) if self._llm_span is not None: + self._ttfb.start("llm", None, self._llm_span.log) return metadata = {**self._latest_llm_metadata, **_metadata_from_processor(processor)} self._llm_metadata = _filter_llm_metadata(metadata) @@ -246,6 +308,7 @@ def _start_llm_span(self, processor: Any) -> None: set_current=False, ) self._llm_parent = self._llm_span.export() + self._ttfb.start("llm", processor, self._llm_span.log) def _capture_llm_tool_calls(self, frame: Any) -> None: calls = getattr(frame, "function_calls", None) or [] @@ -280,6 +343,7 @@ def _end_llm_span(self) -> None: if metadata: event["metadata"] = metadata self._llm_span.log(**event) + self._ttfb.end("llm") self._llm_span.end() self._llm_span = None self._llm_parent = None @@ -342,7 +406,7 @@ def _log_transcription(self, frame: Any) -> None: ) span.end() - def _start_tts_span(self, frame: Any) -> None: + def _start_tts_span(self, frame: Any, processor: Any = None) -> None: self._ensure_pipeline_span() context_id = getattr(frame, "context_id", None) or "__default__" span = start_span( @@ -357,6 +421,7 @@ def _start_tts_span(self, frame: Any) -> None: } span.log(metadata={k: v for k, v in metadata.items() if v is not None}) self._tts_spans[context_id] = span + self._ttfb.start(("tts", context_id), processor, span.log) if context_id == "__default__": self._tts_default_span = span @@ -389,6 +454,7 @@ def _log_tts_audio(self, frame: Any) -> None: def _end_tts_span(self, frame: Any) -> None: context_id = getattr(frame, "context_id", None) or "__default__" + self._ttfb.end(("tts", context_id)) span = self._tts_spans.pop(context_id, None) if span is not None: output = self._pop_tts_audio_output(context_id) @@ -410,6 +476,7 @@ def _start_user_audio_span(self, frame: Any | None = None) -> None: self._ensure_pipeline_span() if self._user_audio_span is not None: return + self._turn_metric_state = {} self._user_audio = bytearray() self._user_audio_metadata = _audio_frame_metadata(frame) if frame is not None else {} self._user_audio_span = start_span( @@ -470,22 +537,37 @@ def _discard_tts_audio(self, context_id: str) -> None: self._tts_audio.pop(context_id, None) self._tts_audio_metadata.pop(context_id, None) + def _log_unmatched_ttfb(self, **event): + self._ensure_pipeline_span() + self._pipeline_span.log(**event) + def _capture_metrics(self, frame: Any, processor: Any) -> None: processor_name = _processor_name(processor) for metric in getattr(frame, "data", []) or []: metric_type = type(metric).__name__ - if metric_type == "LLMUsageMetricsData": + if metric_type in TURN_METRIC_TYPES: + self._ensure_pipeline_span() + owner = self._user_audio_span or self._pipeline_span + state = ( + self._turn_metric_state + if self._user_audio_span is not None + else self._unassociated_turn_metric_state + ) + log_turn_metric(owner, state, metric) + elif metric_type == "LLMUsageMetricsData": metric_processor = getattr(metric, "processor", None) if processor_name is not None and metric_processor is not None and metric_processor != processor_name: continue self._llm_metrics.update(_llm_usage_metrics(getattr(metric, "value", None))) self._llm_metadata.update(_metadata_from_metric(metric)) - elif metric_type == "TTFBMetricsData" and self._llm_span is not None: + self._llm_metadata.update(_metadata_from_processor(processor)) + elif metric_type == "TTFBMetricsData": + owner = self._ttfb.capture(metric, processor) value = getattr(metric, "value", None) - if isinstance(value, (int, float)) and not isinstance(value, bool): + if owner == "llm" and isinstance(value, (int, float)) and not isinstance(value, bool): self._llm_metrics["time_to_first_token"] = value - self._llm_metadata.update(_metadata_from_metric(metric)) - self._llm_metadata.update(_metadata_from_processor(processor)) + self._llm_metadata.update(_metadata_from_metric(metric)) + self._llm_metadata.update(_metadata_from_processor(processor)) self._llm_metadata = _filter_llm_metadata(self._llm_metadata) def _log_error_frame(self, frame: Any) -> None: @@ -493,6 +575,7 @@ def _log_error_frame(self, frame: Any) -> None: error = getattr(frame, "exception", None) or getattr(frame, "error", None) if self._llm_span is not None: self._llm_span.log(error=error) + self._ttfb.end("llm") self._llm_span.end() self._llm_span = None if self._pipeline_span is not None: @@ -529,6 +612,7 @@ def _close_child_spans(self) -> None: self._tts_default_span = None def _close_all_open_spans(self) -> None: + self._ttfb.clear() self._close_child_spans() if self._pipeline_span is not None: self._pipeline_span.end() @@ -661,55 +745,6 @@ def _tools_schema_to_openai_tools(tools: Any) -> list[dict[str, Any]]: return ret -def _metadata_from_processor(processor: Any) -> dict[str, Any]: - metadata: dict[str, Any] = {} - settings = getattr(processor, "_settings", None) - model = getattr(settings, "model", None) or getattr(processor, "model", None) - if isinstance(model, str): - metadata["model"] = model - provider = _provider_from_processor(processor) - if provider: - metadata["provider"] = provider - return metadata - - -def _metadata_from_metric(metric: Any) -> dict[str, Any]: - metadata: dict[str, Any] = {} - model = getattr(metric, "model", None) - if isinstance(model, str): - metadata["model"] = model - processor = getattr(metric, "processor", None) - if isinstance(processor, str): - provider = _provider_from_name(processor) - if provider: - metadata["provider"] = provider - return metadata - - -def _provider_from_processor(processor: Any) -> str | None: - module = getattr(type(processor), "__module__", "") - return _provider_from_name(module) - - -def _provider_from_name(name: str) -> str | None: - lowered = name.lower() - providers = { - "openai": "openai", - "anthropic": "anthropic", - "google": "google", - "gemini": "google", - "mistral": "mistral", - "cohere": "cohere", - "bedrock": "bedrock", - "aws": "bedrock", - "openrouter": "openrouter", - } - for needle, provider in providers.items(): - if needle in lowered: - return provider - return None - - def _processor_name(processor: Any) -> str | None: name = getattr(processor, "name", None) if isinstance(name, str): @@ -728,27 +763,6 @@ def _filter_llm_metadata(metadata: dict[str, Any]) -> dict[str, Any]: return {key: value for key, value in metadata.items() if key in _ALLOWED_LLM_METADATA_FIELDS and value is not None} -def _llm_usage_metrics(usage: Any) -> dict[str, Any]: - metrics: dict[str, Any] = {} - prompt_tokens = getattr(usage, "prompt_tokens", None) - completion_tokens = getattr(usage, "completion_tokens", None) - total_tokens = getattr(usage, "total_tokens", None) - cache_read = getattr(usage, "cache_read_input_tokens", None) - cache_creation = getattr(usage, "cache_creation_input_tokens", None) - reasoning_tokens = getattr(usage, "reasoning_tokens", None) - for key, value in ( - ("prompt_tokens", prompt_tokens), - ("completion_tokens", completion_tokens), - ("tokens", total_tokens), - ("cache_read_input_tokens", cache_read), - ("cache_creation_input_tokens", cache_creation), - ("reasoning_tokens", reasoning_tokens), - ): - if isinstance(value, (int, float)) and not isinstance(value, bool): - metrics[key] = value - return metrics - - def _tool_call_from_pipecat_call(call: Any) -> dict[str, Any] | None: name = getattr(call, "function_name", None) tool_call_id = getattr(call, "tool_call_id", None) diff --git a/py/src/braintrust/integrations/pipecat/ttfb.py b/py/src/braintrust/integrations/pipecat/ttfb.py new file mode 100644 index 000000000..a2a8bc18d --- /dev/null +++ b/py/src/braintrust/integrations/pipecat/ttfb.py @@ -0,0 +1,109 @@ +"""Route native measurements by request identity or an unambiguous live operation.""" + +from collections import OrderedDict, deque + + +class TTFBRouter: + def __init__(self, fallback): + self.fallback = fallback + self.active = {} + self.unmatched = {} + self.pending = deque() + self.completed = OrderedDict() + + def start(self, key, processor, log): + self.completed.pop(key, None) + # Repeated starts without a matching end cannot identify one operation. + if key in self.active: + processor = None + self.active[key] = (processor, log, {}) + pending, self.pending = self.pending, deque() + for metric, source, request in pending: + if request == key: + self.capture_request(metric, source, request) + else: + self.pending.append((metric, source, request)) + + def end(self, key): + operation = self.active.pop(key, None) + if operation and isinstance(key, tuple) and key[0] == "tts": + self.completed[key] = operation + if len(self.completed) > 64: + self.completed.popitem(last=False) + + def clear(self): + while self.pending: + metric, _, _ = self.pending.popleft() + self._capture_request(metric, (None, self.fallback, self.unmatched)) + self.active.clear() + self.completed.clear() + + def capture_request(self, metric, source, key): + """Use emission-time identity, including before start or after stop.""" + operation = self.active.get(key) or self.completed.get(key) + if operation: + processor, log, state = operation + owner = ( + (key, log, state) + if processor is source and metric.processor == source.name + else (None, self.fallback, self.unmatched) + ) + self._capture_request(metric, owner) + else: + if len(self.pending) >= 256: + oldest, _, _ = self.pending.popleft() + self._capture_request(oldest, (None, self.fallback, self.unmatched)) + self.pending.append((metric, source, key)) + + def _capture_request(self, metric, owner): + if type(metric).__name__ == "TTFBMetricsData": + self._capture_ttfb(metric, owner) + else: + self._capture_measurement(metric, owner) + + def owner(self, metric, source, *, operation=None): + name = getattr(metric, "processor", None) + matches = [ + (key, log, state) + for key, (processor, log, state) in self.active.items() + if processor is source + and name is not None + and getattr(processor, "name", None) == name + and (operation is None or key == operation) + ] + if len(matches) == 1: + return matches[0] + return None, self.fallback, self.unmatched + + def capture(self, metric, source): + return self._capture_ttfb(metric, self.owner(metric, source)) + + def _capture_ttfb(self, metric, owner): + key, log, state = owner + values = state.setdefault("values", []) + if len(values) >= 32: + state["omitted"] = state.get("omitted", 0) + 1 + log(metadata={"braintrust.ttfb.omitted": state["omitted"]}) + else: + payload = ( + metric.model_dump(mode="json") + if hasattr(metric, "model_dump") + else {"processor": getattr(metric, "processor", None), "value": metric.value} + ) + values.append(payload) + log(metadata={"contrib.pipecat.ttfb": list(values)}) + return key + + def capture_measurement(self, metric, source): + """Retain native measurements without a frame/routing envelope.""" + self._capture_measurement(metric, self.owner(metric, source)) + + def _capture_measurement(self, metric, owner): + _, log, state = owner + values = state.setdefault("measurements", []) + if len(values) >= 32: + state["measurements_omitted"] = state.get("measurements_omitted", 0) + 1 + log(metadata={"braintrust.measurements.omitted": state["measurements_omitted"]}) + return + values.append({"type": type(metric).__name__, **metric.model_dump(mode="json")}) + log(metadata={"contrib.pipecat.measurements": list(values)}) diff --git a/py/src/braintrust/integrations/pipecat/turn_metrics.py b/py/src/braintrust/integrations/pipecat/turn_metrics.py new file mode 100644 index 000000000..fbb141c08 --- /dev/null +++ b/py/src/braintrust/integrations/pipecat/turn_metrics.py @@ -0,0 +1,23 @@ +"""Native turn analyzer measurements, without inferred EOU timing.""" + +TURN_METRIC_TYPES = {"TurnMetricsData", "SmartTurnMetricsData"} +MAX_TURN_METRICS = 32 + + +def log_turn_metric(span, state, metric): + """Keep bounded predictions on their owner, preserving native units and identity.""" + predictions = state.setdefault("turn_metrics", []) + if len(predictions) >= MAX_TURN_METRICS: + state["turn_metrics_omitted"] = state.get("turn_metrics_omitted", 0) + 1 + span.log(metadata={"braintrust.turn_metrics.omitted": state["turn_metrics_omitted"]}) + return + predictions.append( + { + "type": type(metric).__name__, + **{ + name: getattr(metric, name) + for name in ("processor", "is_complete", "probability", "e2e_processing_time_ms") + }, + } + ) + span.log(metadata={"contrib.pipecat.turn_metrics": list(predictions)}) diff --git a/py/src/braintrust/integrations/pipecat/voice/__init__.py b/py/src/braintrust/integrations/pipecat/voice/__init__.py new file mode 100644 index 000000000..6f1173e50 --- /dev/null +++ b/py/src/braintrust/integrations/pipecat/voice/__init__.py @@ -0,0 +1,6 @@ +"""Internal native voice tracing installed by the Pipecat integration.""" + +from .instrumentation import NativeObserver as VoiceObserver + + +__all__ = ["VoiceObserver"] diff --git a/py/src/braintrust/integrations/pipecat/voice/alignment.py b/py/src/braintrust/integrations/pipecat/voice/alignment.py new file mode 100644 index 000000000..c146fa08c --- /dev/null +++ b/py/src/braintrust/integrations/pipecat/voice/alignment.py @@ -0,0 +1,161 @@ +"""Pipecat output queue provenance hooks.""" + +from collections import deque +from contextvars import ContextVar + +from braintrust._audio import pcm_bytes_to_ms, samples_to_ms + + +def instrument_output(output, alignment, frame_context=None, hooks=None): + """Track equal-rate mono PCM through native chunking without copying audio. + + Unsupported resampling/mixing deliberately has no asserted association. + """ + from .hooks import Hooks + + hooks = hooks or Hooks() + original_start = hooks.original(output, "start") + original_write = hooks.original(output, "write_audio_frame") + token = object() + source_offsets = {} + active = ContextVar("pipecat_output_context", default=None) + + async def start(frame): + result = await original_start(frame) + for sender in output._media_senders.values(): + install_sender(sender) + return result + + def install_sender(sender): + runs, pending = deque(), deque() + handle_audio = hooks.original(sender, "handle_audio_frame") + handle_stop = hooks.original(sender, "handle_tts_stopped") + buffer_audio = hooks.original(sender, "_buffer_audio") + take_chunk = hooks.original(sender, "_take_audio_chunk") + clear_buffer = hooks.original(sender, "_clear_audio_buffer") + create_audio_task = hooks.original(sender, "_create_audio_task") + + async def handle(frame): + context = frame_context(frame) if frame_context else getattr(frame, "context_id", None) + if frame.sample_rate != sender._sample_rate or frame.num_channels != 1 or sender._mixer: + context = None + start = source_offsets.get(context, 0) + if context is not None: + source_offsets[context] = start + len(frame.audio) + owners = alignment.output_owners(context) + current = active.set((owners, start)) + try: + return await handle_audio(frame) + finally: + active.reset(current) + + async def stop(frame): + context = frame_context(frame) if frame_context else frame.context_id + current = active.set(None) # Native stop padding is not generated clip audio. + try: + return await handle_stop(frame) + finally: + active.reset(current) + source_offsets.pop(context, None) + alignment.end_output(context) + + def buffer(audio, *, uninterruptible): + result = buffer_audio(audio, uninterruptible=uninterruptible) + if audio: + owners, start = active.get() or ([], 0) + runs.append([len(audio), owners, start]) + return result + + def take(): + audio, uninterruptible = take_chunk() + left, offset, parts = len(audio), 0, [] + while left and runs: + count, owners, start = runs[0] + amount = min(count, left) + parts.append((offset, amount, owners, start)) + runs[0][2] += amount + runs[0][0] -= amount + if not runs[0][0]: + runs.popleft() + left -= amount + offset += amount + pending.append(parts) + return audio, uninterruptible + + bound_queue = [None] + + def bind_queue(): + if bound_queue[0] is sender._audio_queue: + return + if bound_queue[0] is not None: + hooks.remove(bound_queue[0], "put") + bound_queue[0] = sender._audio_queue + queue_put = hooks.original(sender._audio_queue, "put") + + async def put(frame): + if hasattr(frame, "audio") and pending: + parts = pending.popleft() + frame._braintrust_output_parts = (token, parts) + return await queue_put(frame) + + hooks.set(sender._audio_queue, "put", put) + + def create(): + result = create_audio_task() + bind_queue() + return result + + def clear(): + runs.clear() + pending.clear() + return clear_buffer() + + hooks.set(sender, "handle_audio_frame", handle) + hooks.set(sender, "handle_tts_stopped", stop) + hooks.set(sender, "_buffer_audio", buffer) + hooks.set(sender, "_take_audio_chunk", take) + hooks.set(sender, "_clear_audio_buffer", clear) + hooks.set(sender, "_create_audio_task", create) + bind_queue() + + async def write(frame): + import time + + observed = time.monotonic_ns() + identity = getattr(frame, "_braintrust_output_parts", None) + parts = identity[1] if identity and identity[0] is token else [] + if identity and identity[0] is token: + del frame._braintrust_output_parts + result = await original_write(frame) + if result: + try: + interval = alignment.recording.capture( + 1, frame.audio, frame.sample_rate, frame.num_channels, observed_ns=observed + ) + except Exception: # noqa: BLE001 - instrumentation must preserve transport success + alignment.recording.omit("capture_error") + interval = None + if interval: + for offset, size, owners, start in parts: + # This hook supports only the demo's mono, 24 kHz output. + if frame.sample_rate == 24000 and frame.num_channels == 1: + ranges = [ + [ + interval["start"] + offset // 2, + interval["start"] + (offset + size) // 2, + ] + ] + if owners: + alignment.add_clip_range( + owners[0], + pcm_bytes_to_ms(start, frame.sample_rate, frame.num_channels), + pcm_bytes_to_ms(start + size, frame.sample_rate, frame.num_channels), + samples_to_ms(ranges[0][0]), + samples_to_ms(ranges[0][1]), + ) + for owner in owners: + alignment.add(owner, ranges, 1) + return result + + hooks.set(output, "start", start) + hooks.set(output, "write_audio_frame", write) diff --git a/py/src/braintrust/integrations/pipecat/voice/discovery.py b/py/src/braintrust/integrations/pipecat/voice/discovery.py new file mode 100644 index 000000000..2b8f5db70 --- /dev/null +++ b/py/src/braintrust/integrations/pipecat/voice/discovery.py @@ -0,0 +1,31 @@ +"""Discover a single native voice path without integration-specific setup.""" + +from types import SimpleNamespace + + +def _is(processor, name): + return any(cls.__name__ == name and cls.__module__.startswith("pipecat.") for cls in type(processor).__mro__) + + +def discover(pipeline): + # Parallel/nested paths need explicit ownership; don't guess between them. + processors = list(getattr(pipeline, "processors", ())) + if any(getattr(processor, "processors", ()) for processor in processors): + return None + names = ("BaseInputTransport", "BaseOutputTransport", "LLMUserAggregator", "LLMAssistantAggregator") + matches = [[processor for processor in processors if _is(processor, name)] for name in names] + if any(len(match) != 1 for match in matches): + return None + source, destination, user, assistant = (match[0] for match in matches) + stts = [processor for processor in processors if _is(processor, "SegmentedSTTService")] + realtime = [processor for processor in processors if _is(processor, "OpenAIRealtimeLLMService")] + if len(stts) + len(realtime) != 1: + return None + return dict( + transport=SimpleNamespace(input=lambda: source, output=lambda: destination), + user_aggregator=user, + assistant_aggregator=assistant, + stt=stts[0] if stts else None, + realtime_service=realtime[0] if realtime else None, + tts_services=[processor for processor in processors if _is(processor, "TTSService")], + ) diff --git a/py/src/braintrust/integrations/pipecat/voice/fixtures/README.md b/py/src/braintrust/integrations/pipecat/voice/fixtures/README.md new file mode 100644 index 000000000..8e2b1f43b --- /dev/null +++ b/py/src/braintrust/integrations/pipecat/voice/fixtures/README.md @@ -0,0 +1,20 @@ +# Voice cassette maintenance + +Run from `py/`. Replay needs no credentials: + +```sh +mise exec -- uv run nox -s 'test_pipecat(latest)' -- --vcr-record=none +mise exec -- uv run nox -s 'test_pipecat(1.3.0)' -- --vcr-record=none +``` + +To re-record the voice conversations, set `OPENAI_API_KEY` in your shell, then run: + +```sh +mise exec -- uv run nox -s 'test_pipecat(latest)' -- --vcr-record=all -k voice_conversation +``` + +- Re-recording makes short, billable OpenAI calls. No other API keys are needed. Existing fixtures are replaced only after the conversation passes its assertions. +- HTTP uses VCR; Realtime uses `_test_websocket.py` to record/replay WebSocket messages. `_test_audio_cassettes.py` stores audio separately and restores exact bytes for replay. +- Commit YAML/JSON manifests **and** companion `.audio/` directories under `pipecat/cassettes/latest/` together. Re-recording regenerates these WAVs; do not edit or compress them manually. WebSocket offsets refer to PCM bytes, excluding the WAV header. +- `order.wav` is the cascade caller input: 16 kHz mono PCM16, “Where is my order number one zero four two?”, generated with macOS's Samantha voice. Realtime uses `order-24khz.wav`, a pre-resampled copy of that request. Keeping its input bytes fixed avoids platform-dependent resampler rounding during strict WebSocket replay. Changing either input requires re-recording its conversation. +- After re-recording, review the fixture diff and run replay with `--vcr-record=none` before committing. Missing or truncated audio files fail replay. diff --git a/py/src/braintrust/integrations/pipecat/voice/fixtures/order-24khz.wav b/py/src/braintrust/integrations/pipecat/voice/fixtures/order-24khz.wav new file mode 100644 index 000000000..6cc6ff393 Binary files /dev/null and b/py/src/braintrust/integrations/pipecat/voice/fixtures/order-24khz.wav differ diff --git a/py/src/braintrust/integrations/pipecat/voice/fixtures/order.wav b/py/src/braintrust/integrations/pipecat/voice/fixtures/order.wav new file mode 100644 index 000000000..d7230f1ec Binary files /dev/null and b/py/src/braintrust/integrations/pipecat/voice/fixtures/order.wav differ diff --git a/py/src/braintrust/integrations/pipecat/voice/hooks.py b/py/src/braintrust/integrations/pipecat/voice/hooks.py new file mode 100644 index 000000000..aac020ba7 --- /dev/null +++ b/py/src/braintrust/integrations/pipecat/voice/hooks.py @@ -0,0 +1,133 @@ +"""Per-pipeline reversible method and callback installation.""" + +import inspect +import logging +from contextvars import ContextVar +from functools import wraps + + +class Hooks: + def __init__(self): + self.methods = [] + self.handlers = [] + self.invocations = {} + + def original(self, target, name): + original = getattr(target, name) + if inspect.isasyncgenfunction(original): + return original + state = self.invocations.setdefault((id(target), name), ContextVar(name, default=None)) + + def invoke(*args, **kwargs): + current = state.get() + if current is not None: + current["called"] = True + try: + result = original(*args, **kwargs) + except Exception as error: + if current is not None: + current["error"] = error + raise + if current is not None: + current["result"] = result + return result + + @wraps(original) + async def invoke_async(*args, **kwargs): + current = state.get() + if current is not None: + current["called"] = True + try: + result = await original(*args, **kwargs) + except Exception as error: + if current is not None: + current["error"] = error + raise + if current is not None: + current["result"] = result + return result + + return invoke_async if inspect.iscoroutinefunction(original) else invoke + + def guard(self, target, name, replacement): + state = self.invocations.get((id(target), name)) + if state is None or inspect.isasyncgenfunction(replacement): + return replacement + original = getattr(target, name) + warned = False + + def failure(current): + nonlocal warned + if "error" in current: + raise current["error"] + if not warned: + logging.getLogger(__name__).warning("Pipecat observation failed in %s", name) + warned = True + + @wraps(replacement) + async def guarded_async(*args, **kwargs): + current = {} + token = state.set(current) + try: + return await replacement(*args, **kwargs) + except Exception: + failure(current) + if current.get("called"): + return current.get("result") + return await original(*args, **kwargs) + finally: + state.reset(token) + + @wraps(replacement) + def guarded(*args, **kwargs): + current = {} + token = state.set(current) + try: + return replacement(*args, **kwargs) + except Exception: + failure(current) + if current.get("called"): + return current.get("result") + return original(*args, **kwargs) + finally: + state.reset(token) + + return guarded_async if inspect.iscoroutinefunction(replacement) else guarded + + def set(self, target, name, replacement): + original = target.__dict__.get(name) + present = name in target.__dict__ + replacement = self.guard(target, name, replacement) + setattr(target, name, replacement) + self.methods.append((target, name, replacement, original, present)) + + def remove(self, target, name): + """Restore a replaced object's method without retaining it until shutdown.""" + for index in range(len(self.methods) - 1, -1, -1): + entry = self.methods[index] + if entry[0] is target and entry[1] == name: + _, _, replacement, original, present = self.methods.pop(index) + if target.__dict__.get(name) is replacement: + if present: + setattr(target, name, original) + else: + delattr(target, name) + self.invocations.pop((id(target), name), None) + return + + def event(self, target, name, handler): + target.add_event_handler(name, handler) + self.handlers.append((target, name, handler)) + + def close(self): + for target, name, handler in reversed(self.handlers): + target.remove_event_handler(name, handler) + self.handlers.clear() + for target, name, replacement, original, present in reversed(self.methods): + if target.__dict__.get(name) is replacement: + if present: + setattr(target, name, original) + else: + delattr(target, name) + self.methods.clear() + self.invocations.clear() diff --git a/py/src/braintrust/integrations/pipecat/voice/instrumentation.py b/py/src/braintrust/integrations/pipecat/voice/instrumentation.py new file mode 100644 index 000000000..e5bbcfd72 --- /dev/null +++ b/py/src/braintrust/integrations/pipecat/voice/instrumentation.py @@ -0,0 +1,705 @@ +"""Experimental native Pipecat observer and opt-in recording support.""" + +import asyncio +import dataclasses +import io +import json +import time +import wave +from collections import Counter, deque + +from braintrust._audio import AlignmentPublisher, SegmentedRecording, segment_descriptor +from pipecat.observers.base_observer import BaseObserver # pylint: disable=import-error + +from ..llm_metrics import _llm_usage_metrics, _metadata_from_metric, _metadata_from_processor +from ..ttfb import TTFBRouter +from ..turn_metrics import TURN_METRIC_TYPES, log_turn_metric +from .synthesis import Synthesis, SynthesisRecordings +from .turns import Turns + + +# High-frequency payloads are handled by their owning operation or recorder. +# Only selected diagnostic/lifecycle facts become events. +EVENT_TYPES = { + "StartFrame", + "EndFrame", + "CancelFrame", + "ClientConnectedFrame", + "OutputTransportReadyFrame", + "BotStartedSpeakingFrame", + "BotStoppedSpeakingFrame", + "UserStartedSpeakingFrame", + "UserStoppedSpeakingFrame", + "VADUserStartedSpeakingFrame", + "VADUserStoppedSpeakingFrame", + "InterruptionFrame", + "UserTurnInferenceCompletedFrame", + "EagerEndOfTurnCancelFrame", + "UserMuteStartedFrame", + "UserMuteStoppedFrame", + "STTMuteFrame", + "ErrorFrame", + "FatalErrorFrame", + "MetricsFrame", + "SpeechControlParamsFrame", + "STTMetadataFrame", + "LLMServiceMetadataFrame", + "LLMUpdateSettingsFrame", + "TTSUpdateSettingsFrame", + "STTUpdateSettingsFrame", + "VADParamsUpdateFrame", +} +OPERATION_TYPES = { + "LLMContextFrame", + "LLMFullResponseStartFrame", + "LLMFullResponseEndFrame", + "LLMTextFrame", + "TranscriptionFrame", + "FunctionCallsStartedFrame", + "FunctionCallInProgressFrame", + "FunctionCallResultFrame", + "FunctionCallCancelFrame", + "TTSStartedFrame", + "TTSStoppedFrame", + "TTSTextFrame", +} + + +CONFIG_TYPES = { + "SpeechControlParamsFrame", + "STTMetadataFrame", + "LLMServiceMetadataFrame", + "LLMUpdateSettingsFrame", + "TTSUpdateSettingsFrame", + "STTUpdateSettingsFrame", + "VADParamsUpdateFrame", +} + + +def compact_frame(fields): + """Keep event content; omit transport bookkeeping and empty fields.""" + return { + key: value + for key, value in fields.items() + if key + not in {"id", "name", "broadcast_sibling_id", "interruptible", "transport_destination", "transport_source"} + and value is not None + and value != {} + and value != [] + } + + +def native_value(value): + if isinstance(value, bytes): + return None + if dataclasses.is_dataclass(value): + return { + field.name: native_value(getattr(value, field.name)) + for field in dataclasses.fields(value) + if field.name not in {"audio", "image"} and not field.name.startswith("_") + } + if hasattr(value, "model_dump"): + return native_value(value.model_dump(mode="json")) + if isinstance(value, dict): + return {str(key): native_value(item) for key, item in value.items() if key not in {"audio", "image"}} + if isinstance(value, (tuple, list)): + return [native_value(item) for item in value] + if value is None or isinstance(value, (str, int, float, bool)): + return value + return str(value) + + +def encode_wav(chunks, sample_rate, channels): + output = io.BytesIO() + with wave.Wave_write(output) as wav: + wav.setnchannels(channels) + wav.setsampwidth(2) + wav.setframerate(sample_rate) + for chunk in chunks: + wav.writeframesraw(chunk) + return output.getvalue() + + +class NativeObserver(BaseObserver): + def __init__( + self, + logger, + *, + retain_audio=False, + max_audio_bytes=8 * 1024 * 1024, + audio_format="ogg", + capture_user_audio=None, + capture_agent_audio=None, + root=None, + recording_options=None, + ): + if audio_format not in {"ogg", "wav"}: + raise ValueError("AUDIO_FORMAT must be ogg or wav") + super().__init__(observe_every_push=False) + from .hooks import Hooks + + self.hooks = Hooks() + self.capture_user_audio = retain_audio if capture_user_audio is None else capture_user_audio + self.capture_agent_audio = retain_audio if capture_agent_audio is None else capture_agent_audio + retain_audio = self.capture_user_audio or self.capture_agent_audio + self.logger = logger + self.audio_format = audio_format + self.call_recording = SegmentedRecording( + enabled=retain_audio, + options=recording_options, + audio_format=audio_format, + on_segment=self._publish_call_segment, + on_pending=self._pending_call_segment, + ) + self.realtime = None + self.input_processor = None + self.capture_transport = False + self.root = ( + root if root is not None else logger.start_span(name="pipecat.pipeline", type="task", set_current=False) + ) + self.ttfb = TTFBRouter(self.root.log) + self.tts_requests = None + self.turns = Turns(self.root, self.hooks) + self.turns.on_completed = self._turn_completed + self.alignment = AlignmentPublisher(self.root, self.call_recording) + self.synthesis = SynthesisRecordings( + logger, + self.alignment, + enabled=self.capture_agent_audio, + max_bytes=max_audio_bytes, + audio_format=audio_format, + ) + self.user_capture = None + self.llm_turn = None + self.user_aggregator = None + self.assistant_aggregator = None + self.context_reply_to = None + self.context_tool_results = [] + self.result_origins = {} + self.event_groups = {} + self.configuration = {} + self.filtered_events = Counter() + self.duplicate_events = 0 + self.clock_anchored = False + self.llm = None + self.llm_tool_calls = [] + self.llm_text = [] + self.user = None + self.tools = {} + self.requests = {} + self.tts = {} + self.events = [] + self.omitted_events = 0 + self.event_bytes = 0 + self.retain_audio = retain_audio + self.max_audio_bytes = max_audio_bytes + self.finished = False + self._finish_task = None + self.tool_names = [] + self.started_tools = set() + self.last_context = None + self.seen = set() + self.seen_order = deque() + + def bind( + self, *, transport, user_aggregator, assistant_aggregator, stt=None, realtime_service=None, tts_services=() + ): + """Bind one supported pipeline before it starts; does not alter routing.""" + import importlib.metadata + + if importlib.metadata.version("pipecat-ai") != "1.12.0": + raise ValueError("Experimental voice hooks currently support Pipecat 1.12.0") + if getattr(self, "_bound", False): + raise ValueError("Voice observer is already bound to a pipeline") + if (stt is None) == (realtime_service is None): + raise ValueError("Bind either segmented STT or an OpenAI realtime service") + self.turns.install(user_aggregator, assistant_aggregator) + from .tts_metrics import TTSRequests + + self.tts_requests = TTSRequests(self.hooks, tts_services) + self.user_aggregator, self.assistant_aggregator = user_aggregator, assistant_aggregator + self.capture_transport = True + if stt is not None: + from .user_capture import UserCapture + + self.input_processor = transport.input() + self.user_capture = UserCapture(self, stt, user_aggregator, native_value) + else: + from .realtime import RealtimeCapture + + self.realtime = RealtimeCapture(self, realtime_service, user_aggregator) + if self.capture_agent_audio: + from .alignment import instrument_output + + instrument_output(transport.output(), self.alignment, self.frame_context, self.hooks) + self._bound = True + return self + + def frame_context(self, frame): + identity = getattr(frame, "_braintrust_realtime_context", None) + if identity and self.realtime and identity[0] is self.realtime.frame_token: + return identity[1] + return getattr(frame, "context_id", None) + + async def on_push_frame(self, data): + if not data.first_push or self.finished: + return + frame = data.frame + sibling = getattr(frame, "broadcast_sibling_id", None) + if frame.id in self.seen or (sibling is not None and sibling in self.seen): + self.duplicate_events += 1 + return + if len(self.seen_order) >= 4096: + self.seen.discard(self.seen_order.popleft()) + self.seen.add(frame.id) + self.seen_order.append(frame.id) + if ( + self.capture_user_audio + and not self.user_capture + and data.source is self.input_processor + and type(frame).__name__ + in { + "InputAudioRawFrame", + "UserAudioRawFrame", + } + ): + try: + self.call_recording.capture(0, frame.audio, frame.sample_rate, frame.num_channels) + except Exception: # noqa: BLE001 - capture/export failures must not break the call + self.call_recording.omit("capture_error") + kind = type(frame).__name__ + # Audio data stays out of metadata; retain its format on the owning TTS span. + if kind == "TTSAudioRawFrame": + context = self.frame_context(frame) + state = self.tts.get(context) + if state: + self.synthesis.capture(state, frame) + return + if "AudioRawFrame" in kind: + return + if (self.user_capture or self.realtime) and kind in { + "VADUserStartedSpeakingFrame", + "VADUserStoppedSpeakingFrame", + "TranscriptionFrame", + }: + # These facts are attributed through the actual STT/aggregator path. + return + if kind not in EVENT_TYPES and kind not in OPERATION_TYPES: + self.filtered_events[kind] += 1 + return + fields = native_value(frame) if kind != "LLMContextFrame" else {} + event_owner = self.root + if kind == "MetricsFrame": + request = self.tts_requests.take(frame, data.source) if self.tts_requests else None + if request is not None: + for metric in frame.data: + self.ttfb.capture_request(metric, data.source, request) + return + # Pipeline startup broadcasts zero placeholders for every service. + if frame.data and all( + getattr(metric, "value", None) == 0 + and getattr(metric, "model", None) is None + and getattr(metric, "processor", None) != data.source.name + for metric in frame.data + ): + return + for metric in frame.data: + metric_type = type(metric).__name__ + if metric_type in TURN_METRIC_TYPES and data.source is self.user_aggregator and self.turns.user: + state = self.turns.user + log_turn_metric(state["span"], state, metric) # pylint: disable=unsubscriptable-object + elif metric_type == "TTFBMetricsData": + owner = self.ttfb.capture(metric, data.source) + if owner == "llm": + self.llm.log(metrics={"time_to_first_token": metric.value}) + elif ( + metric_type == "LLMUsageMetricsData" + and self.ttfb.owner(metric, data.source, operation="llm")[0] == "llm" + ): + self.llm.log( + metrics=_llm_usage_metrics(metric.value), + metadata={ + "contrib.pipecat.usage": native_value(metric), + **_metadata_from_processor(data.source), + **_metadata_from_metric(metric), + }, + ) + else: + self.ttfb.capture_measurement(metric, data.source) + return + if kind == "StartFrame": + self.root.log( + metadata={ + f"contrib.pipecat.{name}": getattr(frame, name) + for name in ( + "audio_in_sample_rate", + "audio_out_sample_rate", + "enable_metrics", + "enable_usage_metrics", + ) + } + ) + elif kind == "LLMContextFrame": + self.last_context = native_value(frame.context.get_messages()) + self.context_tool_results = [] + self.context_reply_to = None + if data.source is self.user_aggregator and self.turns.last_user: + self.context_reply_to = self.turns.last_user["span"].span_id + elif data.source is self.assistant_aggregator: + # Only newly observed tool results establish a continuation. + # Old tool messages remaining in context do not create links. + ids = { + m.get("tool_call_id") for m in self.last_context if isinstance(m, dict) and m.get("role") == "tool" + } + self.context_tool_results = sorted(ids & self.result_origins.keys()) + origins = {self.result_origins.pop(i) for i in self.context_tool_results} + if len(origins) == 1: + self.context_reply_to = origins.pop() + elif kind in {"UserStartedSpeakingFrame", "VADUserStartedSpeakingFrame"}: + if kind == "UserStartedSpeakingFrame": + self.turns.start("user", kind) + elif not self.user: + self.user = self.root.start_span( + name="pipecat.user_speaking", + type="task", + set_current=False, + internal={"instrumentation": "pipecat-auto"}, + metadata={"contrib.pipecat.start_frame": kind}, + ) + event_owner = self.user + elif kind in {"UserStoppedSpeakingFrame", "VADUserStoppedSpeakingFrame"}: + if kind == "VADUserStoppedSpeakingFrame" and self.user: + event_owner = self.user + self.user.log(metadata={"contrib.pipecat.end_frame": kind}) + self.user.end() + self.user = None + elif kind == "TranscriptionFrame": + # STT can trigger the next user turn rather than belong to an open + # one. The aggregator supplies authoritative turn text separately. + span = self.root.start_span( + name="pipecat.stt_transcription", + type="task", + set_current=False, + internal={"instrumentation": "pipecat-auto"}, + metadata={f"contrib.pipecat.{key}": value for key, value in fields.items()}, + ) + span.log(input={"text": frame.text}) + span.end() + elif kind == "LLMFullResponseStartFrame": + if self.realtime and self.llm is not None: + # Pipecat brackets both response creation and item arrival. + # One still-open response must not create another turn. + self.filtered_events["repeated_response_start"] += 1 + return + if self.realtime: + turn = self.realtime.user_turns.get(self.realtime.user_item) + self.context_reply_to = turn["span"].span_id if turn else None + self.context_tool_results = self.realtime.pending_tool_results[:] + self.realtime.pending_tool_results.clear() + self.result_origins.clear() + self.llm_tool_calls = [] + self.llm_text = [] + # A new model operation may arrive before the previous aggregator + # callback task finishes. Keep each native response lifecycle distinct. + self.turns.assistant = None + self.llm_turn = self.turns.start("assistant", kind, self.context_reply_to) + self.llm = self.llm_turn["span"].start_span( + name="llm_response", + type="llm", + input=self.last_context, + metadata={ + **self.turns.metadata(self.llm_turn), + "continuation.tool_call_ids": self.context_tool_results, + **_metadata_from_processor(data.source), + }, + set_current=False, + internal={"instrumentation": "pipecat-auto"}, + ) + self.ttfb.start("llm", data.source, self.llm.log) + elif kind == "LLMTextFrame" and self.llm: + self.llm_text.append(frame.text) + elif kind == "FunctionCallsStartedFrame": + self.llm_tool_calls = [ + { + "id": call.tool_call_id, + "type": "function", + "function": {"name": call.function_name, "arguments": json.dumps(native_value(call.arguments))}, + } + for call in frame.function_calls + ] + if self.llm_turn: + self.llm_turn["tool_calls"] = list(self.llm_tool_calls) + self.turns.log_message(self.llm_turn, "") + for call in frame.function_calls: + self.requests.setdefault( + call.tool_call_id, + ( + self.llm or self.root, + self.turns.metadata(self.llm_turn) + or ({"turn.reply_to": self.context_reply_to} if self.context_reply_to else {}), + ), + ) + if self.llm: + self.llm.log(metadata={"contrib.pipecat.function_calls": native_value(frame.function_calls)}) + elif kind == "FunctionCallInProgressFrame": + if frame.tool_call_id in self.started_tools: + return + self.started_tools.add(frame.tool_call_id) + owner, correlation = self.requests.get(frame.tool_call_id, (self.root, {})) + self.tools[frame.tool_call_id] = owner.start_span( + name=frame.function_name, + type="tool", + input=frame.arguments, + metadata={ + **correlation, + "contrib.pipecat.tool_call_id": frame.tool_call_id, + "contrib.pipecat.function_name": frame.function_name, + "contrib.pipecat.arguments": frame.arguments, + "contrib.pipecat.group_id": frame.group_id, + "contrib.pipecat.cancel_on_interruption": frame.cancel_on_interruption, + }, + set_current=False, + internal={"instrumentation": "pipecat-auto"}, + ) + if len(self.tool_names) < 512: + self.tool_names.append(frame.function_name) + elif kind in {"FunctionCallResultFrame", "FunctionCallCancelFrame"}: + tool = self.tools.pop(frame.tool_call_id, None) + if tool: + if kind == "FunctionCallResultFrame": + request = self.requests.pop(frame.tool_call_id, None) + if request: + self.result_origins[frame.tool_call_id] = request[1].get("turn.reply_to") + tool.log( + output=native_value(frame.result), + metadata={"contrib.pipecat.result": native_value(frame.result)}, + ) + if frame.error: + tool.log(error=frame.error) + else: + tool.log(metadata={"contrib.pipecat.cancelled": True}) + tool.end() + elif kind == "LLMFullResponseEndFrame" and self.llm: + text = "".join(self.llm_text) + message = {"role": "assistant", "content": text or None} + if self.llm_tool_calls: + message["tool_calls"] = list(self.llm_tool_calls) + self.llm.log(output=[message], metadata={"contrib.pipecat.text": text} if text else {}) + self.ttfb.end("llm") + self.llm.end() + self.llm = None + elif kind == "TTSStartedFrame": + context = self.frame_context(frame) + if context in self.tts: + return + turn = self.turns.assistant or self.turns.start("assistant", kind) + span = turn["span"].start_span( + name="pipecat.audio_output" if self.realtime else "tts", + type="task", + metadata={ + **self.turns.metadata(turn), + "contrib.pipecat.context_id": getattr(frame, "context_id", None), + **({"openai.response.id": context} if self.realtime and context else {}), + "contrib.pipecat.append_to_context": frame.append_to_context, + **_metadata_from_processor(data.source), + }, + set_current=False, + internal={"instrumentation": "pipecat-auto"}, + ) + self.ttfb.start(("tts", context), data.source, span.log) + self.tts[context] = Synthesis(span) + if self.capture_agent_audio: + self.alignment.begin_output(context, [span, turn["span"]]) + elif kind == "TTSTextFrame": + state = self.tts.get(self.frame_context(frame)) + if state: + state.text.append(frame.text) + text = "".join(state.text) + state.span.log(input={"text": text}, metadata={"contrib.pipecat.text": text}) + elif kind == "TTSStoppedFrame": + self.ttfb.end(("tts", self.frame_context(frame))) + state = self.tts.pop(self.frame_context(frame), None) + if state: + state.span.end() + self.synthesis.complete(state) + elif kind == "InterruptionFrame": + for state in self.tts.values(): + state.span.log(metadata={"contrib.pipecat.end_frame": kind}) + state.span.end() + self.synthesis.complete(state) + for context in self.tts: + self.ttfb.end(("tts", context)) + self.tts.clear() + elif kind in {"ErrorFrame", "FatalErrorFrame"}: + self.root.log(error=str(frame.error)) + + if kind in CONFIG_TYPES: + fields = compact_frame(fields) + identity = (kind, data.source.name) + initial = identity not in self.configuration + if not initial and self.configuration[identity] == fields: + return + if initial and len(self.configuration) >= 32: + self.omitted_events += 1 + return + self.configuration[identity] = fields + self.root.log( + metadata={ + "contrib.pipecat.configuration": [ + {"type": frame_type, "source": source, "fields": values} + for (frame_type, source), values in self.configuration.items() + ] + } + ) + # Explicit update commands are changes even on their first observation. + if initial and kind in {"SpeechControlParamsFrame", "STTMetadataFrame", "LLMServiceMetadataFrame"}: + return + if kind in EVENT_TYPES and kind not in { + "StartFrame", + "EndFrame", + "ClientConnectedFrame", + "OutputTransportReadyFrame", + }: + if kind.startswith("User"): + turn = self.turns.user + event_owner = turn["span"] if turn else self.root + elif kind.startswith("Bot") or kind == "InterruptionFrame": + turn = self.turns.assistant + event_owner = turn["span"] if turn else self.root + self.capture_event(data, fields, event_owner) + + def capture_event(self, data, fields, owner): + if len(self.events) >= 256: + self.omitted_events += 1 + return + if not self.clock_anchored: + self.root.log( + metadata={ + "braintrust.clock": { + "observer_timestamp_ns": data.timestamp, + "observed_unix_ms": time.time() * 1000, + "basis": "observer_callback", + } + } + ) + self.clock_anchored = True + event = { + "contrib.pipecat.frame.type": type(data.frame).__name__, + "contrib.pipecat.observer.timestamp": data.timestamp, + "contrib.pipecat.observer.source": data.source.name, + "contrib.pipecat.frame": compact_frame(fields), + } + size = len(json.dumps(event, ensure_ascii=False).encode("utf-8")) + if len(self.events) < 256 and self.event_bytes + size < 65536: + self.events.append(event) + self.event_bytes += size + group = self.event_groups.setdefault(owner.span_id, (owner, [])) + group[1].append(event) + else: + self.omitted_events += 1 + + async def cleanup(self): + await self.finish() + await super().cleanup() + + async def finish(self): + if self._finish_task is None: + self.finished = True + self._finish_task = asyncio.create_task(self._finish()) + await asyncio.shield(self._finish_task) + + async def _finish(self): + self.call_recording.seal() + self.hooks.close() + try: + await self._finalize() + finally: + self.hooks.close() + # A shielded segment drain can outlive cancellation of this finalizer. + # Its active tail still belongs to that drain until it seals/exports it. + self.call_recording.release_idle_buffer() + await self.synthesis.drain() + for state in self.tts.values(): + self.synthesis.release(state) + if self.user_capture: + await self.user_capture.close() + + async def _finalize(self): + self.ttfb.clear() + self.turns.finish() + for span in [self.llm, self.user, *self.tools.values()]: + if span: + span.end() + for state in self.tts.values(): + state.span.end() + state.omitted = True + state.reason = "pipeline_closed_before_tts_stop" + self.synthesis.complete(state) + for owner, events in self.event_groups.values(): + owner.log(metadata={"contrib.pipecat.events": events}) + self.root.log( + metadata={ + "braintrust.capture.events_omitted": self.omitted_events, + "braintrust.capture.recordings_omitted": self.synthesis.omitted, + "braintrust.capture.filtered_event_counts": dict(self.filtered_events), + "braintrust.capture.duplicate_events": self.duplicate_events, + "braintrust.capture.retained_events": len(self.events), + } + ) + self.root.end() + if self.user_capture: + await self.user_capture.finish() + if self.realtime: + self.realtime.finish() + # Trace lifetime ends before encoding; flush only after attachment updates. + await self.synthesis.drain() + if self.capture_transport: + await self.call_recording.finish() + self._publish_call_manifest() + self.alignment.publish() + await asyncio.to_thread(self.logger.flush) + + def _call_sources(self): + return [ + *( + [{"boundary": "model_input" if self.realtime else "transport_input", "channel_index": 0}] + if self.capture_user_audio + else [] + ), + *([{"boundary": "transport_output", "channel_index": 1}] if self.capture_agent_audio else []), + ] + + def _pending_call_segment(self, segment): + self._publish_call_manifest() + + async def _publish_call_segment(self, segment): + if segment.encoded is None: + self._publish_call_manifest() + return + self.root.log(input={"audio": {segment.segment_id: segment.encoded["attachment"]}}) + self._publish_call_manifest() + self.alignment.publish() + await asyncio.to_thread(self.logger.flush) + + def _publish_call_manifest(self): + descriptors = [ + segment_descriptor(segment, self.root.span_id, self._call_sources()) + for segment in self.call_recording.segments + ] + reason = self.call_recording.reason + if reason or not descriptors: + descriptors.append( + { + "id": "call", + "recording_group_id": "call", + "state": "omitted", + "reason": reason or "no_audio_observed", + "truncated": reason not in (None, "disabled"), + "sources": self._call_sources(), + } + ) + self.root.log(metadata={"audio.recordings": descriptors}) + + def _turn_completed(self, turn): + if turn["role"] == "user" and self.user_capture: + self.user_capture.queue_completed(turn) diff --git a/py/src/braintrust/integrations/pipecat/voice/realtime.py b/py/src/braintrust/integrations/pipecat/voice/realtime.py new file mode 100644 index 000000000..25488f3e7 --- /dev/null +++ b/py/src/braintrust/integrations/pipecat/voice/realtime.py @@ -0,0 +1,321 @@ +"""Pinned OpenAI Realtime hooks: native event IDs and sent-sample provenance. + +Server VAD + uninterrupted PCM input only. Buffer clears invalidate selections. +No STT operation is fabricated for asynchronous provider transcription events. +""" + +from contextvars import ContextVar + +from braintrust._audio import ms_to_samples + +from ..llm_metrics import _metadata_from_processor +from .instrumentation import native_value + + +class RealtimeCapture: + def __init__(self, observer, service, aggregator): + self.observer = observer + self.items = {} + self.consumed = [] + self.associations = [] + self.sent_samples = 0 + self.sent_ranges: list[tuple[int, int, dict[str, int]]] = [] + self.invalid = False + self.omitted = 0 + self.frame_token = object() + self.active = ContextVar("realtime_response", default=None) + self.service = service + self.user_item = None + self.user_turns = {} + self.pending_tool_results = [] + original_tool_result = observer.hooks.original(service, "_send_tool_result") + + async def tool_result(tool_call_id, result): + self.pending_tool_results.append(tool_call_id) + return await original_tool_result(tool_call_id, result) + + observer.hooks.set(service, "_send_tool_result", tool_result) + + async def user_started(aggregator, strategy): + if self.user_item and observer.turns.user: + self.user_turns[self.user_item] = observer.turns.user + if len(self.user_turns) > 256: + self.user_turns.pop(next(iter(self.user_turns))) + observer.turns.user["span"].log(metadata={"openai.item_id": self.user_item}) + + observer.hooks.event(aggregator, "on_user_turn_started", user_started) + original_item = observer.hooks.original(service, "_handle_evt_conversation_item_added") + + async def item_added(evt): + if evt.item.type == "function_call" and observer.llm is None: + turn = self.user_turns.get(self.user_item) + observer.context_reply_to = turn["span"].span_id if turn else None + observer.llm_turn = None + observer.llm_text = [] + observer.llm_tool_calls = [] + observer.llm = observer.root.start_span( + name="llm_response", + type="llm", + set_current=False, + internal={"instrumentation": "pipecat-auto"}, + metadata={ + **_metadata_from_processor(service), + "openai.item_id": evt.item.id, + "openai.call_id": evt.item.call_id, + "turn.reply_to": observer.context_reply_to, + }, + ) + observer.ttfb.start("llm", service, observer.llm.log) + return await original_item(evt) + + observer.hooks.set(service, "_handle_evt_conversation_item_added", item_added) + original_push = observer.hooks.original(service, "push_frame") + + async def push(frame, *args, **kwargs): + context = self.active.get() + if context and type(frame).__name__ in { + "TTSStartedFrame", + "TTSStoppedFrame", + "TTSAudioRawFrame", + "TTSTextFrame", + }: + # Keep provenance on the queued frame, not in a call-long ID map. + frame._braintrust_realtime_context = (self.frame_token, context) + return await original_push(frame, *args, **kwargs) + + observer.hooks.set(service, "push_frame", push) + for name in ("_handle_evt_audio_delta", "_handle_evt_audio_transcript_delta", "_handle_evt_response_done"): + original = observer.hooks.original(service, name) + + async def response(evt, original=original): + response_id = getattr(evt, "response_id", None) or getattr(getattr(evt, "response", None), "id", None) + token = self.active.set(response_id) + response_span = observer.llm + if response_span and response_id: + response_span.log(metadata={"openai.response.id": response_id}) + if getattr(evt, "response", None) is not None: + response_span.log(metadata={"openai.response": native_value(evt.response)}) + observer.llm_tool_calls = [ + { + "id": item.call_id, + "type": "function", + "function": {"name": item.name, "arguments": item.arguments}, + } + for item in evt.response.output + if item.type == "function_call" + ] + try: + result = await original(evt) + if response_span and getattr(evt, "response", None) is not None: + calls = [item for item in evt.response.output if item.type == "function_call"] + if calls: + response_span.log( + output=[ + { + "role": "assistant", + "content": "".join(observer.llm_text) or None, + "tool_calls": [ + { + "id": item.call_id, + "type": "function", + "function": {"name": item.name, "arguments": item.arguments}, + } + for item in calls + ], + } + ] + ) + state = observer.tts.get(response_id) + if state: + state["span"].log( + metadata={ + "openai.response.id": response_id, + "openai.item_id": getattr(evt, "item_id", None), + } + ) + return result + finally: + self.active.reset(token) + + observer.hooks.set(service, name, response) + + for name, field in ( + ("_handle_evt_speech_started", "audio_start_ms"), + ("_handle_evt_speech_stopped", "audio_end_ms"), + ): + original = observer.hooks.original(service, name) + + async def speech(evt, original=original, field=field): + if field == "audio_start_ms": + self.user_item = evt.item_id + if evt.item_id not in self.items and len(self.items) >= 256: + self.items.pop(next(iter(self.items))) + self.omitted += 1 + self.items.setdefault(evt.item_id, {})[field] = getattr(evt, field) + result = await original(evt) + self.publish() + return result + + observer.hooks.set(service, name, speech) + + original_accept = observer.hooks.original(aggregator, "_handle_transcription") + original_commit = observer.hooks.original(aggregator, "_push_aggregation") + original_reset = observer.hooks.original(aggregator, "reset") + + async def accept(frame): + result = await original_accept(frame) + if frame.text.strip() and len(self.consumed) < 256: + self.consumed.append(native_value(frame)) + return result + + async def reset(): + self.consumed.clear() + return await original_reset() + + async def commit(*args, **kwargs): + consumed = self.consumed[:] + turn = observer.turns.user or observer.turns.messages.get(("user", aggregator._user_turn_start_timestamp)) + if turn: + observer.context_reply_to = turn["span"].span_id + result = await original_commit(*args, **kwargs) + if result and turn: + ids = [f.get("result", {}).get("item_id") for f in consumed if isinstance(f.get("result"), dict)] + turn["span"].log( + metadata={ + "contrib.pipecat.transcriptions": consumed, + "openai.item_ids": ids, + "braintrust.user_capture.association": "aggregator_consumed_frames", + } + ) + if len(self.associations) >= 256: + self.associations.pop(0) + self.omitted += 1 + self.associations.append((turn["span"], ids)) + self.publish() + observer.context_reply_to = turn["span"].span_id + return result + + observer.hooks.set(aggregator, "_handle_transcription", accept) + observer.hooks.set(aggregator, "_push_aggregation", commit) + observer.hooks.set(aggregator, "reset", reset) + # Opt-out never installs the per-audio-frame send hook or stores sample ranges. + if observer.capture_user_audio: + original_send = observer.hooks.original(service, "_send_user_audio") + original_event = observer.hooks.original(service, "send_client_event") + original_connect = observer.hooks.original(service, "_connect") + audio_frame = ContextVar("realtime_input_frame", default=None) + self.bound_socket = None + + async def connect(): + result = await original_connect() + socket = service._websocket + if socket is not None and socket is not self.bound_socket: + if self.bound_socket is not None: + self.invalid = True # Provider buffer clock may restart. + self.bound_socket = socket + original_socket_send = observer.hooks.original(socket, "send") + + async def socket_send(message, *args, **kwargs): + frame = audio_frame.get() + try: + result = await original_socket_send(message, *args, **kwargs) + except BaseException: + if frame is not None: + self.invalid = True + raise + if frame is not None: + record_sent(frame) + return result + + observer.hooks.set(socket, "send", socket_send) + return result + + def record_sent(frame): + if frame.sample_rate != 24000 or frame.num_channels != 1: + self.invalid = True + try: + interval = observer.call_recording.capture(0, frame.audio, frame.sample_rate, frame.num_channels) + except Exception: + observer.call_recording.omit("capture_error") + interval = None + size = len(frame.audio) // 2 + if interval and not self.invalid: + merged = False + if self.sent_ranges: + start, end, previous = self.sent_ranges[-1] + if end == self.sent_samples and previous["end"] == interval["start"]: + self.sent_ranges[-1] = ( + start, + self.sent_samples + size, + {"start": previous["start"], "end": interval["end"]}, + ) + merged = True + if not merged: + if len(self.sent_ranges) >= 16000: + self.sent_ranges.pop(0) + self.omitted += 1 + self.sent_ranges.append((self.sent_samples, self.sent_samples + size, interval)) + self.sent_samples += size + if self.associations: + self.publish() + + async def send_audio(frame): + token = audio_frame.set(frame) + try: + return await original_send(frame) + finally: + audio_frame.reset(token) + + observer.hooks.set(service, "_connect", connect) + + async def send_event(evt): + if evt.type == "input_audio_buffer.clear": + self.publish() + self.invalid = True + return await original_event(evt) + + observer.hooks.set(service, "_send_user_audio", send_audio) + observer.hooks.set(service, "send_client_event", send_event) + + def publish(self, final=False): + pending = [] + completed = set() + for span, ids in self.associations: + events = [{"item_id": item, **self.items.get(item, {})} for item in ids] + ready = all("audio_start_ms" in event and "audio_end_ms" in event for event in events) + if ready and self.observer.capture_user_audio and not self.invalid: + ready = all(ms_to_samples(event["audio_end_ms"]) <= self.sent_samples for event in events) + if not ready and not final: + pending.append((span, ids)) + continue + span.log(metadata={"openai.input_audio_segments": events}) + if self.observer.capture_user_audio and not self.invalid: + for event in events: + if "audio_start_ms" not in event or "audio_end_ms" not in event: + continue + start, end = ms_to_samples(event["audio_start_ms"]), ms_to_samples(event["audio_end_ms"]) + ranges = [] + for a, b, interval in self.sent_ranges: + left, right = max(start, a), min(end, b) + if right > left: + ranges.append([interval["start"] + left - a, interval["start"] + right - a]) + if sum(b - a for a, b in ranges) == end - start: + self.observer.alignment.add(span, ranges, 0) + else: + self.omitted += 1 + completed.update(ids) + for item in completed.difference(item for _, ids in pending for item in ids): + self.items.pop(item, None) + self.associations = pending + + def finish(self): + self.publish(final=True) + self.observer.root.log( + metadata={ + "braintrust.realtime.alignment_invalidated": self.invalid, + "braintrust.realtime.associations_omitted": self.omitted, + } + ) + self.sent_ranges.clear() + self.items.clear() + self.user_turns.clear() diff --git a/py/src/braintrust/integrations/pipecat/voice/recording.py b/py/src/braintrust/integrations/pipecat/voice/recording.py new file mode 100644 index 000000000..380d3a51e --- /dev/null +++ b/py/src/braintrust/integrations/pipecat/voice/recording.py @@ -0,0 +1,22 @@ +"""Pipecat transport capture hooks.""" + +import time + + +def capture_transport_output(output, recording): + """Observe only successful writes and preserve the transport's pacing/result.""" + original = output.write_audio_frame + + async def write(frame): + if recording.reason: + return await original(frame) + observed_ns = time.monotonic_ns() + result = await original(frame) + if result: + try: + recording.capture(1, frame.audio, frame.sample_rate, frame.num_channels, observed_ns=observed_ns) + except Exception: # noqa: BLE001 - capture/export failures must not break the call + recording.omit("capture_error") + return result + + output.write_audio_frame = write diff --git a/py/src/braintrust/integrations/pipecat/voice/synthesis.py b/py/src/braintrust/integrations/pipecat/voice/synthesis.py new file mode 100644 index 000000000..2534a58aa --- /dev/null +++ b/py/src/braintrust/integrations/pipecat/voice/synthesis.py @@ -0,0 +1,189 @@ +"""Generated speech capture and publication, independent of frame dispatch.""" + +import asyncio +from dataclasses import dataclass, field + +from braintrust._audio import ( + RecordingBusy, + RecordingJobs, + UploadFailed, + encode_audio, + encode_in_worker, + prepare_recording, + source_budget, + upload_recording, +) +from braintrust.logger import Span + + +@dataclass +class Synthesis: + span: Span + chunks: list[bytes] = field(default_factory=list) + text: list[str] = field(default_factory=list) + omitted: bool = False + reason: str | None = None + rate: int | None = None + channels: int | None = None + observed_format: tuple[int, int] | None = None + last_pts: int | None = None + + +class SynthesisRecordings: + def __init__(self, logger, alignment, *, enabled, max_bytes, audio_format): + self.logger = logger + self.alignment = alignment + self.enabled = enabled + self.max_bytes = max_bytes + self.audio_format = audio_format + self.bytes = 0 + self.omitted = 0 + self.jobs = RecordingJobs() + + def __del__(self): + retained = getattr(self, "bytes", 0) + if retained: + source_budget.release(retained) + self.bytes = 0 + + async def drain(self): + await self.jobs.drain() + + def capture(self, state, frame): + audio_format = (frame.sample_rate, frame.num_channels) + state.last_pts = frame.pts + if state.observed_format != audio_format: + state.observed_format = audio_format + state.span.log( + metadata={ + "contrib.pipecat.sample_rate": frame.sample_rate, + "contrib.pipecat.num_channels": frame.num_channels, + "contrib.pipecat.frame.pts": frame.pts, + } + ) + if state.chunks and (state.rate, state.channels) != audio_format: + state.omitted = True + state.reason = "audio_format_changed" + self.release(state) + if self.enabled and not state.omitted: + if self.bytes + len(frame.audio) <= self.max_bytes and source_budget.reserve(len(frame.audio)): + state.chunks.append(frame.audio) + self.bytes += len(frame.audio) + state.rate, state.channels = ( + frame.sample_rate, + frame.num_channels, + ) + else: + state.omitted = True + state.reason = ( + "capture_byte_limit" + if self.bytes + len(frame.audio) > self.max_bytes + else "process_capture_byte_limit" + ) + self.release(state) + + def complete(self, state): + if state.observed_format is not None: + state.span.log(metadata={"contrib.pipecat.frame.pts": state.last_pts}) + if len(self.jobs) < 256: + self.jobs.submit( + lambda: self._publish(state), + lambda: self.release(state), + after_release=lambda: asyncio.to_thread(self.logger.flush), + ) + if state.chunks and not state.omitted: + state.span.log( + metadata={ + "audio.recordings": [ + {"id": "tts-clip", "state": "pending", "sources": [{"boundary": "tts_output"}]} + ] + } + ) + return + self.omitted += 1 + state.span.log( + metadata={ + "audio.recordings": [ + { + "id": "tts-clip", + "state": "omitted", + "reason": "recording_count_limit", + "sources": [{"boundary": "tts_output"}], + } + ] + } + ) + self.release(state) + + def release(self, state): + size = sum(map(len, state.chunks)) + state.chunks.clear() + source_budget.release(size) + self.bytes -= size + + async def _publish(self, state): + span = state.span + recording = {"id": "tts-clip", "sources": [{"boundary": "tts_output"}]} + if state.chunks and not state.omitted: + try: + encoded = await encode_in_worker( + prepare_recording, + "tts-clip", + encode_audio, + state.chunks, + state.rate, + state.channels, + self.audio_format, + ) + await upload_recording(encoded) + except Exception as error: # noqa: BLE001 - capture/export failures must not break the call + span.log( + metadata={ + "audio.recordings": [ + { + **recording, + "state": "omitted", + "reason": str(error) + if isinstance(error, (RecordingBusy, UploadFailed)) + else "encoding_failed", + } + ] + } + ) + return + recording.update( + state="ready", + attachment={ + "span_id": span.span_id, + "ref": "/output/0/content/1/file/file_data", + }, + mime_type=encoded["mime_type"], + duration_ms=encoded["duration_ms"], + channel_count=encoded["channel_count"], + ) + span.log( + output=[ + { + "role": "assistant", + "content": [ + {"type": "text", "text": "".join(state.text)}, + { + "type": "file", + "file": { + "file_data": encoded["attachment"], + "filename": f"tts-clip.{encoded['extension']}", + }, + }, + ], + } + ], + metadata={"braintrust.recording.processing": encoded["processing"]}, + ) + else: + recording.update( + state="omitted", + reason=(state.reason or "no_audio_observed") if self.enabled else "disabled", + ) + span.log(metadata={"audio.recordings": [recording]}) + if recording["state"] == "ready": + self.alignment.publish_clip(span, recording) diff --git a/py/src/braintrust/integrations/pipecat/voice/test_alignment.py b/py/src/braintrust/integrations/pipecat/voice/test_alignment.py new file mode 100644 index 000000000..419a6ecb8 --- /dev/null +++ b/py/src/braintrust/integrations/pipecat/voice/test_alignment.py @@ -0,0 +1,140 @@ +import asyncio +import io +import unittest +from unittest.mock import patch + +import numpy as np +import soundfile as sf # pylint: disable=import-error +from braintrust._audio.export import AlignmentPublisher +from braintrust._audio.recording import CallRecording +from pipecat.frames.frames import TTSAudioRawFrame, TTSStoppedFrame # pylint: disable=import-error +from pipecat.transports.base_output import BaseOutputTransport # pylint: disable=import-error +from pipecat.transports.base_transport import TransportParams # pylint: disable=import-error + +from .alignment import instrument_output +from .test_instrumentation import Span + + +class AlignmentTests(unittest.IsolatedAsyncioTestCase): + async def make_output(self, enabled=True): + class Output: + success = True + + def create_task(self, coroutine): + coroutine.close() + return object() + + async def start(self, frame): + pass + + async def write_audio_frame(self, frame): + return self.success + + output = Output() + params = TransportParams(audio_out_enabled=True) + sender = BaseOutputTransport.MediaSender( + output, + destination=None, + sample_rate=24000, + audio_chunk_size=8, + params=params, + ) + sender._audio_queue = asyncio.Queue() + output._media_senders = {None: sender} + recording = CallRecording(enabled=enabled) + alignment = AlignmentPublisher(Span(), recording) + instrument_output(output, alignment) + await output.start(None) + self.addCleanup(sender._executor.shutdown) + return output, sender, alignment + + async def test_native_chunk_crossing_contexts_and_flush_padding(self): + output, sender, alignment = await self.make_output() + first, second = Span(), Span() + alignment.begin_output("a", [first]) + alignment.begin_output("b", [second]) + + def audio(context, samples): + return TTSAudioRawFrame( + audio=np.array(samples, dtype=" 512: + self.messages.pop(next(iter(self.messages))) + self.states = [pending for pending in self.states if pending is not state] + if getattr(self, role) is state: + setattr(self, role, None) + + def finish(self): + for state in self.states: + if not state["ended"]: + state["span"].log(metadata={"turn.incomplete": True}) + state["span"].end() + state["ended"] = True diff --git a/py/src/braintrust/integrations/pipecat/voice/user_capture.py b/py/src/braintrust/integrations/pipecat/voice/user_capture.py new file mode 100644 index 000000000..db3118c2a --- /dev/null +++ b/py/src/braintrust/integrations/pipecat/voice/user_capture.py @@ -0,0 +1,407 @@ +"""Pinned Pipecat 1.12 cascade hooks: STT segment -> accepted frames -> turn. + +The native aggregator owns membership. No transcript or timestamp matching. +""" + +import asyncio +import io +import time +import wave +from collections import deque + +from braintrust._audio import ( + ClipTimeline, + InputRanges, + RecordingBusy, + RecordingJobs, + UploadFailed, + encode_audio, + encode_in_worker, + pcm_bytes_to_ms, + prepare_recording, + samples_to_ms, + source_budget, + upload_recording, +) +from pipecat.frames.frames import TranscriptionFrame # pylint: disable=import-error + +from ..llm_metrics import _metadata_from_processor + + +def encode_segments(segments, audio_format, mappings=None): + chunks = [] + rate = channels = None + timeline = ClipTimeline() + position_ms = 0 + for index, segment in enumerate(segments): + with wave.open(io.BytesIO(segment), "rb") as source: + if source.getsampwidth() != 2: + raise ValueError("Expected PCM16 STT input") + current = source.getframerate(), source.getnchannels() + if rate is not None and current != (rate, channels): + raise ValueError("Mixed STT input formats") + rate, channels = current + chunks.append(source.readframes(source.getnframes())) + for start, end, call_start, call_end in mappings[index] if mappings else []: + timeline.add( + position_ms + pcm_bytes_to_ms(start, rate, channels), + position_ms + pcm_bytes_to_ms(end, rate, channels), + samples_to_ms(call_start), + samples_to_ms(call_end), + ) + position_ms += source.getnframes() / rate * 1000 + encoded = encode_audio(chunks, rate, channels, audio_format) + encoded["clip_timeline"] = timeline + return encoded + + +class UserCapture: + def __init__(self, observer, stt, aggregator, native_value): + self.observer = observer + self.accepted = [] + self.by_frame = {} + self.batches = {} + self.bytes = 0 + self.boundaries = [] + self.segment_boundaries = deque(maxlen=256) + self.omitted = 0 + self.input_ranges = InputRanges() if observer.capture_user_audio else None + self.segments = [] + self.jobs = RecordingJobs() + + original_start = observer.hooks.original(stt, "_handle_user_started_speaking") + original_stop = observer.hooks.original(stt, "_handle_user_stopped_speaking") + original_run = observer.hooks.original(stt, "run_stt") + original_accept = observer.hooks.original(aggregator, "_handle_transcription") + original_commit = observer.hooks.original(aggregator, "_push_aggregation") + original_reset = observer.hooks.original(aggregator, "reset") + original_audio = observer.hooks.original(stt, "process_audio_frame") + + async def process_audio(frame, direction): + try: + interval = observer.call_recording.capture(0, frame.audio, frame.sample_rate, frame.num_channels) + except Exception as error: # noqa: BLE001 - recording must not stop input processing + observer.call_recording.omit("capture_error") + interval = None + result = await original_audio(frame, direction) + self.input_ranges.append(len(frame.audio), interval) + # Mirror the actual retained native buffer length, not a guessed VAD time. + self.input_ranges.trim(len(stt._audio_buffer)) + return result + + async def speech_start(frame): + self.boundaries = [ + { + "contrib.pipecat.frame.type": type(frame).__name__, + "contrib.pipecat.frame": native_value(frame), + } + ] + return await original_start(frame) + + async def speech_stop(frame): + if stt.is_usable: + pieces = self.input_ranges.drain_mapped() if self.input_ranges else [] + self.segment_boundaries.append( + { + "ranges": [[a, b] for _, _, a, b in pieces], + "clip_ranges": pieces, + "events": [ + *self.boundaries, + { + "contrib.pipecat.frame.type": type(frame).__name__, + "contrib.pipecat.frame": native_value(frame), + }, + ], + } + ) + elif self.input_ranges: + self.input_ranges.drain() + self.boundaries = [] + return await original_stop(frame) + + async def run(audio): + queued = self.segment_boundaries.popleft() if self.segment_boundaries else {"events": [], "ranges": []} + segment = { + "service_metadata": _metadata_from_processor(stt), + "boundaries": queued["events"], + "ranges": queued["ranges"], + "clip_ranges": queued.get("clip_ranges", []), + "audio": None, + "reason": "disabled", + "start": time.time(), + "end": None, + "frames": [], + "span": None, + } + tracked = len(self.segments) < 256 + if tracked: + self.segments.append(segment) + else: + self.omitted += 1 + segment["reason"] = "capture_backlog_limit" + if observer.capture_user_audio and tracked: + if self.bytes + len(audio) <= observer.max_audio_bytes and source_budget.reserve(len(audio)): + segment.update(audio=audio, reason=None) + self.bytes += len(audio) + else: + segment["reason"] = ( + "capture_byte_limit" + if self.bytes + len(audio) > observer.max_audio_bytes + else "process_capture_byte_limit" + ) + segment["ttfb_metadata"] = {} + + def log_ttfb(**event): + segment["ttfb_metadata"].update(event["metadata"]) + if segment["span"] is not None: + segment["span"].log(**event) + + observer.ttfb.start(("stt", id(segment)), stt, log_ttfb) + try: + async for frame in original_run(audio): + try: + segment["end"] = time.time() + if isinstance(frame, TranscriptionFrame): + segment["frames"].append(native_value(frame)) + if len(self.by_frame) < 256: + self.by_frame[frame.id] = segment + else: + self.omitted += 1 + elif getattr(frame, "error", None): + segment["error"] = str(frame.error) + except Exception: # noqa: BLE001 - always deliver the native STT frame + self.omitted += 1 + yield frame + except BaseException as error: + segment["error"] = type(error).__name__ + raise + finally: + observer.ttfb.end(("stt", id(segment))) + if segment["end"] is None: + segment["end"] = time.time() + + async def accept(frame): + result = await original_accept(frame) + if frame.text.strip(): + if len(self.accepted) < 256: + self.accepted.append((native_value(frame), self.by_frame.pop(frame.id, None))) + else: + self.omitted += 1 + return result + + async def reset(): + self.accepted = [] + return await original_reset() + + async def commit(*args, **kwargs): + # Snapshot before native reset/context push. The original method + # can schedule turn-stop callbacks while it awaits downstream work. + accepted = self.accepted[:] + turn = observer.turns.user + result = await original_commit(*args, **kwargs) + if result and accepted: + owner = turn["span"] if turn else observer.root + segments = [] + for _, segment in accepted: + if segment is not None and not any(s is segment for s in segments): + segments.append(segment) + self.create_stt(segment, owner, turn) + owner.log( + metadata={ + "contrib.pipecat.transcriptions": [frame for frame, _ in accepted], + "contrib.pipecat.speech_events": [event for s in segments for event in s["boundaries"]], + "braintrust.user_capture.association": "aggregator_consumed_frames", + } + ) + if turn and (turn["span"].span_id in self.batches or len(self.batches) < 256): + batch = self.batches.setdefault( + turn["span"].span_id, + {"turn": turn, "segments": [], "frames": []}, + ) + batch["frames"].extend(frame for frame, _ in accepted) + for segment in segments: + if not any(s is segment for s in batch["segments"]): + batch["segments"].append(segment) + owner.log( + metadata={ + "contrib.pipecat.transcriptions": batch["frames"], + "contrib.pipecat.speech_events": [ + event for s in batch["segments"] for event in s["boundaries"] + ], + } + ) + else: + if turn: + self.omitted += 1 + owner.log( + metadata={ + "audio.recordings": [ + { + "id": "user-clip", + "state": "omitted", + "reason": "capture_backlog_limit", + } + ] + } + ) + self.release_segments(segments) + if turn: + self.queue_completed(turn) + return result + + observer.hooks.set(stt, "_handle_user_started_speaking", speech_start) + observer.hooks.set(stt, "_handle_user_stopped_speaking", speech_stop) + observer.hooks.set(stt, "run_stt", run) + if observer.capture_user_audio: + observer.hooks.set(stt, "process_audio_frame", process_audio) + observer.hooks.set(aggregator, "_handle_transcription", accept) + observer.hooks.set(aggregator, "_push_aggregation", commit) + observer.hooks.set(aggregator, "reset", reset) + + def __del__(self): + retained = getattr(self, "bytes", 0) + if retained: + source_budget.release(retained) + self.bytes = 0 + + def create_stt(self, segment, owner, turn=None): + if segment["span"] is not None: + return + metadata = { + "contrib.pipecat.transcriptions": segment["frames"], + "contrib.pipecat.function": "run_stt", + **segment.get("service_metadata", {}), + **segment.get("ttfb_metadata", {}), + } + if turn: + metadata.update(self.observer.turns.metadata(turn)) + span = owner.start_span( + name="stt", + type="task", + start_time=segment["start"], + set_current=False, + internal={"instrumentation": "pipecat-auto"}, + metadata=metadata, + ) + text = " ".join(f["text"] for f in segment["frames"]) + span.log(output=[{"role": "user", "content": text}]) + if segment.get("error"): + span.log(error=segment["error"]) + span.end(end_time=segment["end"] or time.time()) + segment["span"] = span + self.segments = [pending for pending in self.segments if pending is not segment] + self.observer.alignment.add(span, segment["ranges"], 0) + if turn: + self.observer.alignment.add(owner, segment["ranges"], 0) + span.log(input={"recording_span_id": owner.span_id, "recording_id": "user-clip"}) + + async def finish(self): + for segment in list(self.segments): + self.create_stt(segment, self.observer.root) + for batch in list(self.batches.values()): + self.queue_completed(batch["turn"], force=True) + await self.jobs.drain() + self.observer.root.log(metadata={"braintrust.user_capture.events_omitted": self.omitted}) + self.release() + + async def close(self): + await self.jobs.drain() + self.release() + + def queue_completed(self, turn, force=False): + batch = self.batches.get(turn["span"].span_id) + if not batch or batch.get("queued") or (not force and not turn.get("ended")): + return + batch["queued"] = True + + def release(): + self.release_segments(batch["segments"]) + self.batches.pop(turn["span"].span_id, None) + + self.jobs.submit( + lambda: self._publish_batch(batch), + release, + after_release=lambda: asyncio.to_thread(self.observer.logger.flush), + ) + + if any(s["audio"] is not None for s in batch["segments"]): + turn["span"].log( + metadata={ + "audio.recordings": [ + {"id": "user-clip", "state": "pending", "sources": [{"boundary": "stt_input"}]} + ] + } + ) + + async def _publish_batch(self, batch): + turn, segments = batch["turn"], batch["segments"] + descriptor = {"id": "user-clip", "sources": [{"boundary": "stt_input"}]} + reason = next((s["reason"] for s in segments if s["reason"]), None) + audio = [s["audio"] for s in segments if s["audio"] is not None] + if reason or not audio: + descriptor.update(state="omitted", reason=reason or "no_audio_observed") + else: + try: + encoded = await encode_in_worker( + prepare_recording, + "user-clip", + encode_segments, + audio, + self.observer.audio_format, + [s["clip_ranges"] for s in segments if s["audio"] is not None], + ) + await upload_recording(encoded) + except Exception as error: # noqa: BLE001 - recording failure must not break the call + descriptor.update( + state="omitted", + reason=str(error) if isinstance(error, (RecordingBusy, UploadFailed)) else "encoding_failed", + ) + else: + descriptor.update( + state="ready", + attachment={ + "span_id": turn["span"].span_id, + "ref": "/input/0/content/1/file/file_data", + }, + mime_type=encoded["mime_type"], + duration_ms=encoded["duration_ms"], + channel_count=encoded["channel_count"], + ) + origin = self.observer.call_recording.origin_unix_ms + if origin is not None: + descriptor["timeline"] = encoded["clip_timeline"].descriptor(origin) + attachment = encoded["attachment"] + turn["span"].log( + input=[ + { + "role": "user", + "content": [ + {"type": "text", "text": turn.get("content") or ""}, + { + "type": "file", + "file": { + "file_data": attachment, + "filename": f"user-clip.{encoded['extension']}", + }, + }, + ], + } + ] + ) + turn["span"].log(metadata={"audio.recordings": [descriptor]}) + + def release_segments(self, segments): + for segment in segments: + if segment["audio"] is not None: + size = len(segment["audio"]) + segment["audio"] = None + self.bytes -= size + source_budget.release(size) + + def release(self): + self.batches.clear() + self.by_frame.clear() + self.accepted.clear() + self.segments.clear() + source_budget.release(self.bytes) + self.bytes = 0 diff --git a/py/uv.lock b/py/uv.lock index 09698dcf4..127d5cbe8 100644 --- a/py/uv.lock +++ b/py/uv.lock @@ -804,6 +804,12 @@ all = [ { name = "uv" }, { name = "uvicorn" }, ] +audio = [ + { name = "numpy", version = "2.2.6", source = { registry = "https://pypi.org/simple" }, marker = "python_full_version < '3.11' or (extra == 'group-10-braintrust-lint' and extra == 'group-10-braintrust-test-agentscope') or (extra == 'group-10-braintrust-lint' and extra == 'group-10-braintrust-test-agno') or (extra == 'group-10-braintrust-lint' and extra == 'group-10-braintrust-test-crewai') or (extra == 'group-10-braintrust-lint' and extra == 'group-10-braintrust-test-deepagents') or (extra == 'group-10-braintrust-lint' and extra == 'group-10-braintrust-test-langchain') or (extra == 'group-10-braintrust-lint' and extra == 'group-10-braintrust-test-litellm') or (extra == 'group-10-braintrust-lint' and extra == 'group-10-braintrust-test-livekit-agents') or (extra == 'group-10-braintrust-lint' and extra == 'group-10-braintrust-test-openai-agents') or (extra == 'group-10-braintrust-lint' and extra == 'group-10-braintrust-test-pydantic-ai-logfire') or (extra == 'group-10-braintrust-lint' and extra == 'group-10-braintrust-test-strands') or (extra == 'group-10-braintrust-test-agentscope' and extra == 'group-10-braintrust-test-agno') or (extra == 'group-10-braintrust-test-agentscope' and extra == 'group-10-braintrust-test-crewai') or (extra == 'group-10-braintrust-test-agentscope' and extra == 'group-10-braintrust-test-deepagents') or (extra == 'group-10-braintrust-test-agentscope' and extra == 'group-10-braintrust-test-langchain') or (extra == 'group-10-braintrust-test-agentscope' and extra == 'group-10-braintrust-test-litellm') or (extra == 'group-10-braintrust-test-agentscope' and extra == 'group-10-braintrust-test-livekit-agents') or (extra == 'group-10-braintrust-test-agentscope' and extra == 'group-10-braintrust-test-openai-agents') or (extra == 'group-10-braintrust-test-agentscope' and extra == 'group-10-braintrust-test-strands') or (extra == 'group-10-braintrust-test-agno' and extra == 'group-10-braintrust-test-crewai') or (extra == 'group-10-braintrust-test-agno' and extra == 'group-10-braintrust-test-deepagents') or (extra == 'group-10-braintrust-test-agno' and extra == 'group-10-braintrust-test-langchain') or (extra == 'group-10-braintrust-test-agno' and extra == 'group-10-braintrust-test-litellm') or (extra == 'group-10-braintrust-test-agno' and extra == 'group-10-braintrust-test-livekit-agents') or (extra == 'group-10-braintrust-test-agno' and extra == 'group-10-braintrust-test-openai-agents') or (extra == 'group-10-braintrust-test-agno' and extra == 'group-10-braintrust-test-strands') or (extra == 'group-10-braintrust-test-crewai' and extra == 'group-10-braintrust-test-deepagents') or (extra == 'group-10-braintrust-test-crewai' and extra == 'group-10-braintrust-test-langchain') or (extra == 'group-10-braintrust-test-crewai' and extra == 'group-10-braintrust-test-litellm') or (extra == 'group-10-braintrust-test-crewai' and extra == 'group-10-braintrust-test-livekit-agents') or (extra == 'group-10-braintrust-test-crewai' and extra == 'group-10-braintrust-test-openai-agents') or (extra == 'group-10-braintrust-test-crewai' and extra == 'group-10-braintrust-test-strands') or (extra == 'group-10-braintrust-test-deepagents' and extra == 'group-10-braintrust-test-langchain') or (extra == 'group-10-braintrust-test-deepagents' and extra == 'group-10-braintrust-test-litellm') or (extra == 'group-10-braintrust-test-deepagents' and extra == 'group-10-braintrust-test-livekit-agents') or (extra == 'group-10-braintrust-test-deepagents' and extra == 'group-10-braintrust-test-openai-agents') or (extra == 'group-10-braintrust-test-deepagents' and extra == 'group-10-braintrust-test-strands') or (extra == 'group-10-braintrust-test-langchain' and extra == 'group-10-braintrust-test-litellm') or (extra == 'group-10-braintrust-test-langchain' and extra == 'group-10-braintrust-test-livekit-agents') or (extra == 'group-10-braintrust-test-langchain' and extra == 'group-10-braintrust-test-openai-agents') or (extra == 'group-10-braintrust-test-langchain' and extra == 'group-10-braintrust-test-strands') or (extra == 'group-10-braintrust-test-litellm' and extra == 'group-10-braintrust-test-livekit-agents') or (extra == 'group-10-braintrust-test-litellm' and extra == 'group-10-braintrust-test-openai-agents') or (extra == 'group-10-braintrust-test-litellm' and extra == 'group-10-braintrust-test-strands') or (extra == 'group-10-braintrust-test-livekit-agents' and extra == 'group-10-braintrust-test-openai-agents') or (extra == 'group-10-braintrust-test-livekit-agents' and extra == 'group-10-braintrust-test-pydantic-ai-logfire') or (extra == 'group-10-braintrust-test-livekit-agents' and extra == 'group-10-braintrust-test-pydantic-ai-otel-events') or (extra == 'group-10-braintrust-test-livekit-agents' and extra == 'group-10-braintrust-test-strands') or (extra == 'group-10-braintrust-test-openai-agents' and extra == 'group-10-braintrust-test-strands')" }, + { name = "numpy", version = "2.4.6", source = { registry = "https://pypi.org/simple" }, marker = "python_full_version == '3.11.*' or (extra == 'group-10-braintrust-lint' and extra == 'group-10-braintrust-test-agentscope') or (extra == 'group-10-braintrust-lint' and extra == 'group-10-braintrust-test-agno') or (extra == 'group-10-braintrust-lint' and extra == 'group-10-braintrust-test-crewai') or (extra == 'group-10-braintrust-lint' and extra == 'group-10-braintrust-test-deepagents') or (extra == 'group-10-braintrust-lint' and extra == 'group-10-braintrust-test-langchain') or (extra == 'group-10-braintrust-lint' and extra == 'group-10-braintrust-test-litellm') or (extra == 'group-10-braintrust-lint' and extra == 'group-10-braintrust-test-livekit-agents') or (extra == 'group-10-braintrust-lint' and extra == 'group-10-braintrust-test-openai-agents') or (extra == 'group-10-braintrust-lint' and extra == 'group-10-braintrust-test-pydantic-ai-logfire') or (extra == 'group-10-braintrust-lint' and extra == 'group-10-braintrust-test-strands') or (extra == 'group-10-braintrust-test-agentscope' and extra == 'group-10-braintrust-test-agno') or (extra == 'group-10-braintrust-test-agentscope' and extra == 'group-10-braintrust-test-crewai') or (extra == 'group-10-braintrust-test-agentscope' and extra == 'group-10-braintrust-test-deepagents') or (extra == 'group-10-braintrust-test-agentscope' and extra == 'group-10-braintrust-test-langchain') or (extra == 'group-10-braintrust-test-agentscope' and extra == 'group-10-braintrust-test-litellm') or (extra == 'group-10-braintrust-test-agentscope' and extra == 'group-10-braintrust-test-livekit-agents') or (extra == 'group-10-braintrust-test-agentscope' and extra == 'group-10-braintrust-test-openai-agents') or (extra == 'group-10-braintrust-test-agentscope' and extra == 'group-10-braintrust-test-strands') or (extra == 'group-10-braintrust-test-agno' and extra == 'group-10-braintrust-test-crewai') or (extra == 'group-10-braintrust-test-agno' and extra == 'group-10-braintrust-test-deepagents') or (extra == 'group-10-braintrust-test-agno' and extra == 'group-10-braintrust-test-langchain') or (extra == 'group-10-braintrust-test-agno' and extra == 'group-10-braintrust-test-litellm') or (extra == 'group-10-braintrust-test-agno' and extra == 'group-10-braintrust-test-livekit-agents') or (extra == 'group-10-braintrust-test-agno' and extra == 'group-10-braintrust-test-openai-agents') or (extra == 'group-10-braintrust-test-agno' and extra == 'group-10-braintrust-test-strands') or (extra == 'group-10-braintrust-test-crewai' and extra == 'group-10-braintrust-test-deepagents') or (extra == 'group-10-braintrust-test-crewai' and extra == 'group-10-braintrust-test-langchain') or (extra == 'group-10-braintrust-test-crewai' and extra == 'group-10-braintrust-test-litellm') or (extra == 'group-10-braintrust-test-crewai' and extra == 'group-10-braintrust-test-livekit-agents') or (extra == 'group-10-braintrust-test-crewai' and extra == 'group-10-braintrust-test-openai-agents') or (extra == 'group-10-braintrust-test-crewai' and extra == 'group-10-braintrust-test-strands') or (extra == 'group-10-braintrust-test-deepagents' and extra == 'group-10-braintrust-test-langchain') or (extra == 'group-10-braintrust-test-deepagents' and extra == 'group-10-braintrust-test-litellm') or (extra == 'group-10-braintrust-test-deepagents' and extra == 'group-10-braintrust-test-livekit-agents') or (extra == 'group-10-braintrust-test-deepagents' and extra == 'group-10-braintrust-test-openai-agents') or (extra == 'group-10-braintrust-test-deepagents' and extra == 'group-10-braintrust-test-strands') or (extra == 'group-10-braintrust-test-langchain' and extra == 'group-10-braintrust-test-litellm') or (extra == 'group-10-braintrust-test-langchain' and extra == 'group-10-braintrust-test-livekit-agents') or (extra == 'group-10-braintrust-test-langchain' and extra == 'group-10-braintrust-test-openai-agents') or (extra == 'group-10-braintrust-test-langchain' and extra == 'group-10-braintrust-test-strands') or (extra == 'group-10-braintrust-test-litellm' and extra == 'group-10-braintrust-test-livekit-agents') or (extra == 'group-10-braintrust-test-litellm' and extra == 'group-10-braintrust-test-openai-agents') or (extra == 'group-10-braintrust-test-litellm' and extra == 'group-10-braintrust-test-strands') or (extra == 'group-10-braintrust-test-livekit-agents' and extra == 'group-10-braintrust-test-openai-agents') or (extra == 'group-10-braintrust-test-livekit-agents' and extra == 'group-10-braintrust-test-pydantic-ai-logfire') or (extra == 'group-10-braintrust-test-livekit-agents' and extra == 'group-10-braintrust-test-pydantic-ai-otel-events') or (extra == 'group-10-braintrust-test-livekit-agents' and extra == 'group-10-braintrust-test-strands') or (extra == 'group-10-braintrust-test-openai-agents' and extra == 'group-10-braintrust-test-strands')" }, + { name = "numpy", version = "2.5.3", source = { registry = "https://pypi.org/simple" }, marker = "python_full_version >= '3.12' or (extra == 'group-10-braintrust-lint' and extra == 'group-10-braintrust-test-agentscope') or (extra == 'group-10-braintrust-lint' and extra == 'group-10-braintrust-test-agno') or (extra == 'group-10-braintrust-lint' and extra == 'group-10-braintrust-test-crewai') or (extra == 'group-10-braintrust-lint' and extra == 'group-10-braintrust-test-deepagents') or (extra == 'group-10-braintrust-lint' and extra == 'group-10-braintrust-test-langchain') or (extra == 'group-10-braintrust-lint' and extra == 'group-10-braintrust-test-litellm') or (extra == 'group-10-braintrust-lint' and extra == 'group-10-braintrust-test-livekit-agents') or (extra == 'group-10-braintrust-lint' and extra == 'group-10-braintrust-test-openai-agents') or (extra == 'group-10-braintrust-lint' and extra == 'group-10-braintrust-test-pydantic-ai-logfire') or (extra == 'group-10-braintrust-lint' and extra == 'group-10-braintrust-test-strands') or (extra == 'group-10-braintrust-test-agentscope' and extra == 'group-10-braintrust-test-agno') or (extra == 'group-10-braintrust-test-agentscope' and extra == 'group-10-braintrust-test-crewai') or (extra == 'group-10-braintrust-test-agentscope' and extra == 'group-10-braintrust-test-deepagents') or (extra == 'group-10-braintrust-test-agentscope' and extra == 'group-10-braintrust-test-langchain') or (extra == 'group-10-braintrust-test-agentscope' and extra == 'group-10-braintrust-test-litellm') or (extra == 'group-10-braintrust-test-agentscope' and extra == 'group-10-braintrust-test-livekit-agents') or (extra == 'group-10-braintrust-test-agentscope' and extra == 'group-10-braintrust-test-openai-agents') or (extra == 'group-10-braintrust-test-agentscope' and extra == 'group-10-braintrust-test-strands') or (extra == 'group-10-braintrust-test-agno' and extra == 'group-10-braintrust-test-crewai') or (extra == 'group-10-braintrust-test-agno' and extra == 'group-10-braintrust-test-deepagents') or (extra == 'group-10-braintrust-test-agno' and extra == 'group-10-braintrust-test-langchain') or (extra == 'group-10-braintrust-test-agno' and extra == 'group-10-braintrust-test-litellm') or (extra == 'group-10-braintrust-test-agno' and extra == 'group-10-braintrust-test-livekit-agents') or (extra == 'group-10-braintrust-test-agno' and extra == 'group-10-braintrust-test-openai-agents') or (extra == 'group-10-braintrust-test-agno' and extra == 'group-10-braintrust-test-strands') or (extra == 'group-10-braintrust-test-crewai' and extra == 'group-10-braintrust-test-deepagents') or (extra == 'group-10-braintrust-test-crewai' and extra == 'group-10-braintrust-test-langchain') or (extra == 'group-10-braintrust-test-crewai' and extra == 'group-10-braintrust-test-litellm') or (extra == 'group-10-braintrust-test-crewai' and extra == 'group-10-braintrust-test-livekit-agents') or (extra == 'group-10-braintrust-test-crewai' and extra == 'group-10-braintrust-test-openai-agents') or (extra == 'group-10-braintrust-test-crewai' and extra == 'group-10-braintrust-test-strands') or (extra == 'group-10-braintrust-test-deepagents' and extra == 'group-10-braintrust-test-langchain') or (extra == 'group-10-braintrust-test-deepagents' and extra == 'group-10-braintrust-test-litellm') or (extra == 'group-10-braintrust-test-deepagents' and extra == 'group-10-braintrust-test-livekit-agents') or (extra == 'group-10-braintrust-test-deepagents' and extra == 'group-10-braintrust-test-openai-agents') or (extra == 'group-10-braintrust-test-deepagents' and extra == 'group-10-braintrust-test-strands') or (extra == 'group-10-braintrust-test-langchain' and extra == 'group-10-braintrust-test-litellm') or (extra == 'group-10-braintrust-test-langchain' and extra == 'group-10-braintrust-test-livekit-agents') or (extra == 'group-10-braintrust-test-langchain' and extra == 'group-10-braintrust-test-openai-agents') or (extra == 'group-10-braintrust-test-langchain' and extra == 'group-10-braintrust-test-strands') or (extra == 'group-10-braintrust-test-litellm' and extra == 'group-10-braintrust-test-livekit-agents') or (extra == 'group-10-braintrust-test-litellm' and extra == 'group-10-braintrust-test-openai-agents') or (extra == 'group-10-braintrust-test-litellm' and extra == 'group-10-braintrust-test-strands') or (extra == 'group-10-braintrust-test-livekit-agents' and extra == 'group-10-braintrust-test-openai-agents') or (extra == 'group-10-braintrust-test-livekit-agents' and extra == 'group-10-braintrust-test-pydantic-ai-logfire') or (extra == 'group-10-braintrust-test-livekit-agents' and extra == 'group-10-braintrust-test-pydantic-ai-otel-events') or (extra == 'group-10-braintrust-test-livekit-agents' and extra == 'group-10-braintrust-test-strands') or (extra == 'group-10-braintrust-test-openai-agents' and extra == 'group-10-braintrust-test-strands')" }, + { name = "soundfile" }, +] cli = [ { name = "boto3" }, { name = "python-dotenv" }, @@ -912,6 +918,15 @@ test-agno = [ { name = "pytest-asyncio" }, { name = "pytest-vcr" }, ] +test-audio = [ + { name = "numpy", version = "2.2.6", source = { registry = "https://pypi.org/simple" }, marker = "python_full_version < '3.11' or (extra == 'group-10-braintrust-lint' and extra == 'group-10-braintrust-test-agentscope') or (extra == 'group-10-braintrust-lint' and extra == 'group-10-braintrust-test-agno') or (extra == 'group-10-braintrust-lint' and extra == 'group-10-braintrust-test-crewai') or (extra == 'group-10-braintrust-lint' and extra == 'group-10-braintrust-test-deepagents') or (extra == 'group-10-braintrust-lint' and extra == 'group-10-braintrust-test-langchain') or (extra == 'group-10-braintrust-lint' and extra == 'group-10-braintrust-test-litellm') or (extra == 'group-10-braintrust-lint' and extra == 'group-10-braintrust-test-livekit-agents') or (extra == 'group-10-braintrust-lint' and extra == 'group-10-braintrust-test-openai-agents') or (extra == 'group-10-braintrust-lint' and extra == 'group-10-braintrust-test-pydantic-ai-logfire') or (extra == 'group-10-braintrust-lint' and extra == 'group-10-braintrust-test-strands') or (extra == 'group-10-braintrust-test-agentscope' and extra == 'group-10-braintrust-test-agno') or (extra == 'group-10-braintrust-test-agentscope' and extra == 'group-10-braintrust-test-crewai') or (extra == 'group-10-braintrust-test-agentscope' and extra == 'group-10-braintrust-test-deepagents') or (extra == 'group-10-braintrust-test-agentscope' and extra == 'group-10-braintrust-test-langchain') or (extra == 'group-10-braintrust-test-agentscope' and extra == 'group-10-braintrust-test-litellm') or (extra == 'group-10-braintrust-test-agentscope' and extra == 'group-10-braintrust-test-livekit-agents') or (extra == 'group-10-braintrust-test-agentscope' and extra == 'group-10-braintrust-test-openai-agents') or (extra == 'group-10-braintrust-test-agentscope' and extra == 'group-10-braintrust-test-strands') or (extra == 'group-10-braintrust-test-agno' and extra == 'group-10-braintrust-test-crewai') or (extra == 'group-10-braintrust-test-agno' and extra == 'group-10-braintrust-test-deepagents') or (extra == 'group-10-braintrust-test-agno' and extra == 'group-10-braintrust-test-langchain') or (extra == 'group-10-braintrust-test-agno' and extra == 'group-10-braintrust-test-litellm') or (extra == 'group-10-braintrust-test-agno' and extra == 'group-10-braintrust-test-livekit-agents') or (extra == 'group-10-braintrust-test-agno' and extra == 'group-10-braintrust-test-openai-agents') or (extra == 'group-10-braintrust-test-agno' and extra == 'group-10-braintrust-test-strands') or (extra == 'group-10-braintrust-test-crewai' and extra == 'group-10-braintrust-test-deepagents') or (extra == 'group-10-braintrust-test-crewai' and extra == 'group-10-braintrust-test-langchain') or (extra == 'group-10-braintrust-test-crewai' and extra == 'group-10-braintrust-test-litellm') or (extra == 'group-10-braintrust-test-crewai' and extra == 'group-10-braintrust-test-livekit-agents') or (extra == 'group-10-braintrust-test-crewai' and extra == 'group-10-braintrust-test-openai-agents') or (extra == 'group-10-braintrust-test-crewai' and extra == 'group-10-braintrust-test-strands') or (extra == 'group-10-braintrust-test-deepagents' and extra == 'group-10-braintrust-test-langchain') or (extra == 'group-10-braintrust-test-deepagents' and extra == 'group-10-braintrust-test-litellm') or (extra == 'group-10-braintrust-test-deepagents' and extra == 'group-10-braintrust-test-livekit-agents') or (extra == 'group-10-braintrust-test-deepagents' and extra == 'group-10-braintrust-test-openai-agents') or (extra == 'group-10-braintrust-test-deepagents' and extra == 'group-10-braintrust-test-strands') or (extra == 'group-10-braintrust-test-langchain' and extra == 'group-10-braintrust-test-litellm') or (extra == 'group-10-braintrust-test-langchain' and extra == 'group-10-braintrust-test-livekit-agents') or (extra == 'group-10-braintrust-test-langchain' and extra == 'group-10-braintrust-test-openai-agents') or (extra == 'group-10-braintrust-test-langchain' and extra == 'group-10-braintrust-test-strands') or (extra == 'group-10-braintrust-test-litellm' and extra == 'group-10-braintrust-test-livekit-agents') or (extra == 'group-10-braintrust-test-litellm' and extra == 'group-10-braintrust-test-openai-agents') or (extra == 'group-10-braintrust-test-litellm' and extra == 'group-10-braintrust-test-strands') or (extra == 'group-10-braintrust-test-livekit-agents' and extra == 'group-10-braintrust-test-openai-agents') or (extra == 'group-10-braintrust-test-livekit-agents' and extra == 'group-10-braintrust-test-pydantic-ai-logfire') or (extra == 'group-10-braintrust-test-livekit-agents' and extra == 'group-10-braintrust-test-pydantic-ai-otel-events') or (extra == 'group-10-braintrust-test-livekit-agents' and extra == 'group-10-braintrust-test-strands') or (extra == 'group-10-braintrust-test-openai-agents' and extra == 'group-10-braintrust-test-strands')" }, + { name = "numpy", version = "2.4.6", source = { registry = "https://pypi.org/simple" }, marker = "python_full_version == '3.11.*' or (extra == 'group-10-braintrust-lint' and extra == 'group-10-braintrust-test-agentscope') or (extra == 'group-10-braintrust-lint' and extra == 'group-10-braintrust-test-agno') or (extra == 'group-10-braintrust-lint' and extra == 'group-10-braintrust-test-crewai') or (extra == 'group-10-braintrust-lint' and extra == 'group-10-braintrust-test-deepagents') or (extra == 'group-10-braintrust-lint' and extra == 'group-10-braintrust-test-langchain') or (extra == 'group-10-braintrust-lint' and extra == 'group-10-braintrust-test-litellm') or (extra == 'group-10-braintrust-lint' and extra == 'group-10-braintrust-test-livekit-agents') or (extra == 'group-10-braintrust-lint' and extra == 'group-10-braintrust-test-openai-agents') or (extra == 'group-10-braintrust-lint' and extra == 'group-10-braintrust-test-pydantic-ai-logfire') or (extra == 'group-10-braintrust-lint' and extra == 'group-10-braintrust-test-strands') or (extra == 'group-10-braintrust-test-agentscope' and extra == 'group-10-braintrust-test-agno') or (extra == 'group-10-braintrust-test-agentscope' and extra == 'group-10-braintrust-test-crewai') or (extra == 'group-10-braintrust-test-agentscope' and extra == 'group-10-braintrust-test-deepagents') or (extra == 'group-10-braintrust-test-agentscope' and extra == 'group-10-braintrust-test-langchain') or (extra == 'group-10-braintrust-test-agentscope' and extra == 'group-10-braintrust-test-litellm') or (extra == 'group-10-braintrust-test-agentscope' and extra == 'group-10-braintrust-test-livekit-agents') or (extra == 'group-10-braintrust-test-agentscope' and extra == 'group-10-braintrust-test-openai-agents') or (extra == 'group-10-braintrust-test-agentscope' and extra == 'group-10-braintrust-test-strands') or (extra == 'group-10-braintrust-test-agno' and extra == 'group-10-braintrust-test-crewai') or (extra == 'group-10-braintrust-test-agno' and extra == 'group-10-braintrust-test-deepagents') or (extra == 'group-10-braintrust-test-agno' and extra == 'group-10-braintrust-test-langchain') or (extra == 'group-10-braintrust-test-agno' and extra == 'group-10-braintrust-test-litellm') or (extra == 'group-10-braintrust-test-agno' and extra == 'group-10-braintrust-test-livekit-agents') or (extra == 'group-10-braintrust-test-agno' and extra == 'group-10-braintrust-test-openai-agents') or (extra == 'group-10-braintrust-test-agno' and extra == 'group-10-braintrust-test-strands') or (extra == 'group-10-braintrust-test-crewai' and extra == 'group-10-braintrust-test-deepagents') or (extra == 'group-10-braintrust-test-crewai' and extra == 'group-10-braintrust-test-langchain') or (extra == 'group-10-braintrust-test-crewai' and extra == 'group-10-braintrust-test-litellm') or (extra == 'group-10-braintrust-test-crewai' and extra == 'group-10-braintrust-test-livekit-agents') or (extra == 'group-10-braintrust-test-crewai' and extra == 'group-10-braintrust-test-openai-agents') or (extra == 'group-10-braintrust-test-crewai' and extra == 'group-10-braintrust-test-strands') or (extra == 'group-10-braintrust-test-deepagents' and extra == 'group-10-braintrust-test-langchain') or (extra == 'group-10-braintrust-test-deepagents' and extra == 'group-10-braintrust-test-litellm') or (extra == 'group-10-braintrust-test-deepagents' and extra == 'group-10-braintrust-test-livekit-agents') or (extra == 'group-10-braintrust-test-deepagents' and extra == 'group-10-braintrust-test-openai-agents') or (extra == 'group-10-braintrust-test-deepagents' and extra == 'group-10-braintrust-test-strands') or (extra == 'group-10-braintrust-test-langchain' and extra == 'group-10-braintrust-test-litellm') or (extra == 'group-10-braintrust-test-langchain' and extra == 'group-10-braintrust-test-livekit-agents') or (extra == 'group-10-braintrust-test-langchain' and extra == 'group-10-braintrust-test-openai-agents') or (extra == 'group-10-braintrust-test-langchain' and extra == 'group-10-braintrust-test-strands') or (extra == 'group-10-braintrust-test-litellm' and extra == 'group-10-braintrust-test-livekit-agents') or (extra == 'group-10-braintrust-test-litellm' and extra == 'group-10-braintrust-test-openai-agents') or (extra == 'group-10-braintrust-test-litellm' and extra == 'group-10-braintrust-test-strands') or (extra == 'group-10-braintrust-test-livekit-agents' and extra == 'group-10-braintrust-test-openai-agents') or (extra == 'group-10-braintrust-test-livekit-agents' and extra == 'group-10-braintrust-test-pydantic-ai-logfire') or (extra == 'group-10-braintrust-test-livekit-agents' and extra == 'group-10-braintrust-test-pydantic-ai-otel-events') or (extra == 'group-10-braintrust-test-livekit-agents' and extra == 'group-10-braintrust-test-strands') or (extra == 'group-10-braintrust-test-openai-agents' and extra == 'group-10-braintrust-test-strands')" }, + { name = "numpy", version = "2.5.3", source = { registry = "https://pypi.org/simple" }, marker = "python_full_version >= '3.12' or (extra == 'group-10-braintrust-lint' and extra == 'group-10-braintrust-test-agentscope') or (extra == 'group-10-braintrust-lint' and extra == 'group-10-braintrust-test-agno') or (extra == 'group-10-braintrust-lint' and extra == 'group-10-braintrust-test-crewai') or (extra == 'group-10-braintrust-lint' and extra == 'group-10-braintrust-test-deepagents') or (extra == 'group-10-braintrust-lint' and extra == 'group-10-braintrust-test-langchain') or (extra == 'group-10-braintrust-lint' and extra == 'group-10-braintrust-test-litellm') or (extra == 'group-10-braintrust-lint' and extra == 'group-10-braintrust-test-livekit-agents') or (extra == 'group-10-braintrust-lint' and extra == 'group-10-braintrust-test-openai-agents') or (extra == 'group-10-braintrust-lint' and extra == 'group-10-braintrust-test-pydantic-ai-logfire') or (extra == 'group-10-braintrust-lint' and extra == 'group-10-braintrust-test-strands') or (extra == 'group-10-braintrust-test-agentscope' and extra == 'group-10-braintrust-test-agno') or (extra == 'group-10-braintrust-test-agentscope' and extra == 'group-10-braintrust-test-crewai') or (extra == 'group-10-braintrust-test-agentscope' and extra == 'group-10-braintrust-test-deepagents') or (extra == 'group-10-braintrust-test-agentscope' and extra == 'group-10-braintrust-test-langchain') or (extra == 'group-10-braintrust-test-agentscope' and extra == 'group-10-braintrust-test-litellm') or (extra == 'group-10-braintrust-test-agentscope' and extra == 'group-10-braintrust-test-livekit-agents') or (extra == 'group-10-braintrust-test-agentscope' and extra == 'group-10-braintrust-test-openai-agents') or (extra == 'group-10-braintrust-test-agentscope' and extra == 'group-10-braintrust-test-strands') or (extra == 'group-10-braintrust-test-agno' and extra == 'group-10-braintrust-test-crewai') or (extra == 'group-10-braintrust-test-agno' and extra == 'group-10-braintrust-test-deepagents') or (extra == 'group-10-braintrust-test-agno' and extra == 'group-10-braintrust-test-langchain') or (extra == 'group-10-braintrust-test-agno' and extra == 'group-10-braintrust-test-litellm') or (extra == 'group-10-braintrust-test-agno' and extra == 'group-10-braintrust-test-livekit-agents') or (extra == 'group-10-braintrust-test-agno' and extra == 'group-10-braintrust-test-openai-agents') or (extra == 'group-10-braintrust-test-agno' and extra == 'group-10-braintrust-test-strands') or (extra == 'group-10-braintrust-test-crewai' and extra == 'group-10-braintrust-test-deepagents') or (extra == 'group-10-braintrust-test-crewai' and extra == 'group-10-braintrust-test-langchain') or (extra == 'group-10-braintrust-test-crewai' and extra == 'group-10-braintrust-test-litellm') or (extra == 'group-10-braintrust-test-crewai' and extra == 'group-10-braintrust-test-livekit-agents') or (extra == 'group-10-braintrust-test-crewai' and extra == 'group-10-braintrust-test-openai-agents') or (extra == 'group-10-braintrust-test-crewai' and extra == 'group-10-braintrust-test-strands') or (extra == 'group-10-braintrust-test-deepagents' and extra == 'group-10-braintrust-test-langchain') or (extra == 'group-10-braintrust-test-deepagents' and extra == 'group-10-braintrust-test-litellm') or (extra == 'group-10-braintrust-test-deepagents' and extra == 'group-10-braintrust-test-livekit-agents') or (extra == 'group-10-braintrust-test-deepagents' and extra == 'group-10-braintrust-test-openai-agents') or (extra == 'group-10-braintrust-test-deepagents' and extra == 'group-10-braintrust-test-strands') or (extra == 'group-10-braintrust-test-langchain' and extra == 'group-10-braintrust-test-litellm') or (extra == 'group-10-braintrust-test-langchain' and extra == 'group-10-braintrust-test-livekit-agents') or (extra == 'group-10-braintrust-test-langchain' and extra == 'group-10-braintrust-test-openai-agents') or (extra == 'group-10-braintrust-test-langchain' and extra == 'group-10-braintrust-test-strands') or (extra == 'group-10-braintrust-test-litellm' and extra == 'group-10-braintrust-test-livekit-agents') or (extra == 'group-10-braintrust-test-litellm' and extra == 'group-10-braintrust-test-openai-agents') or (extra == 'group-10-braintrust-test-litellm' and extra == 'group-10-braintrust-test-strands') or (extra == 'group-10-braintrust-test-livekit-agents' and extra == 'group-10-braintrust-test-openai-agents') or (extra == 'group-10-braintrust-test-livekit-agents' and extra == 'group-10-braintrust-test-pydantic-ai-logfire') or (extra == 'group-10-braintrust-test-livekit-agents' and extra == 'group-10-braintrust-test-pydantic-ai-otel-events') or (extra == 'group-10-braintrust-test-livekit-agents' and extra == 'group-10-braintrust-test-strands') or (extra == 'group-10-braintrust-test-openai-agents' and extra == 'group-10-braintrust-test-strands')" }, + { name = "pytest" }, + { name = "pytest-asyncio" }, + { name = "pytest-vcr" }, + { name = "soundfile" }, +] test-cli = [ { name = "httpx" }, { name = "pytest" }, @@ -994,6 +1009,7 @@ test-pipecat = [ { name = "pytest" }, { name = "pytest-asyncio" }, { name = "pytest-vcr" }, + { name = "soundfile" }, { name = "websockets" }, ] test-pydantic-ai-logfire = [ @@ -1039,6 +1055,7 @@ requires-dist = [ { name = "chevron" }, { name = "exceptiongroup", specifier = ">=1.2.0" }, { name = "jsonschema" }, + { name = "numpy", marker = "extra == 'audio'", specifier = ">=1.26" }, { name = "openai-agents", marker = "extra == 'all'" }, { name = "openai-agents", marker = "extra == 'openai-agents'" }, { name = "opentelemetry-api", marker = "extra == 'all'" }, @@ -1054,6 +1071,7 @@ requires-dist = [ { name = "python-dotenv", marker = "extra == 'cli'" }, { name = "python-slugify" }, { name = "requests" }, + { name = "soundfile", marker = "extra == 'audio'", specifier = ">=0.13.1" }, { name = "sseclient-py" }, { name = "starlette", marker = "extra == 'all'" }, { name = "starlette", marker = "extra == 'cli'" }, @@ -1067,7 +1085,7 @@ requires-dist = [ { name = "uvicorn", marker = "extra == 'cli'" }, { name = "wrapt" }, ] -provides-extras = ["cli", "doc", "openai-agents", "otel", "performance", "temporal", "all"] +provides-extras = ["audio", "cli", "doc", "openai-agents", "otel", "performance", "temporal", "all"] [package.metadata.requires-dev] api-codegen = [ @@ -1146,6 +1164,13 @@ test-agno = [ { name = "pytest-asyncio", specifier = "==1.3.0" }, { name = "pytest-vcr", specifier = "==1.0.2" }, ] +test-audio = [ + { name = "numpy", specifier = ">=1.26" }, + { name = "pytest", specifier = "==9.1.1" }, + { name = "pytest-asyncio", specifier = "==1.3.0" }, + { name = "pytest-vcr", specifier = "==1.0.2" }, + { name = "soundfile", specifier = ">=0.13.1" }, +] test-cli = [ { name = "httpx", specifier = "==0.28.1" }, { name = "pytest", specifier = "==9.1.1" }, @@ -1224,6 +1249,7 @@ test-pipecat = [ { name = "pytest", specifier = "==9.1.1" }, { name = "pytest-asyncio", specifier = "==1.3.0" }, { name = "pytest-vcr", specifier = "==1.0.2" }, + { name = "soundfile", specifier = ">=0.13.1" }, { name = "websockets", specifier = "==15.0.1" }, ] test-pydantic-ai-logfire = [ @@ -7899,6 +7925,29 @@ wheels = [ { url = "https://files.pythonhosted.org/packages/60/a4/b0c21c9f215a6fd9606b8f8748c21212dc098e5d5a2d93068c50edcf19b4/sounddevice-0.5.6-py3-none-win_arm64.whl", hash = "sha256:c8ae19173e5f27f8c12d4b5eee2dbfe542cee125d591e663e0fb4dfb75246d45", size = 1009630, upload-time = "2026-08-17T07:55:03.689Z" }, ] +[[package]] +name = "soundfile" +version = "0.14.0" +source = { registry = "https://pypi.org/simple" } +dependencies = [ + { name = "cffi" }, + { name = "numpy", version = "2.2.6", source = { registry = "https://pypi.org/simple" }, marker = "python_full_version < '3.11' or (extra == 'group-10-braintrust-lint' and extra == 'group-10-braintrust-test-agentscope') or (extra == 'group-10-braintrust-lint' and extra == 'group-10-braintrust-test-agno') or (extra == 'group-10-braintrust-lint' and extra == 'group-10-braintrust-test-crewai') or (extra == 'group-10-braintrust-lint' and extra == 'group-10-braintrust-test-deepagents') or (extra == 'group-10-braintrust-lint' and extra == 'group-10-braintrust-test-langchain') or (extra == 'group-10-braintrust-lint' and extra == 'group-10-braintrust-test-litellm') or (extra == 'group-10-braintrust-lint' and extra == 'group-10-braintrust-test-livekit-agents') or (extra == 'group-10-braintrust-lint' and extra == 'group-10-braintrust-test-openai-agents') or (extra == 'group-10-braintrust-lint' and extra == 'group-10-braintrust-test-pydantic-ai-logfire') or (extra == 'group-10-braintrust-lint' and extra == 'group-10-braintrust-test-strands') or (extra == 'group-10-braintrust-test-agentscope' and extra == 'group-10-braintrust-test-agno') or (extra == 'group-10-braintrust-test-agentscope' and extra == 'group-10-braintrust-test-crewai') or (extra == 'group-10-braintrust-test-agentscope' and extra == 'group-10-braintrust-test-deepagents') or (extra == 'group-10-braintrust-test-agentscope' and extra == 'group-10-braintrust-test-langchain') or (extra == 'group-10-braintrust-test-agentscope' and extra == 'group-10-braintrust-test-litellm') or (extra == 'group-10-braintrust-test-agentscope' and extra == 'group-10-braintrust-test-livekit-agents') or (extra == 'group-10-braintrust-test-agentscope' and extra == 'group-10-braintrust-test-openai-agents') or (extra == 'group-10-braintrust-test-agentscope' and extra == 'group-10-braintrust-test-strands') or (extra == 'group-10-braintrust-test-agno' and extra == 'group-10-braintrust-test-crewai') or (extra == 'group-10-braintrust-test-agno' and extra == 'group-10-braintrust-test-deepagents') or (extra == 'group-10-braintrust-test-agno' and extra == 'group-10-braintrust-test-langchain') or (extra == 'group-10-braintrust-test-agno' and extra == 'group-10-braintrust-test-litellm') or (extra == 'group-10-braintrust-test-agno' and extra == 'group-10-braintrust-test-livekit-agents') or (extra == 'group-10-braintrust-test-agno' and extra == 'group-10-braintrust-test-openai-agents') or (extra == 'group-10-braintrust-test-agno' and extra == 'group-10-braintrust-test-strands') or (extra == 'group-10-braintrust-test-crewai' and extra == 'group-10-braintrust-test-deepagents') or (extra == 'group-10-braintrust-test-crewai' and extra == 'group-10-braintrust-test-langchain') or (extra == 'group-10-braintrust-test-crewai' and extra == 'group-10-braintrust-test-litellm') or (extra == 'group-10-braintrust-test-crewai' and extra == 'group-10-braintrust-test-livekit-agents') or (extra == 'group-10-braintrust-test-crewai' and extra == 'group-10-braintrust-test-openai-agents') or (extra == 'group-10-braintrust-test-crewai' and extra == 'group-10-braintrust-test-strands') or (extra == 'group-10-braintrust-test-deepagents' and extra == 'group-10-braintrust-test-langchain') or (extra == 'group-10-braintrust-test-deepagents' and extra == 'group-10-braintrust-test-litellm') or (extra == 'group-10-braintrust-test-deepagents' and extra == 'group-10-braintrust-test-livekit-agents') or (extra == 'group-10-braintrust-test-deepagents' and extra == 'group-10-braintrust-test-openai-agents') or (extra == 'group-10-braintrust-test-deepagents' and extra == 'group-10-braintrust-test-strands') or (extra == 'group-10-braintrust-test-langchain' and extra == 'group-10-braintrust-test-litellm') or (extra == 'group-10-braintrust-test-langchain' and extra == 'group-10-braintrust-test-livekit-agents') or (extra == 'group-10-braintrust-test-langchain' and extra == 'group-10-braintrust-test-openai-agents') or (extra == 'group-10-braintrust-test-langchain' and extra == 'group-10-braintrust-test-strands') or (extra == 'group-10-braintrust-test-litellm' and extra == 'group-10-braintrust-test-livekit-agents') or (extra == 'group-10-braintrust-test-litellm' and extra == 'group-10-braintrust-test-openai-agents') or (extra == 'group-10-braintrust-test-litellm' and extra == 'group-10-braintrust-test-strands') or (extra == 'group-10-braintrust-test-livekit-agents' and extra == 'group-10-braintrust-test-openai-agents') or (extra == 'group-10-braintrust-test-livekit-agents' and extra == 'group-10-braintrust-test-pydantic-ai-logfire') or (extra == 'group-10-braintrust-test-livekit-agents' and extra == 'group-10-braintrust-test-pydantic-ai-otel-events') or (extra == 'group-10-braintrust-test-livekit-agents' and extra == 'group-10-braintrust-test-strands') or (extra == 'group-10-braintrust-test-openai-agents' and extra == 'group-10-braintrust-test-strands')" }, + { name = "numpy", version = "2.4.6", source = { registry = "https://pypi.org/simple" }, marker = "python_full_version == '3.11.*' or (extra == 'group-10-braintrust-lint' and extra == 'group-10-braintrust-test-agentscope') or (extra == 'group-10-braintrust-lint' and extra == 'group-10-braintrust-test-agno') or (extra == 'group-10-braintrust-lint' and extra == 'group-10-braintrust-test-crewai') or (extra == 'group-10-braintrust-lint' and extra == 'group-10-braintrust-test-deepagents') or (extra == 'group-10-braintrust-lint' and extra == 'group-10-braintrust-test-langchain') or (extra == 'group-10-braintrust-lint' and extra == 'group-10-braintrust-test-litellm') or (extra == 'group-10-braintrust-lint' and extra == 'group-10-braintrust-test-livekit-agents') or (extra == 'group-10-braintrust-lint' and extra == 'group-10-braintrust-test-openai-agents') or (extra == 'group-10-braintrust-lint' and extra == 'group-10-braintrust-test-pydantic-ai-logfire') or (extra == 'group-10-braintrust-lint' and extra == 'group-10-braintrust-test-strands') or (extra == 'group-10-braintrust-test-agentscope' and extra == 'group-10-braintrust-test-agno') or (extra == 'group-10-braintrust-test-agentscope' and extra == 'group-10-braintrust-test-crewai') or (extra == 'group-10-braintrust-test-agentscope' and extra == 'group-10-braintrust-test-deepagents') or (extra == 'group-10-braintrust-test-agentscope' and extra == 'group-10-braintrust-test-langchain') or (extra == 'group-10-braintrust-test-agentscope' and extra == 'group-10-braintrust-test-litellm') or (extra == 'group-10-braintrust-test-agentscope' and extra == 'group-10-braintrust-test-livekit-agents') or (extra == 'group-10-braintrust-test-agentscope' and extra == 'group-10-braintrust-test-openai-agents') or (extra == 'group-10-braintrust-test-agentscope' and extra == 'group-10-braintrust-test-strands') or (extra == 'group-10-braintrust-test-agno' and extra == 'group-10-braintrust-test-crewai') or (extra == 'group-10-braintrust-test-agno' and extra == 'group-10-braintrust-test-deepagents') or (extra == 'group-10-braintrust-test-agno' and extra == 'group-10-braintrust-test-langchain') or (extra == 'group-10-braintrust-test-agno' and extra == 'group-10-braintrust-test-litellm') or (extra == 'group-10-braintrust-test-agno' and extra == 'group-10-braintrust-test-livekit-agents') or (extra == 'group-10-braintrust-test-agno' and extra == 'group-10-braintrust-test-openai-agents') or (extra == 'group-10-braintrust-test-agno' and extra == 'group-10-braintrust-test-strands') or (extra == 'group-10-braintrust-test-crewai' and extra == 'group-10-braintrust-test-deepagents') or (extra == 'group-10-braintrust-test-crewai' and extra == 'group-10-braintrust-test-langchain') or (extra == 'group-10-braintrust-test-crewai' and extra == 'group-10-braintrust-test-litellm') or (extra == 'group-10-braintrust-test-crewai' and extra == 'group-10-braintrust-test-livekit-agents') or (extra == 'group-10-braintrust-test-crewai' and extra == 'group-10-braintrust-test-openai-agents') or (extra == 'group-10-braintrust-test-crewai' and extra == 'group-10-braintrust-test-strands') or (extra == 'group-10-braintrust-test-deepagents' and extra == 'group-10-braintrust-test-langchain') or (extra == 'group-10-braintrust-test-deepagents' and extra == 'group-10-braintrust-test-litellm') or (extra == 'group-10-braintrust-test-deepagents' and extra == 'group-10-braintrust-test-livekit-agents') or (extra == 'group-10-braintrust-test-deepagents' and extra == 'group-10-braintrust-test-openai-agents') or (extra == 'group-10-braintrust-test-deepagents' and extra == 'group-10-braintrust-test-strands') or (extra == 'group-10-braintrust-test-langchain' and extra == 'group-10-braintrust-test-litellm') or (extra == 'group-10-braintrust-test-langchain' and extra == 'group-10-braintrust-test-livekit-agents') or (extra == 'group-10-braintrust-test-langchain' and extra == 'group-10-braintrust-test-openai-agents') or (extra == 'group-10-braintrust-test-langchain' and extra == 'group-10-braintrust-test-strands') or (extra == 'group-10-braintrust-test-litellm' and extra == 'group-10-braintrust-test-livekit-agents') or (extra == 'group-10-braintrust-test-litellm' and extra == 'group-10-braintrust-test-openai-agents') or (extra == 'group-10-braintrust-test-litellm' and extra == 'group-10-braintrust-test-strands') or (extra == 'group-10-braintrust-test-livekit-agents' and extra == 'group-10-braintrust-test-openai-agents') or (extra == 'group-10-braintrust-test-livekit-agents' and extra == 'group-10-braintrust-test-pydantic-ai-logfire') or (extra == 'group-10-braintrust-test-livekit-agents' and extra == 'group-10-braintrust-test-pydantic-ai-otel-events') or (extra == 'group-10-braintrust-test-livekit-agents' and extra == 'group-10-braintrust-test-strands') or (extra == 'group-10-braintrust-test-openai-agents' and extra == 'group-10-braintrust-test-strands')" }, + { name = "numpy", version = "2.5.3", source = { registry = "https://pypi.org/simple" }, marker = "python_full_version >= '3.12' or (extra == 'group-10-braintrust-lint' and extra == 'group-10-braintrust-test-agentscope') or (extra == 'group-10-braintrust-lint' and extra == 'group-10-braintrust-test-agno') or (extra == 'group-10-braintrust-lint' and extra == 'group-10-braintrust-test-crewai') or (extra == 'group-10-braintrust-lint' and extra == 'group-10-braintrust-test-deepagents') or (extra == 'group-10-braintrust-lint' and extra == 'group-10-braintrust-test-langchain') or (extra == 'group-10-braintrust-lint' and extra == 'group-10-braintrust-test-litellm') or (extra == 'group-10-braintrust-lint' and extra == 'group-10-braintrust-test-livekit-agents') or (extra == 'group-10-braintrust-lint' and extra == 'group-10-braintrust-test-openai-agents') or (extra == 'group-10-braintrust-lint' and extra == 'group-10-braintrust-test-pydantic-ai-logfire') or (extra == 'group-10-braintrust-lint' and extra == 'group-10-braintrust-test-strands') or (extra == 'group-10-braintrust-test-agentscope' and extra == 'group-10-braintrust-test-agno') or (extra == 'group-10-braintrust-test-agentscope' and extra == 'group-10-braintrust-test-crewai') or (extra == 'group-10-braintrust-test-agentscope' and extra == 'group-10-braintrust-test-deepagents') or (extra == 'group-10-braintrust-test-agentscope' and extra == 'group-10-braintrust-test-langchain') or (extra == 'group-10-braintrust-test-agentscope' and extra == 'group-10-braintrust-test-litellm') or (extra == 'group-10-braintrust-test-agentscope' and extra == 'group-10-braintrust-test-livekit-agents') or (extra == 'group-10-braintrust-test-agentscope' and extra == 'group-10-braintrust-test-openai-agents') or (extra == 'group-10-braintrust-test-agentscope' and extra == 'group-10-braintrust-test-strands') or (extra == 'group-10-braintrust-test-agno' and extra == 'group-10-braintrust-test-crewai') or (extra == 'group-10-braintrust-test-agno' and extra == 'group-10-braintrust-test-deepagents') or (extra == 'group-10-braintrust-test-agno' and extra == 'group-10-braintrust-test-langchain') or (extra == 'group-10-braintrust-test-agno' and extra == 'group-10-braintrust-test-litellm') or (extra == 'group-10-braintrust-test-agno' and extra == 'group-10-braintrust-test-livekit-agents') or (extra == 'group-10-braintrust-test-agno' and extra == 'group-10-braintrust-test-openai-agents') or (extra == 'group-10-braintrust-test-agno' and extra == 'group-10-braintrust-test-strands') or (extra == 'group-10-braintrust-test-crewai' and extra == 'group-10-braintrust-test-deepagents') or (extra == 'group-10-braintrust-test-crewai' and extra == 'group-10-braintrust-test-langchain') or (extra == 'group-10-braintrust-test-crewai' and extra == 'group-10-braintrust-test-litellm') or (extra == 'group-10-braintrust-test-crewai' and extra == 'group-10-braintrust-test-livekit-agents') or (extra == 'group-10-braintrust-test-crewai' and extra == 'group-10-braintrust-test-openai-agents') or (extra == 'group-10-braintrust-test-crewai' and extra == 'group-10-braintrust-test-strands') or (extra == 'group-10-braintrust-test-deepagents' and extra == 'group-10-braintrust-test-langchain') or (extra == 'group-10-braintrust-test-deepagents' and extra == 'group-10-braintrust-test-litellm') or (extra == 'group-10-braintrust-test-deepagents' and extra == 'group-10-braintrust-test-livekit-agents') or (extra == 'group-10-braintrust-test-deepagents' and extra == 'group-10-braintrust-test-openai-agents') or (extra == 'group-10-braintrust-test-deepagents' and extra == 'group-10-braintrust-test-strands') or (extra == 'group-10-braintrust-test-langchain' and extra == 'group-10-braintrust-test-litellm') or (extra == 'group-10-braintrust-test-langchain' and extra == 'group-10-braintrust-test-livekit-agents') or (extra == 'group-10-braintrust-test-langchain' and extra == 'group-10-braintrust-test-openai-agents') or (extra == 'group-10-braintrust-test-langchain' and extra == 'group-10-braintrust-test-strands') or (extra == 'group-10-braintrust-test-litellm' and extra == 'group-10-braintrust-test-livekit-agents') or (extra == 'group-10-braintrust-test-litellm' and extra == 'group-10-braintrust-test-openai-agents') or (extra == 'group-10-braintrust-test-litellm' and extra == 'group-10-braintrust-test-strands') or (extra == 'group-10-braintrust-test-livekit-agents' and extra == 'group-10-braintrust-test-openai-agents') or (extra == 'group-10-braintrust-test-livekit-agents' and extra == 'group-10-braintrust-test-pydantic-ai-logfire') or (extra == 'group-10-braintrust-test-livekit-agents' and extra == 'group-10-braintrust-test-pydantic-ai-otel-events') or (extra == 'group-10-braintrust-test-livekit-agents' and extra == 'group-10-braintrust-test-strands') or (extra == 'group-10-braintrust-test-openai-agents' and extra == 'group-10-braintrust-test-strands')" }, + { name = "typing-extensions" }, +] +sdist = { url = "https://files.pythonhosted.org/packages/d2/db/949331952a6fb1c5b12e9de80fd08747966c2039d1a61db4764fbd3981c2/soundfile-0.14.0.tar.gz", hash = "sha256:ba1c1a2d618bca5c406647c83b89f07cc8810fa506a50622a6993ba130c1de11", size = 47842, upload-time = "2026-06-06T08:58:47.869Z" } +wheels = [ + { url = "https://files.pythonhosted.org/packages/b1/d1/5e338af9ca6ed0786cd5bb03f6d60de1c325728c1189014f3b59aae7403c/soundfile-0.14.0-py2.py3-none-any.whl", hash = "sha256:8ba81ae3a89fd5ab3bef8a8eb481fbbe794e806309675a89b4df48b8d31908a8", size = 26799, upload-time = "2026-06-06T08:58:33.269Z" }, + { url = "https://files.pythonhosted.org/packages/7e/72/c6b21e58d3113596e7e8de0a08d6f1d95173492cfbca0a4db14148cbba2a/soundfile-0.14.0-py2.py3-none-macosx_10_9_x86_64.whl", hash = "sha256:19be05428da76ed61a4cad29b8e4bcf43a3e5c100089d2ec81dc961eed1b0dd4", size = 1144568, upload-time = "2026-06-06T08:58:35.231Z" }, + { url = "https://files.pythonhosted.org/packages/63/7a/dfdd6f8c748988427119f75eb860a3cedd858d1aea1fe28f39ad8559ef22/soundfile-0.14.0-py2.py3-none-macosx_11_0_arm64.whl", hash = "sha256:d828d35a059626da52f1415b5faee610aeab393319cb3fc4a9aef47b619fc14c", size = 1103726, upload-time = "2026-06-06T08:58:37.948Z" }, + { url = "https://files.pythonhosted.org/packages/4a/f8/fc39fad6f879633461d27394cd1ddaf1f769ffa0597dca35872f51b16461/soundfile-0.14.0-py2.py3-none-manylinux_2_28_aarch64.whl", hash = "sha256:e85724a90bc99a6e8062c0b4ddf725f53b2a3b70afd4da875e9d2cfc4e92f377", size = 1238050, upload-time = "2026-06-06T08:58:39.932Z" }, + { url = "https://files.pythonhosted.org/packages/7b/a2/70fd4432b924684c372df8b0a45708c36c057ef3596c9eb53e0a806b980b/soundfile-0.14.0-py2.py3-none-manylinux_2_28_x86_64.whl", hash = "sha256:1e38bac1853412871318e82a1ba69a8be677619b56025bbfcccdb41b6cafe82d", size = 1315963, upload-time = "2026-06-06T08:58:41.716Z" }, + { url = "https://files.pythonhosted.org/packages/d9/34/c9e80783d83eab739a9531fdee03675d53e0bf1b2ccb4bb3af5844675046/soundfile-0.14.0-py2.py3-none-win32.whl", hash = "sha256:0a6ae43c50c71b4e020cc55382925cb89451c1ed1a0c3d0f5d802da269226849", size = 902199, upload-time = "2026-06-06T08:58:43.289Z" }, + { url = "https://files.pythonhosted.org/packages/ed/97/b39c18ac1df45e755ca22b8b00e872929da5d107998a207a5e4ac831bfda/soundfile-0.14.0-py2.py3-none-win_amd64.whl", hash = "sha256:299491d3499460fb1b74bb4bd78b57ffc2d243a5fafa7b6ec1b264875c78453e", size = 1021480, upload-time = "2026-06-06T08:58:45.016Z" }, + { url = "https://files.pythonhosted.org/packages/f4/83/55c65e61cf457805ce2ec157c1c6ae17715d0851aa2374422de0538838ca/soundfile-0.14.0-py2.py3-none-win_arm64.whl", hash = "sha256:e090704718e124e7c844695236f1fce8d18a5e761eaf7c82dfcd124620805f98", size = 888858, upload-time = "2026-06-06T08:58:46.593Z" }, +] + [[package]] name = "sqlalchemy" version = "2.0.54"