"""Audio helpers for the voice pipeline. Wire format everywhere: SLIN — signed 16-bit LE mono PCM @ 8 kHz (AudioSocket). Whisper wants 16 kHz; TTS providers are asked for 8 kHz where possible and resampled otherwise. """ from __future__ import annotations import audioop import io import math import wave SAMPLE_RATE = 8000 SAMPLE_WIDTH = 2 # bytes, 16-bit FRAME_MS = 20 FRAME_BYTES = SAMPLE_RATE * SAMPLE_WIDTH * FRAME_MS // 1000 # 320 def resample(pcm: bytes, from_rate: int, to_rate: int) -> bytes: if from_rate == to_rate or not pcm: return pcm out, _ = audioop.ratecv(pcm, SAMPLE_WIDTH, 1, from_rate, to_rate, None) return out def rms(pcm: bytes) -> int: return audioop.rms(pcm, SAMPLE_WIDTH) if pcm else 0 def duration_s(pcm: bytes, rate: int = SAMPLE_RATE) -> float: return len(pcm) / (rate * SAMPLE_WIDTH) def silence(ms: int, rate: int = SAMPLE_RATE) -> bytes: return b"\x00" * (rate * SAMPLE_WIDTH * ms // 1000) def tone(freq: int, ms: int, rate: int = SAMPLE_RATE, amplitude: int = 12000) -> bytes: """Test helper: sine tone (registers as speech for the energy VAD).""" n = rate * ms // 1000 return b"".join( int(amplitude * math.sin(2 * math.pi * freq * i / rate)).to_bytes( 2, "little", signed=True ) for i in range(n) ) def pcm_to_wav(pcm: bytes, rate: int = SAMPLE_RATE) -> bytes: buf = io.BytesIO() with wave.open(buf, "wb") as w: w.setnchannels(1) w.setsampwidth(SAMPLE_WIDTH) w.setframerate(rate) w.writeframes(pcm) return buf.getvalue() def mix(a: bytes, b: bytes) -> bytes: """Mix two PCM streams (unequal lengths ok).""" if len(a) < len(b): a, b = b, a if not b: return a mixed = audioop.add(a[: len(b)], b, SAMPLE_WIDTH) return mixed + a[len(b):] class CallRecorder: """Accumulates caller + agent audio on a shared timeline and writes a WAV.""" def __init__(self) -> None: self._caller = bytearray() self._agent = bytearray() def add_caller(self, pcm: bytes) -> None: self._caller.extend(pcm) # keep agent track in sync (silence while caller talks) if len(self._agent) < len(self._caller): self._agent.extend(b"\x00" * (len(self._caller) - len(self._agent))) def add_agent(self, pcm: bytes) -> None: self._agent.extend(pcm) if len(self._caller) < len(self._agent): self._caller.extend(b"\x00" * (len(self._agent) - len(self._caller))) def to_wav(self) -> bytes: return pcm_to_wav(mix(bytes(self._caller), bytes(self._agent))) @property def seconds(self) -> float: return duration_s(bytes(self._caller))