Files
gogo-telefon/gogo/voice/audio.py

95 lines
2.7 KiB
Python
Raw Normal View History

"""Audio helpers for the voice pipeline.
Wire format everywhere: SLIN signed 16-bit LE mono PCM @ 8 kHz (AudioSocket).
Whisper wants 16 kHz; TTS providers are asked for 8 kHz where possible and
resampled otherwise.
"""
from __future__ import annotations
import audioop
import io
import math
import wave
SAMPLE_RATE = 8000
SAMPLE_WIDTH = 2 # bytes, 16-bit
FRAME_MS = 20
FRAME_BYTES = SAMPLE_RATE * SAMPLE_WIDTH * FRAME_MS // 1000 # 320
def resample(pcm: bytes, from_rate: int, to_rate: int) -> bytes:
if from_rate == to_rate or not pcm:
return pcm
out, _ = audioop.ratecv(pcm, SAMPLE_WIDTH, 1, from_rate, to_rate, None)
return out
def rms(pcm: bytes) -> int:
return audioop.rms(pcm, SAMPLE_WIDTH) if pcm else 0
def duration_s(pcm: bytes, rate: int = SAMPLE_RATE) -> float:
return len(pcm) / (rate * SAMPLE_WIDTH)
def silence(ms: int, rate: int = SAMPLE_RATE) -> bytes:
return b"\x00" * (rate * SAMPLE_WIDTH * ms // 1000)
def tone(freq: int, ms: int, rate: int = SAMPLE_RATE, amplitude: int = 12000) -> bytes:
"""Test helper: sine tone (registers as speech for the energy VAD)."""
n = rate * ms // 1000
return b"".join(
int(amplitude * math.sin(2 * math.pi * freq * i / rate)).to_bytes(
2, "little", signed=True
)
for i in range(n)
)
def pcm_to_wav(pcm: bytes, rate: int = SAMPLE_RATE) -> bytes:
buf = io.BytesIO()
with wave.open(buf, "wb") as w:
w.setnchannels(1)
w.setsampwidth(SAMPLE_WIDTH)
w.setframerate(rate)
w.writeframes(pcm)
return buf.getvalue()
def mix(a: bytes, b: bytes) -> bytes:
"""Mix two PCM streams (unequal lengths ok)."""
if len(a) < len(b):
a, b = b, a
if not b:
return a
mixed = audioop.add(a[: len(b)], b, SAMPLE_WIDTH)
return mixed + a[len(b):]
class CallRecorder:
"""Accumulates caller + agent audio on a shared timeline and writes a WAV."""
def __init__(self) -> None:
self._caller = bytearray()
self._agent = bytearray()
def add_caller(self, pcm: bytes) -> None:
self._caller.extend(pcm)
# keep agent track in sync (silence while caller talks)
if len(self._agent) < len(self._caller):
self._agent.extend(b"\x00" * (len(self._caller) - len(self._agent)))
def add_agent(self, pcm: bytes) -> None:
self._agent.extend(pcm)
if len(self._caller) < len(self._agent):
self._caller.extend(b"\x00" * (len(self._agent) - len(self._caller)))
def to_wav(self) -> bytes:
return pcm_to_wav(mix(bytes(self._caller), bytes(self._agent)))
@property
def seconds(self) -> float:
return duration_s(bytes(self._caller))