Skip to navigation

Audio I/O

Audio formats Hydra accepts and returns.
View as Markdown

Hydra is strict about audio formats. Get this wrong and you’ll either see invalid_audio errors or distorted playback.

Input (client → server)

PropertyValue
CodecPCM16 (16-bit signed integer)
EndiannessLittle-endian
ChannelsMono
Sample rate16 000 Hz
Encoding on the wireBase64, inside input_audio_buffer.append.audio

Python

import asyncio, base64, json, wave
with wave.open("input_16khz_mono.wav", "rb") as w:
assert w.getframerate() == 16000 and w.getnchannels() == 1
while True:
pcm = w.readframes(320) # 20 ms at 16 kHz
if not pcm:
break
await ws.send(json.dumps({
"type": "input_audio_buffer.append",
"audio": base64.b64encode(pcm).decode(),
}))
await asyncio.sleep(0.02) # pace at real-time

Browser (AudioWorklet)

The mic delivers float32 samples; you need to (a) convert to int16 and (b) base64-encode each chunk. Use an AudioWorklet - the deprecated ScriptProcessorNode works for a prototype but blocks the main thread under load.

// my-mic-worklet.js
class MicWorklet extends AudioWorkletProcessor {
constructor() { super(); this._buf = []; this._frames = 0; }
process(inputs) {
const ch = inputs[0]?.[0];
if (!ch) return true;
this._buf.push(new Float32Array(ch));
this._frames += ch.length;
if (this._frames >= 320) { // 20 ms at 16 kHz
const out = new Float32Array(this._frames);
let o = 0;
for (const c of this._buf) { out.set(c, o); o += c.length; }
this.port.postMessage(out);
this._buf = []; this._frames = 0;
}
return true;
}
}
registerProcessor("mic-worklet", MicWorklet);
// main thread
const stream = await navigator.mediaDevices.getUserMedia({ audio: true });
const ctx = new AudioContext({ sampleRate: 16000 });
await ctx.audioWorklet.addModule("my-mic-worklet.js");
const src = ctx.createMediaStreamSource(stream);
const node = new AudioWorkletNode(ctx, "mic-worklet");
node.port.onmessage = (e) => {
const pcm16 = floatTo16BitPCM(e.data);
ws.send(JSON.stringify({
type: "input_audio_buffer.append",
audio: arrayBufferToBase64(pcm16.buffer),
}));
};
src.connect(node);
function floatTo16BitPCM(f32) {
const out = new Int16Array(f32.length);
for (let i = 0; i < f32.length; i++) {
const s = Math.max(-1, Math.min(1, f32[i]));
out[i] = s < 0 ? s * 0x8000 : s * 0x7fff;
}
return out;
}
function arrayBufferToBase64(buf) {
let bin = "";
const b = new Uint8Array(buf);
for (let i = 0; i < b.length; i++) bin += String.fromCharCode(b[i]);
return btoa(bin);
}

Output (server → client)

Sample rate is per-model. Read it from session.configured before initializing your audio pipeline.

?model=output_audio_sample_rate
hydra-v1.0 (default)48000
hydra-v1.124000

?model=hydra (bare, no version) currently routes to hydra-v1.0. This parameter will be deprecated in the future.

PropertyValue
CodecPCM16
EndiannessLittle-endian
ChannelsMono
Sample rateRead from session.configured.session.output_audio_sample_rate (48000 on hydra-v1.0, 24000 on hydra-v1.1)
Encoding on the wireBase64, inside response.output_audio.delta.delta

Python

out_chunks = []
out_rate = None # populated from session.configured
async for raw in ws:
evt = json.loads(raw)
if evt["type"] == "session.configured":
out_rate = evt["session"]["output_audio_sample_rate"]
elif evt["type"] == "response.output_audio.delta":
out_chunks.append(base64.b64decode(evt["delta"]))
# Later, write to a WAV at the rate the server advertised
with wave.open("reply.wav", "wb") as w:
w.setnchannels(1); w.setsampwidth(2); w.setframerate(out_rate)
w.writeframes(b"".join(out_chunks))

Browser (gapless playback)

Schedule each chunk against a running playCursor so chunks play back-to-back with no audible gap. The AudioContext is created after session.configured arrives so its sample rate matches the server’s.

let playCtx = null;
let playCursor = 0;
let outRate = null;
ws.onmessage = (ev) => {
const evt = JSON.parse(ev.data);
if (evt.type === "session.configured") {
outRate = evt.session.output_audio_sample_rate;
playCtx = new AudioContext({ sampleRate: outRate });
playCursor = playCtx.currentTime;
return;
}
if (evt.type === "response.output_audio.delta") {
playPCM16(b64ToInt16(evt.delta));
}
};
function playPCM16(int16) {
const buf = playCtx.createBuffer(1, int16.length, outRate);
const ch = buf.getChannelData(0);
for (let i = 0; i < int16.length; i++) ch[i] = int16[i] / 0x8000;
const src = playCtx.createBufferSource();
src.buffer = buf;
src.connect(playCtx.destination);
const start = Math.max(playCtx.currentTime, playCursor);
src.start(start);
playCursor = start + buf.duration;
}
function b64ToInt16(b64) {
// base64 → ArrayBuffer → little-endian Int16Array
const bin = atob(b64);
const buf = new ArrayBuffer(bin.length);
const view = new DataView(buf);
for (let i = 0; i < bin.length; i++) view.setUint8(i, bin.charCodeAt(i));
const out = new Int16Array(bin.length / 2);
for (let i = 0; i < out.length; i++) out[i] = view.getInt16(i * 2, true);
return out;
}

For barge-in, you reset playCursor = playCtx.currentTime when a fresh response.created arrives - see Turn detection & barge-in.

Common gotchas

  • Sending audio before session.configured - frames are silently dropped; the server does not queue them and does not emit an error. Always wait for the session.configured echo before starting the mic.
  • Sample-rate mismatch - sending 24 kHz audio while claiming PCM16 16 kHz produces unintelligible transcription on the model side. Resample explicitly.
  • Stereo input - Hydra expects mono. If you have stereo, downmix before encoding.

Streaming a WAV file (for CI / regression tests)

Hydra is built for live mic streams. For test fixtures, regression tests, or batch jobs you sometimes want to replay a known WAV instead. The pattern paces a 16 kHz mono PCM16 WAV at real-time speed, then collects the response audio to disk.

import asyncio, base64, json, os, wave
import websockets
URL = f"wss://api.smallest.ai/waves/v1/s2s?model=hydra-v1.1&api_key={os.environ['SMALLEST_API_KEY']}"
WAV_IN, WAV_OUT = "input_16khz_mono.wav", "reply.wav"
async def main():
chunks = []
out_rate = None # populated from session.configured
configured = asyncio.Event() # gate audio streaming on session.configured
done = asyncio.Event() # set by reader when response.done arrives
async with websockets.connect(URL, max_size=None) as ws:
async def reader():
nonlocal out_rate
async for raw in ws:
evt = json.loads(raw)
t = evt["type"]
if t == "session.created":
await ws.send(json.dumps({
"type": "session.configure",
"session": {"instructions": "Reply briefly.", "voice": "aria"},
}))
elif t == "session.configured":
out_rate = evt["session"]["output_audio_sample_rate"]
configured.set()
elif t == "response.output_audio.delta":
chunks.append(base64.b64decode(evt["delta"]))
elif t == "response.done":
print(f"[{evt['response']['status']}]")
done.set()
elif t == "error":
print("ERROR:", evt["error"])
recv_task = asyncio.create_task(reader())
await configured.wait() # don't stream audio before the server is ready
with wave.open(WAV_IN, "rb") as w:
assert w.getframerate() == 16000 and w.getnchannels() == 1
while pcm := w.readframes(320): # 20 ms at 16 kHz
if done.is_set():
break
await ws.send(json.dumps({
"type": "input_audio_buffer.append",
"audio": base64.b64encode(pcm).decode(),
}))
await asyncio.sleep(0.02) # pace at real-time
try:
await asyncio.wait_for(done.wait(), timeout=15)
except asyncio.TimeoutError:
pass
recv_task.cancel()
with wave.open(WAV_OUT, "wb") as w:
w.setnchannels(1); w.setsampwidth(2); w.setframerate(out_rate)
w.writeframes(b"".join(chunks))
print(f"wrote {WAV_OUT} ({out_rate} Hz)")
asyncio.run(main())

Don’t have a 16 kHz mono WAV? Convert with ffmpeg:

ffmpeg -i any-input.wav -ac 1 -ar 16000 -sample_fmt s16 input_16khz_mono.wav

This pattern is for testing only - it doesn’t exercise full-duplex behaviour (no overlap, no barge-in). For interactive use, see the quickstart.

Next