Docs / Guides / Python

Guides · Live API

Python

Connect over a plain WebSocket to send live microphone audio through the .wave Live API and play the model’s responses as they arrive.

Install and configure

python -m pip install websockets sounddevice
export DOTWAVE_API_KEY="wk_…"

Connect with your key as a bearer header. The Live socket takes no query parameters.

import os
import websockets

LIVE_URL = "wss://api.dotwave.ai/v1/live/sessions"
AUTH = {"Authorization": f"Bearer {os.environ['DOTWAVE_API_KEY']}"}

Full-duplex conversation

This example captures mono PCM, sends chunks as the microphone produces them, plays returned PCM, and prints both sides of the conversation.

import asyncio, base64, json, queue
import sounddevice as sd
import websockets

async def send_microphone(ws):
    chunks = queue.SimpleQueue()

    def captured(indata, frames, time, status):
        if status:
            print(status)
        chunks.put(bytes(indata))

    with sd.RawInputStream(
        samplerate=24000, channels=1, dtype="int16",
        blocksize=1920, callback=captured,
    ):
        while True:
            audio = await asyncio.to_thread(chunks.get)
            await ws.send(json.dumps({
                "type": "session.input_audio.append",
                "audio": base64.b64encode(audio).decode(),
            }))

async def receive(ws):
    with sd.RawOutputStream(
        samplerate=24000, channels=1, dtype="int16",
        blocksize=1920,
    ) as speaker:
        async for raw in ws:
            event = json.loads(raw)
            kind = event.get("type", "")
            if kind == "session.output_audio.delta":
                await asyncio.to_thread(
                    speaker.write, base64.b64decode(event["delta"])
                )
            elif kind in ("session.input_transcript.delta", "session.output_transcript.delta"):
                print(event.get("delta", ""), end="", flush=True)
            elif kind == "error":
                raise RuntimeError(event["error"]["message"])

async def main():
    async with websockets.connect(LIVE_URL, additional_headers=AUTH) as ws:
        # session.start must be the first event, or the socket refuses it.
        await ws.send(json.dumps({
            "type": "session.start",
            "session": {
                "model": "nemotron-voicechat",
                "audio": {
                    "input": {
                        "format": {"type": "audio/pcm", "rate": 24000},
                    },
                    "output": {
                        "format": {"type": "audio/pcm", "rate": 24000},
                        "encoding": "base64",
                    },
                },
            },
        }))
        await asyncio.gather(
            send_microphone(ws),
            receive(ws),
        )

asyncio.run(main())

Send a recorded turn

A file must enter the socket at the same pace as live audio. Replace the microphone sender with this WAV sender; the receive loop still plays the model response. Live appends audio straight onto the session, so there is no buffer to commit when the turn ends.

import asyncio, base64, json, time, wave

async def send_wave(ws, path):
    with wave.open(path, "rb") as source:
        assert source.getnchannels() == 1
        assert source.getsampwidth() == 2
        assert source.getframerate() == 24000

        frames_per_chunk = 1920
        started = time.monotonic()
        sent = 0

        while audio := source.readframes(frames_per_chunk):
            await ws.send(json.dumps({
                "type": "session.input_audio.append",
                "audio": base64.b64encode(audio).decode(),
            }))
            sent += len(audio) // 2
            deadline = started + sent / source.getframerate()
            await asyncio.sleep(max(0, deadline - time.monotonic()))

Limits

Session length, voices and supported fields depend on the model: see its page in Models. A field the model does not support is refused, never silently ignored.