Get started
Quickstart
Create an API key, check it, then make a first call. Each example runs from a terminal with Python and a 24 kHz mono WAV file.
1. Create an API key
Create an account. Your API key is shown once after signup; new accounts start with free credits, no card required. Keep the key on your server, in an environment variable:
export DOTWAVE_API_KEY="wk_…"2. Check the key
List the models your key can use:
curl https://api.dotwave.ai/v1/models \
-H "Authorization: Bearer $DOTWAVE_API_KEY"3. Make a first call
Choose the tab for your API. The Live API and Deepgram-compatible API examples use the websockets package; the Realtime API example uses the OpenAI SDK, pip install "openai[realtime]".
import asyncio, base64, json, os, time, wave
import websockets
URL = "wss://api.dotwave.ai/v1/live/sessions"
HEADERS = {"Authorization": f"Bearer {os.environ['DOTWAVE_API_KEY']}"}
async def send_audio(ws, path):
# 24 kHz mono PCM16, then 5 s of silence while the model answers.
with wave.open(path, "rb") as f:
assert f.getframerate() == 24000 and f.getsampwidth() == 2
assert f.getnchannels() == 1
audio = f.readframes(f.getnframes()) + bytes(48000 * 5)
# 3,840 bytes at a time, paced against a clock at the speed of speech.
started = time.monotonic()
for offset in range(0, len(audio), 3840):
await ws.send(json.dumps({
"type": "session.input_audio.append",
"audio": base64.b64encode(audio[offset:offset + 3840]).decode(),
}))
await asyncio.sleep(max(0, started + (offset + 3840) / 48000 - time.monotonic()))
await ws.send(json.dumps({"type": "session.close"}))
async def main():
async with websockets.connect(URL, additional_headers=HEADERS) as ws:
# session.start must be the first event, or the socket refuses it.
await ws.send(json.dumps({
"type": "session.start",
"session": {"model": "nemotron-voicechat"},
}))
sender = asyncio.create_task(send_audio(ws, "speech-24000-mono.wav"))
reply = bytearray()
async for raw in ws:
event = json.loads(raw)
kind = event.get("type", "")
if kind == "session.output_audio.delta":
reply += base64.b64decode(event["delta"])
elif kind.endswith("_transcript.delta"): # both sides' transcripts
print(event.get("delta", ""), end="", flush=True)
elif kind in ("session.closed", "error"):
break
sender.cancel()
with wave.open("reply.wav", "wb") as f:
f.setnchannels(1); f.setsampwidth(2); f.setframerate(24000)
f.writeframes(bytes(reply))
asyncio.run(main())import asyncio, base64, os, time, wave
from openai import AsyncOpenAI
client = AsyncOpenAI(
api_key=os.environ["DOTWAVE_API_KEY"],
base_url="https://api.dotwave.ai/v1",
)
async def send_wav(conn, path):
# 24 kHz mono PCM16, paced against a clock at the speed of speech.
with wave.open(path, "rb") as wav:
started, sent = time.monotonic(), 0
while chunk := wav.readframes(1920): # 1,920 samples at a time
audio = base64.b64encode(chunk).decode()
await conn.input_audio_buffer.append(audio=audio)
sent += len(chunk) // 2
await asyncio.sleep(max(0, started + sent / 24000 - time.monotonic()))
async def print_transcripts(conn):
async for event in conn:
if event.type == "conversation.item.input_audio_transcription.delta":
print(event.delta, end="", flush=True)
elif event.type == "conversation.item.input_audio_transcription.completed":
print()
async def main():
async with client.realtime.connect(model="nemotron-asr-streaming") as conn:
await conn.session.update(session={
"type": "transcription",
"audio": {"input": {
"format": {"type": "audio/pcm", "rate": 24000},
"transcription": {
"model": "nemotron-asr-streaming",
"language": "pt-BR",
},
}},
})
printer = asyncio.create_task(print_transcripts(conn))
await send_wav(conn, "fala-24k.wav")
await asyncio.sleep(4) # the last segment completes after 3.2 s of silence
printer.cancel()
asyncio.run(main())import asyncio, json, os, time, wave
import websockets
URL = ("wss://api.dotwave.ai/v1/listen"
"?encoding=linear16&sample_rate=24000&language=pt-BR")
HEADERS = {"Authorization": f"Token {os.environ['DOTWAVE_API_KEY']}"}
async def send_wav(ws, path):
# 24 kHz mono PCM16 as binary messages, paced at the speed of speech.
with wave.open(path, "rb") as wav:
started, sent = time.monotonic(), 0
while chunk := wav.readframes(1920): # 1,920 samples at a time
await ws.send(chunk)
sent += len(chunk) // 2
await asyncio.sleep(max(0, started + sent / 24000 - time.monotonic()))
# Send what is still open as a final, then close.
await ws.send(json.dumps({"type": "CloseStream"}))
async def main():
async with websockets.connect(URL, additional_headers=HEADERS) as ws:
sender = asyncio.create_task(send_wav(ws, "fala-24k.wav"))
async for raw in ws:
message = json.loads(raw)
if message.get("type") == "Results" and message["is_final"]:
print(message["channel"]["alternatives"][0]["transcript"])
await sender
asyncio.run(main())Next
- Live API
- Events, audio and close codes of a full-duplex session: Live API.
- Realtime API
- Formats, segments and languages of a transcription session: Realtime API.
- Deepgram-compatible API
- Query parameters, messages and plugins: Deepgram-compatible API.
- Browsers
- Create a client secret on your server: Authentication.