Guides · Realtime API
OpenAI SDK
The Realtime API uses OpenAI’s Realtime transcription events, so the OpenAI SDK connects by changing only its base URL and key.
Python
Install the SDK with its WebSocket support, and set your key:
python -m pip install "openai[realtime]"
export DOTWAVE_API_KEY="wk_…"This example transcribes a 24 kHz mono WAV file, sending it at the speed of speech. Name the language with its tag, here pt-BR, or leave it out to let the model detect the language.
import asyncio, base64, os, time, wave
from openai import AsyncOpenAI
client = AsyncOpenAI(
api_key=os.environ["DOTWAVE_API_KEY"],
base_url="https://api.dotwave.ai/v1",
)
async def send_wav(conn, path):
# 24 kHz mono PCM16, paced against a clock at the speed of speech.
with wave.open(path, "rb") as wav:
started, sent = time.monotonic(), 0
while chunk := wav.readframes(1920): # 1,920 samples at a time
audio = base64.b64encode(chunk).decode()
await conn.input_audio_buffer.append(audio=audio)
sent += len(chunk) // 2
await asyncio.sleep(max(0, started + sent / 24000 - time.monotonic()))
async def print_transcripts(conn):
async for event in conn:
if event.type == "conversation.item.input_audio_transcription.delta":
print(event.delta, end="", flush=True)
elif event.type == "conversation.item.input_audio_transcription.completed":
print()
async def main():
async with client.realtime.connect(model="nemotron-asr-streaming") as conn:
await conn.session.update(session={
"type": "transcription",
"audio": {"input": {
"format": {"type": "audio/pcm", "rate": 24000},
"transcription": {
"model": "nemotron-asr-streaming",
"language": "pt-BR",
},
}},
})
printer = asyncio.create_task(print_transcripts(conn))
await send_wav(conn, "fala-24k.wav")
await asyncio.sleep(4) # the last segment completes after 3.2 s of silence
printer.cancel()
asyncio.run(main())TypeScript
In Node.js, the SDK’s Realtime client runs on the ws package:
npm install openai ws
export DOTWAVE_API_KEY="wk_…"import OpenAI from 'openai';
import { OpenAIRealtimeWS } from 'openai/realtime/ws';
import { open } from 'node:fs/promises';
import { setTimeout as delay } from 'node:timers/promises';
const client = new OpenAI({
apiKey: process.env.DOTWAVE_API_KEY,
baseURL: 'https://api.dotwave.ai/v1',
});
const rt = new OpenAIRealtimeWS({ model: 'nemotron-asr-streaming' }, client);
rt.on('conversation.item.input_audio_transcription.delta', event => {
process.stdout.write(event.delta);
});
rt.on('conversation.item.input_audio_transcription.completed', () => {
process.stdout.write('\n');
});
rt.on('error', error => console.error(error.message));
// Raw 24 kHz mono PCM16, paced against a clock at the speed of speech.
async function sendPcm(path: string) {
const file = await open(path, 'r');
const chunk = Buffer.alloc(3840); // 1,920 samples
const started = performance.now();
let sent = 0;
try {
while (true) {
const { bytesRead } = await file.read(chunk, 0, chunk.length, null);
if (!bytesRead) break;
rt.send({
type: 'input_audio_buffer.append',
audio: chunk.subarray(0, bytesRead).toString('base64'),
});
sent += bytesRead;
const audioMs = sent / (24000 * 2) * 1000;
await delay(Math.max(0, started + audioMs - performance.now()));
}
} finally {
await file.close();
}
}
rt.socket.on('open', async () => {
rt.send({
type: 'session.update',
session: {
type: 'transcription',
audio: {
input: {
format: { type: 'audio/pcm', rate: 24000 },
transcription: {
model: 'nemotron-asr-streaming',
language: 'pt-BR',
},
},
},
},
});
await sendPcm('fala-24k.pcm');
await delay(4000); // the last segment completes after 3.2 s of silence
rt.close();
});Next
The WebSocket page covers audio formats, pacing and segments, and client secrets cover browsers.