# OpenAI SDK

The Realtime API uses OpenAI’s Realtime transcription events, so the OpenAI SDK connects by changing only its base URL and key.

[Create an account](https://api.dotwave.ai/auth/signup) · [Realtime API](https://dotwave.ai/docs/realtime/)

## Python

Install the SDK with its WebSocket support, and set your key:

```bash
python -m pip install "openai[realtime]"
export DOTWAVE_API_KEY="wk_…"
```

This example transcribes a 24 kHz mono WAV file, sending it at the speed of speech. Name the language with its tag, here `pt-BR`, or leave it out to let the model detect the language.

```python
import asyncio, base64, os, time, wave
from openai import AsyncOpenAI

client = AsyncOpenAI(
    api_key=os.environ["DOTWAVE_API_KEY"],
    base_url="https://api.dotwave.ai/v1",
)

async def send_wav(conn, path):
    # 24 kHz mono PCM16, paced against a clock at the speed of speech.
    with wave.open(path, "rb") as wav:
        started, sent = time.monotonic(), 0
        while chunk := wav.readframes(1920):  # 1,920 samples at a time
            audio = base64.b64encode(chunk).decode()
            await conn.input_audio_buffer.append(audio=audio)
            sent += len(chunk) // 2
            await asyncio.sleep(max(0, started + sent / 24000 - time.monotonic()))

async def print_transcripts(conn):
    async for event in conn:
        if event.type == "conversation.item.input_audio_transcription.delta":
            print(event.delta, end="", flush=True)
        elif event.type == "conversation.item.input_audio_transcription.completed":
            print()

async def main():
    async with client.realtime.connect(model="nemotron-asr-streaming") as conn:
        await conn.session.update(session={
            "type": "transcription",
            "audio": {"input": {
                "format": {"type": "audio/pcm", "rate": 24000},
                "transcription": {
                    "model": "nemotron-asr-streaming",
                    "language": "pt-BR",
                },
            }},
        })
        printer = asyncio.create_task(print_transcripts(conn))
        await send_wav(conn, "fala-24k.wav")
        await asyncio.sleep(4)  # the last segment completes after 3.2 s of silence
        printer.cancel()

asyncio.run(main())
```

## TypeScript

In Node.js, the SDK’s Realtime client runs on the `ws` package:

```bash
npm install openai ws
export DOTWAVE_API_KEY="wk_…"
```

```typescript
import OpenAI from 'openai';
import { OpenAIRealtimeWS } from 'openai/realtime/ws';
import { open } from 'node:fs/promises';
import { setTimeout as delay } from 'node:timers/promises';

const client = new OpenAI({
  apiKey: process.env.DOTWAVE_API_KEY,
  baseURL: 'https://api.dotwave.ai/v1',
});
const rt = new OpenAIRealtimeWS({ model: 'nemotron-asr-streaming' }, client);

rt.on('conversation.item.input_audio_transcription.delta', event => {
  process.stdout.write(event.delta);
});
rt.on('conversation.item.input_audio_transcription.completed', () => {
  process.stdout.write('\n');
});
rt.on('error', error => console.error(error.message));

// Raw 24 kHz mono PCM16, paced against a clock at the speed of speech.
async function sendPcm(path: string) {
  const file = await open(path, 'r');
  const chunk = Buffer.alloc(3840); // 1,920 samples
  const started = performance.now();
  let sent = 0;
  try {
    while (true) {
      const { bytesRead } = await file.read(chunk, 0, chunk.length, null);
      if (!bytesRead) break;
      rt.send({
        type: 'input_audio_buffer.append',
        audio: chunk.subarray(0, bytesRead).toString('base64'),
      });
      sent += bytesRead;
      const audioMs = sent / (24000 * 2) * 1000;
      await delay(Math.max(0, started + audioMs - performance.now()));
    }
  } finally {
    await file.close();
  }
}

rt.socket.on('open', async () => {
  rt.send({
    type: 'session.update',
    session: {
      type: 'transcription',
      audio: {
        input: {
          format: { type: 'audio/pcm', rate: 24000 },
          transcription: {
            model: 'nemotron-asr-streaming',
            language: 'pt-BR',
          },
        },
      },
    },
  });
  await sendPcm('fala-24k.pcm');
  await delay(4000); // the last segment completes after 3.2 s of silence
  rt.close();
});
```

## Next

The [WebSocket](https://dotwave.ai/docs/realtime/websocket/) page covers audio formats, pacing and segments, and [client secrets](https://dotwave.ai/docs/realtime/client-secrets/) cover browsers.
