wss://stt.navana.aiA WebSocket that accepts PCM audio frames and returns transcript frames in real time.
Authentication
X-Api-Key: <key> on the upgrade request. See
Authentication.
Protocol
sequenceDiagram
participant C as Client
participant B as Bodhi
C->>B: upgrade + X-Api-Key
C->>B: {"config": {...}}
loop
C->>B: binary PCM16
B-->>C: {"type":"partial"|"complete", ...}
end
C->>B: {"eof": 1}
B-->>C: final frames, last one has eos: true
B->>C: close
1. Config frame, client to server
The first message, as JSON text. Nothing is transcribed until it arrives.
{
"config": {
"sample_rate": 8000,
"transaction_id": "a-uuid-you-generate",
"model": "hi-banking-v2-8khz"
}
}| Field | Type | Required | Description |
|---|---|---|---|
sample_rate |
number | Yes | Sample rate of the PCM frames you will send. Must match your audio, since nothing verifies it. |
transaction_id |
string | Yes | An id you generate, to correlate this session with your own logs. Must be a valid UUID. |
model |
string | Yes | Transcription model. See Models and languages. |
aux |
boolean | No | true to receive segment_meta, which carries timings, confidence, tokens, and per-word confidence on final frames. |
parse_number |
boolean | No | Turn on inverse text normalisation, converting spoken form to written form, so “पच्चीस लाख” becomes 2500000. Beta, and only affects Hindi, Malayalam, Kannada, Gujarati, and Marathi models, silently ignored on the rest. See Inverse text normalisation. |
hotwords |
array | No | Context biasing, as [{"phrase": "बोधी", "score": 2.5}]. Boosts recognition of domain-specific or uncommon phrases. score is optional and defaults to 1.5; around 2.5 is recommended for longer phrases. |
exclude_partial |
boolean | No | true to receive only "complete" frames, suppressing partials entirely. Defaults to false. |
endpoint_silence_duration |
number | No | Seconds of silence before an in-progress segment is finalized. Accepts 0.44 to 1.2, defaults to 0.44. |
2. Audio frames, client to server
Binary WebSocket frames carrying 16-bit signed PCM, mono, at the sample rate you declared in the config frame.
There is no required frame size. Smaller frames lower latency, larger ones
reduce overhead, and around 100 ms is a reasonable default. At 8 kHz mono that
is 1600 bytes per frame, since one second of audio is sample_rate × 2 bytes.
3. Transcript frames, server to client
{
"call_id": "0f8c1b2e-4a55-4c7e-9d31-1b6a0e2f7c44",
"segment_id": 0,
"eos": false,
"type": "partial",
"text": "the words recognised so far",
"segment_meta": {
"start_time": 0,
"confidence": 0.91,
"tokens": ["the", "words", "recognised", "so", "far"],
"timestamps": [0.2, 0.5, 0.8, 1.1, 1.3],
"words": [
{ "word": "the", "confidence": 0.94 },
{ "word": "words", "confidence": 0.91 }
]
}
}| Field | Type | Description |
|---|---|---|
call_id |
string | Identifier for this streaming connection. |
segment_id |
number | Which speech segment this frame is about. Successive frames with the same id replace each other, so do not append. |
eos |
boolean | true on the frame that ends the stream. This is how you know nothing more is coming. |
type |
string | "partial" while the segment is still refining, "complete" once it is final. |
text |
string | The transcript processed so far for this segment. |
segment_meta |
object | Present when you set aux: true. |
segment_meta.start_time |
number | Where this segment starts, in seconds from the start of the audio. |
segment_meta.confidence |
number | Segment-level confidence, 0 to 1. |
segment_meta.tokens |
string[] | The individual tokens recognized from the audio, in order. |
segment_meta.timestamps |
number[] | When each token was detected, in seconds. Pairs positionally with tokens. |
segment_meta.words |
array | Only populated when type is "complete". Per-word breakdown as [{"word": "…", "confidence": 0.91}]. |
4. EOF frame, client to server
{ "eof": 1 }Signals that you have finished sending audio. Remaining transcripts flush, and
the final frame carries eos: true.
Example
import asyncio, json, os, uuid, wave
from websockets.asyncio.client import connect
CHUNK_MS = 100 # frame size is up to you; 100 ms is a reasonable default
async def transcribe(path: str) -> str:
wf = wave.open(path, "rb")
chunk_frames = int(wf.getframerate() * CHUNK_MS / 1000)
async with connect(
"wss://stt.navana.ai",
additional_headers={"X-Api-Key": os.environ["BODHI_API_KEY"]},
) as socket:
await socket.send(json.dumps({"config": {
"sample_rate": wf.getframerate(),
"transaction_id": str(uuid.uuid4()),
"model": "hi-banking-v2-8khz",
"aux": True,
}}))
segments: dict[int, str] = {}
async def send_audio():
while chunk := wf.readframes(chunk_frames):
await socket.send(chunk)
await asyncio.sleep(CHUNK_MS / 1000) # pace like live capture
await socket.send(json.dumps({"eof": 1}))
async def receive():
async for raw in socket:
frame = json.loads(raw)
if "error" in frame:
raise RuntimeError(frame)
segments[frame["segment_id"]] = frame["text"]
if frame.get("eos"):
return
await asyncio.gather(send_audio(), receive())
return " ".join(segments[k] for k in sorted(segments))import WebSocket from 'ws'
import { randomUUID } from 'node:crypto'
import { readFile } from 'node:fs/promises'
const CHUNK_MS = 100 // frame size is up to you; 100 ms is a reasonable default
// Walk the RIFF chunks for the sample rate and the start of the samples. Don't
// assume a fixed 44-byte header: any WAV carrying an extra chunk, such as a
// LIST, puts the audio elsewhere, and the misalignment turns into a garbled
// transcript rather than an error.
function readWav(buf: Buffer): { sampleRate: number; samples: Buffer } {
let offset = 12
let sampleRate = 0
while (offset < buf.length) {
const id = buf.toString('ascii', offset, offset + 4)
const size = buf.readUInt32LE(offset + 4)
const body = offset + 8
if (id === 'fmt ') sampleRate = buf.readUInt32LE(body + 4)
if (id === 'data') return { sampleRate, samples: buf.subarray(body, body + size) }
offset = body + size + (size % 2)
}
throw new Error('no data chunk in WAV')
}
export async function transcribe(path: string): Promise<string> {
const { sampleRate, samples } = readWav(await readFile(path))
const bytesPerChunk = (sampleRate * 2 * CHUNK_MS) / 1000
const segments = new Map<number, string>()
const socket = new WebSocket('wss://stt.navana.ai', {
headers: { 'X-Api-Key': process.env.BODHI_API_KEY! },
})
const transcript = () =>
[...segments].sort((a, b) => a[0] - b[0]).map(([, t]) => t).join(' ')
return new Promise<string>((resolve, reject) => {
socket.on('error', reject)
// Settle on whichever comes first, `eos` or the socket closing.
socket.on('close', () => resolve(transcript()))
socket.on('open', async () => {
socket.send(JSON.stringify({ config: {
sample_rate: sampleRate,
transaction_id: randomUUID(),
model: 'hi-banking-v2-8khz',
aux: true,
}}))
for (let i = 0; i < samples.length; i += bytesPerChunk) {
socket.send(samples.subarray(i, i + bytesPerChunk))
await new Promise((r) => setTimeout(r, CHUNK_MS)) // pace like live capture
}
socket.send(JSON.stringify({ eof: 1 }))
})
socket.on('message', (raw) => {
const frame = JSON.parse(String(raw))
if (frame.error) return reject(new Error(JSON.stringify(frame)))
segments.set(frame.segment_id, frame.text)
if (frame.eos) {
socket.close()
resolve(transcript())
}
})
})
}package main
// go get github.com/gorilla/websocket
import (
"encoding/binary"
"io"
"net/http"
"os"
"time"
"github.com/google/uuid"
"github.com/gorilla/websocket"
)
const chunkMS = 100 // frame size is up to you; 100 ms is a reasonable default
// openWAV walks the RIFF chunks to find the sample rate and the start of the
// samples. Don't skip a fixed 44 bytes: any WAV carrying an extra chunk, such
// as a LIST, puts the audio somewhere else, and the misalignment turns into a
// garbled transcript rather than an error.
func openWAV(path string) (*os.File, int, error) {
f, err := os.Open(path)
if err != nil {
return nil, 0, err
}
var riff [12]byte
if _, err := io.ReadFull(f, riff[:]); err != nil {
return nil, 0, err
}
sampleRate := 0
for {
var ch [8]byte
if _, err := io.ReadFull(f, ch[:]); err != nil {
return nil, 0, err
}
id, size := string(ch[:4]), int64(binary.LittleEndian.Uint32(ch[4:]))
switch id {
case "fmt ":
fmtChunk := make([]byte, size)
if _, err := io.ReadFull(f, fmtChunk); err != nil {
return nil, 0, err
}
sampleRate = int(binary.LittleEndian.Uint32(fmtChunk[4:8]))
case "data":
return f, sampleRate, nil // positioned at the first sample
default:
if _, err := f.Seek(size+size%2, io.SeekCurrent); err != nil {
return nil, 0, err
}
}
}
}
func main() {
// PCM16 LE mono. Take the rate from the file rather than hardcoding it, so
// `config` always matches the audio you actually send.
f, sampleRate, err := openWAV("recording.wav")
if err != nil {
panic(err)
}
defer f.Close()
header := http.Header{}
header.Set("X-Api-Key", os.Getenv("BODHI_API_KEY"))
conn, _, err := websocket.DefaultDialer.Dial("wss://stt.navana.ai", header)
if err != nil {
panic(err)
}
defer conn.Close()
conn.WriteJSON(map[string]any{"config": map[string]any{
"sample_rate": sampleRate,
"transaction_id": uuid.NewString(),
"model": "hi-banking-v2-8khz",
"aux": true,
}})
segments := map[int]string{}
go func() {
for {
var frame struct {
SegmentID int `json:"segment_id"`
Type string `json:"type"`
Text string `json:"text"`
Error string `json:"error"`
EOS bool `json:"eos"`
}
if err := conn.ReadJSON(&frame); err != nil {
return
}
if frame.Error != "" {
return
}
segments[frame.SegmentID] = frame.Text
if frame.EOS {
return
}
}
}()
chunk := make([]byte, sampleRate*2*chunkMS/1000)
for {
n, err := io.ReadFull(f, chunk)
if n > 0 {
conn.WriteMessage(websocket.BinaryMessage, chunk[:n])
time.Sleep(chunkMS * time.Millisecond) // pace like live capture
}
if err != nil {
break
}
}
conn.WriteJSON(map[string]int{"eof": 1})
}Errors
The upgrade is rejected with a normal HTTP response, so you get a readable status rather than a silent close.
| Status | Cause |
|---|---|
400 |
Malformed config frame, an invalid transaction_id, or a model that does not exist. |
401 |
Missing or incorrect API key. |
402 |
The account’s credit balance is exhausted. |
403 |
The account is inactive, or the key lacks the required scope. |
500 |
Unexpected server error. |
503 |
The service is unavailable or temporarily overloaded. |
Notes
eosis the end-of-stream signal. Wait for it rather than a fixed delay after youreofframe, since how much audio is still being transcribed when you stop sending varies. Settle your client on whichever comes first, theeosframe or the socket closing, so a session can never leave it waiting.- Sample rate is not verified. Declaring one rate and sending another produces a poor transcript rather than an error.
parse_numberonly affects five languages, namely Hindi, Malayalam, Kannada, Gujarati, and Marathi.