SmallestAISttLiveRealtimeClient streams audio to Pulse over a WebSocket and
returns speaker-labelled transcripts at roughly one second of latency. Audio
goes up as binary frames; results come back as JSON.
Only frames with is_final set carry the words array, so speaker attribution
must never be rendered from interim frames. Send finalize to flush the last
utterance and close_stream to end the session — the server then emits a final
frame with is_last set.
Only pulse is served here. pulse-pro has no streaming worker and returns
400 before the WebSocket upgrade completes.
This example assumes using SmallestAI; is in scope and apiKey contains your SmallestAI API key.
usingvarclient=newSmallestAIClient(apiKey);varwav=awaitBuildTwoSpeakerWavAsync(client,TestContext.CancellationToken);varpcm=ReadPcm(wav);awaitrealtime.ConnectAsync(model:SttLiveModel.Pulse,language:"en",encoding:SttLiveEncoding.Linear16,sampleRate:FixtureSampleRate,wordTimestamps:SttLiveWordTimestamps.True,diarize:SttLiveDiarize.True,sentenceTimestamps:SttLiveSentenceTimestamps.True,cancellationToken:TestContext.CancellationToken);varupload=Task.Run(async()=>{for(varoffset=0;offset<pcm.Length;offset+=4096){varlength=Math.Min(4096,pcm.Length-offset);awaitrealtime.SendAudioAsync(pcm.AsMemory(offset,length),TestContext.CancellationToken);}awaitrealtime.SendSttFinalizeAsync(TestContext.CancellationToken);awaitrealtime.SendSttCloseStreamAsync(TestContext.CancellationToken);});varfinals=newList<SttTranscriptionEvent>();varsawLast=false;awaitforeach(var@eventinrealtime.ReceiveUpdatesAsync(TestContext.CancellationToken)){if(@event.IsFinal==true){finals.Add(@event);}if(@event.IsLast==true){sawLast=true;break;}}awaitupload;// Word-level speaker attribution arrives only on final frames.varspeakers=finals.SelectMany(@event=>@event.Words??[]).Where(word=>word.Speaker!=null).Select(word=>word.Speaker!.Value).Distinct().ToList();