Install any skill in seconds. Free to start, no credit card required.
Get Started Free →Implement real-time streaming transcription with Deepgram WebSocket. Use when building live transcription, voice interfaces, real-time captioning, or voice AI applications. Trigger: "deepgram streaming", "real-time transcription", "live transcription", "websocket transcription", "voice streaming", "deepgram live".
.claude/skills/jeremylongshore-deepgram-core-workflow-b/SKILL.md| Test case | Without → With | Effect | Δ tokens | Δ turns |
|---|---|---|---|---|
| case-01 | ✗→✓ | ▲ Improved | 35% | 0% |
| case-02 | ✗→✓ | ▲ Improved | 68% | 0% |
| case-03 | ✗→✓ | ▲ Improved | 90% | 0% |
| case-13 | ✗→✓ | ▲ Improved | 87% | 0% |
| case-15 | ✗→✓ | ▲ Improved | 175% | 0% |
Real-time streaming transcription using Deepgram's WebSocket API. The SDK manages the WebSocket connection via listen.live(). Covers microphone capture, interim/final result handling, speaker diarization, UtteranceEnd detection, auto-reconnect, and building an SSE endpoint for browser clients.
@deepgram/sdk installed, DEEPGRAM_API_KEY configuredrec), file stream, or WebSocket audio from browsersox installed (apt install sox / brew install sox)typescriptimport { createClient, LiveTranscriptionEvents } from '@deepgram/sdk'; const deepgram = createClient(process.env.DEEPGRAM_API_KEY!); const connection = deepgram.listen.live({ model: 'nova-3', language: 'en', smart_format: true, punctuate: true, interim_results: true, // Show in-progress results utterance_end_ms: 1000, // Silence threshold for utterance end vad_events: true, // Voice activity detection events encoding: 'linear16', // 16-bit PCM sample_rate: 16000, // 16 kHz channels: 1, // Mono }); // Connection lifecycle events connection.on(LiveTranscriptionEvents.Open, () => { console.log('WebSocket connected to Deepgram'); }); connection.on(LiveTranscriptionEvents.Close, () => { console.log('WebSocket closed'); }); connection.on(LiveTranscriptionEvents.Error, (err) => { console.error('Deepgram error:', err); }); // Transcript events connection.on(LiveTranscriptionEvents.Transcript, (data) => { const transcript = data.channel.alternatives[0]?.transcript; if (!transcript) return; if (data.is_final) { console.log(`[FINAL] ${transcript}`); } else { process.stdout.write(`\r[interim] ${transcript}`); } }); // UtteranceEnd — fires when speaker pauses connection.on(LiveTranscriptionEvents.UtteranceEnd, () => { console.log('\n--- utterance end ---'); });
typescriptimport { spawn } from 'child_process'; function startMicrophone(connection: any) { // Sox captures from default mic: 16kHz, 16-bit signed LE, mono const mic = spawn('rec', [ '-q', // Quiet (no progress) '-r', '16000', // Sample rate '-e', 'signed', // Encoding '-b', '16', // Bit depth '-c', '1', // Mono '-t', 'raw', // Raw PCM output '-', // Output to stdout ]); mic.stdout.on('data', (chunk: Buffer) => { if (connection.getReadyState() === 1) { // WebSocket.OPEN connection.send(chunk); } }); mic.on('error', (err) => { console.error('Microphone error:', err.message); console.log('Install sox: apt install sox / brew install sox'); }); return mic; } // Usage const mic = startMicrophone(connection); // Graceful shutdown process.on('SIGINT', () => { mic.kill(); connection.finish(); // Sends CloseStream message, waits for final results setTimeout(() => process.exit(0), 2000); });
typescriptconst connection = deepgram.listen.live({ model: 'nova-3', smart_format: true, diarize: true, interim_results: false, // Only final for cleaner diarization utterance_end_ms: 1500, encoding: 'linear16', sample_rate: 16000, channels: 1, }); connection.on(LiveTranscriptionEvents.Transcript, (data) => { if (!data.is_final) return; const words = data.channel.alternatives[0]?.words ?? []; if (words.length === 0) return; // Group consecutive words by speaker let currentSpeaker = words[0].speaker; let segment = ''; for (const word of words) { if (word.speaker !== currentSpeaker) { console.log(`Speaker ${currentSpeaker}: ${segment.trim()}`); currentSpeaker = word.speaker; segment = ''; } segment += ` ${word.punctuated_word ?? word.word}`; } console.log(`Speaker ${currentSpeaker}: ${segment.trim()}`); });
typescriptclass ReconnectingLiveTranscription { private client: ReturnType<typeof createClient>; private connection: any = null; private reconnectAttempts = 0; private maxReconnectAttempts = 10; private baseDelay = 1000; constructor(apiKey: string, private options: Record<string, any>) { this.client = createClient(apiKey); } connect() { this.connection = this.client.listen.live(this.options); this.connection.on(LiveTranscriptionEvents.Open, () => { console.log('Connected'); this.reconnectAttempts = 0; // Reset on success }); this.connection.on(LiveTranscriptionEvents.Close, () => { this.scheduleReconnect(); }); this.connection.on(LiveTranscriptionEvents.Error, (err: Error) => { console.error('Connection error:', err.message); this.scheduleReconnect(); }); return this.connection; } private scheduleReconnect() { if (this.reconnectAttempts >= this.maxReconnectAttempts) { console.error('Max reconnection attempts reached'); return; } const delay = this.baseDelay * Math.pow(2, this.reconnectAttempts) + Math.random() * 1000; // Jitter this.reconnectAttempts++; console.log(`Reconnecting in ${Math.round(delay)}ms (attempt ${this.reconnectAttempts})`); setTimeout(() => this.connect(), delay); } send(chunk: Buffer) { if (this.connection?.getReadyState() === 1) { this.connection.send(chunk); } } close() { this.maxReconnectAttempts = 0; // Prevent reconnect this.connection?.finish(); } }
typescriptimport express from 'express'; import { createClient, LiveTranscriptionEvents } from '@deepgram/sdk'; const app = express(); app.get('/api/transcribe/stream', (req, res) => { res.setHeader('Content-Type', 'text/event-stream'); res.setHeader('Cache-Control', 'no-cache'); res.setHeader('Connection', 'keep-alive'); const deepgram = createClient(process.env.DEEPGRAM_API_KEY!); const connection = deepgram.listen.live({ model: 'nova-3', smart_format: true, interim_results: true, encoding: 'linear16', sample_rate: 16000, channels: 1, }); connection.on(LiveTranscriptionEvents.Transcript, (data) => { const transcript = data.channel.alternatives[0]?.transcript; if (transcript) { res.write(`data: ${JSON.stringify({ transcript, is_final: data.is_final, speech_final: data.speech_final, })}\n\n`); } }); // Client provides audio via a paired WebSocket (see browser setup) req.on('close', () => { connection.finish(); }); });
typescript// Deepgram closes idle connections after ~10s of no audio. // Send KeepAlive messages during silence periods. connection.on(LiveTranscriptionEvents.Open, () => { const keepAliveInterval = setInterval(() => { if (connection.getReadyState() === 1) { connection.keepAlive(); } }, 8000); // Every 8 seconds connection.on(LiveTranscriptionEvents.Close, () => { clearInterval(keepAliveInterval); }); });
| Issue | Cause | Solution | |-------|-------|----------| | WebSocket closes immediately | Invalid API key or bad encoding params | Check key, verify encoding/sample_rate match audio | | No transcripts received | Audio not being sent or wrong format | Verify connection.send(chunk) is called with raw PCM | | High latency | Network congestion | Use interim_results: true for perceived speed | | rec command not found | Sox not installed | apt install sox or brew install sox | | Connection drops after 10s | No audio + no KeepAlive | Send connection.keepAlive() every 8s | | Garbled output | Sample rate mismatch | Ensure audio sample rate matches sample_rate option |
Proceed to deepgram-data-handling for transcript processing and storage patterns.
| Case | Status | Duration (ms) | Turns | Tokens | Tool calls | ||||||||
|---|---|---|---|---|---|---|---|---|---|---|---|---|---|
| Without | With | Δ | Without | With | Δ | Without | With | Δ | Without | With | Δ | ||
case-01 | fail→pass | 16,387 | 11,320 | -31% | 1 | 1 | 0% | 3,676 | 4,972 | +35% | 0 | 0 | — |
case-02 | fail→pass | 11,267 | 6,366 | -43% | 1 | 1 | 0% | 2,286 | 3,830 | +68% | 0 | 0 | — |
case-03 | fail→pass | 12,526 | 8,088 | -35% | 1 | 1 | 0% | 2,172 | 4,131 | +90% | 0 | 0 | — |
case-04 | pass→pass | 16,405 | 11,919 | -27% | 1 | 1 | 0% | 3,153 | 4,725 | +50% | 0 | 0 | — |
case-05 | pass→pass | 21,925 | 15,147 | -31% | 1 | 1 | 0% | 4,706 | 5,935 | +26% | 0 | 0 | — |
case-06 | fail→fail | 11,783 | 7,276 | -38% | 1 | 1 | 0% | 2,326 | 3,866 | +66% | 0 | 0 | — |
case-07 | pass→pass | 12,769 | 10,342 | -19% | 1 | 1 | 0% | 2,589 | 4,692 | +81% | 0 | 0 | — |
case-08 | pass→pass | 12,149 | 4,949 | -59% | 1 | 1 | 0% | 2,391 | 3,409 | +43% | 0 | 0 | — |
case-09 | pass→pass | 13,067 | 7,945 | -39% | 1 | 1 | 0% | 2,460 | 4,055 | +65% | 0 | 0 | — |
case-10 | pass→pass | 4,500 | 2,960 | -34% | 1 | 1 | 0% | 877 | 3,084 | +252% | 0 | 0 | — |
case-11 | pass→pass | 11,003 | 5,844 | -47% | 1 | 1 | 0% | 2,070 | 3,574 | +73% | 0 | 0 | — |
case-12 | fail→fail | 14,206 | 8,377 | -41% | 1 | 1 | 0% | 2,832 | 4,377 | +55% | 0 | 0 | — |
case-13 | fail→pass | 11,299 | 7,534 | -33% | 1 | 1 | 0% | 2,159 | 4,030 | +87% | 0 | 0 | — |
case-14 | pass→pass | 12,170 | 6,257 | -49% | 1 | 1 | 0% | 2,348 | 3,655 | +56% | 0 | 0 | — |
case-15 | fail→pass | 5,963 | 2,866 | -52% | 1 | 1 | 0% | 1,092 | 3,000 | +175% | 0 | 0 | — |
case-16 | pass→pass | 6,017 | 2,715 | -55% | 1 | 1 | 0% | 1,147 | 2,939 | +156% | 0 | 0 | — |
case-17 | fail→pass | 9,564 | 3,253 | -66% | 1 | 1 | 0% | 1,677 | 3,088 | +84% | 0 | 0 | — |
case-18 | fail→pass | 10,602 | 5,904 | -44% | 1 | 1 | 0% | 1,853 | 3,601 | +94% | 0 | 0 | — |
case-19 | pass→pass | 4,693 | 2,605 | -44% | 1 | 1 | 0% | 724 | 2,896 | +300% | 0 | 0 | — |
case-20 | pass→pass | 4,448 | 2,328 | -48% | 1 | 1 | 0% | 819 | 2,963 | +262% | 0 | 0 | — |
case-21 | pass→pass | 11,655 | 6,708 | -42% | 1 | 1 | 0% | 2,180 | 3,808 | +75% | 0 | 0 | — |
case-22 | pass→pass | 14,170 | 10,142 | -28% | 1 | 1 | 0% | 2,846 | 4,775 | +68% | 0 | 0 | — |
case-23 | pass→pass | 11,481 | 9,573 | -17% | 1 | 1 | 0% | 2,193 | 4,469 | +104% | 0 | 0 | — |
DecimalAI ran this skill against gemini-3.6-flash twice over the same eval suite — once with the skill loaded and once without — and compared the two runs case by case. 23 cases were attempted. The headline lift of +30 percentage points is the difference between those two pass rates over the 23 comparable cases.
Without the skill loaded, the model failed this case. With it loaded, the same prompt on the same model passed. This is one improved case from the latest verified run; every case, including any that regressed, is in the table above.
Other measured skills in the registry, with their headline benchmark lift.