Install any skill in seconds. Free to start, no credit card required.
Get Started Free →Deploy Deepgram integrations to production environments. Use when deploying to cloud platforms, configuring containers, or setting up Deepgram in Docker/Kubernetes/serverless. Trigger: "deploy deepgram", "deepgram docker", "deepgram kubernetes", "deepgram production deploy", "deepgram cloud run", "deepgram lambda".
.claude/skills/jeremylongshore-deepgram-deploy-integration/SKILL.md| Test case | Without → With | Effect | Δ tokens | Δ turns |
|---|---|---|---|---|
| case-05 | ✗→✓ | ▲ Improved | 95% | 0% |
| case-06 | ✗→✓ | ▲ Improved | 70% | 0% |
| case-07 | ✗→✓ | ▲ Improved | 98% | 0% |
| case-10 | ✗→✓ | ▲ Improved | 325% | 0% |
| case-15 | ✗→✓ | ▲ Improved | 53% | 0% |
Deploy Deepgram transcription services to Docker, Kubernetes, AWS Lambda, and Google Cloud Run. Includes production Dockerfile, K8s manifests with secret management, serverless handlers for event-driven transcription, and health check patterns.
dockerfile# Multi-stage build for minimal production image FROM node:20-alpine AS builder WORKDIR /app COPY package*.json ./ RUN npm ci --production=false COPY tsconfig.json ./ COPY src/ ./src/ RUN npm run build FROM node:20-alpine AS runtime # Security: non-root user RUN addgroup -g 1001 -S app && adduser -S app -u 1001 WORKDIR /app # Production dependencies only COPY package*.json ./ RUN npm ci --production && npm cache clean --force # Copy built application COPY --from=builder /app/dist ./dist # Health check (tests Deepgram connectivity) HEALTHCHECK --interval=30s --timeout=10s --start-period=5s --retries=3 \ CMD wget -q --spider http://localhost:3000/health || exit 1 USER app EXPOSE 3000 CMD ["node", "dist/server.js"]
yaml# docker-compose.yml version: '3.8' services: deepgram-service: build: . ports: - "3000:3000" environment: - NODE_ENV=production - DEEPGRAM_API_KEY=${DEEPGRAM_API_KEY} - DEEPGRAM_MODEL=nova-3 healthcheck: test: ["CMD", "wget", "-q", "--spider", "http://localhost:3000/health"] interval: 30s timeout: 10s retries: 3 restart: unless-stopped deploy: resources: limits: memory: 512M cpus: '1.0' redis: image: redis:7-alpine ports: - "6379:6379" volumes: - redis-data:/data volumes: redis-data:
yaml# k8s/deployment.yaml apiVersion: apps/v1 kind: Deployment metadata: name: deepgram-service labels: app: deepgram-service spec: replicas: 3 selector: matchLabels: app: deepgram-service template: metadata: labels: app: deepgram-service spec: containers: - name: deepgram-service image: your-registry/deepgram-service:latest ports: - containerPort: 3000 env: - name: NODE_ENV value: production - name: DEEPGRAM_API_KEY valueFrom: secretKeyRef: name: deepgram-secrets key: api-key - name: DEEPGRAM_MODEL value: nova-3 resources: requests: memory: "256Mi" cpu: "250m" limits: memory: "512Mi" cpu: "1000m" livenessProbe: httpGet: path: /health port: 3000 initialDelaySeconds: 10 periodSeconds: 30 readinessProbe: httpGet: path: /health port: 3000 initialDelaySeconds: 5 periodSeconds: 10 --- apiVersion: v1 kind: Service metadata: name: deepgram-service spec: selector: app: deepgram-service ports: - port: 80 targetPort: 3000 type: ClusterIP --- apiVersion: autoscaling/v2 kind: HorizontalPodAutoscaler metadata: name: deepgram-service-hpa spec: scaleTargetRef: apiVersion: apps/v1 kind: Deployment name: deepgram-service minReplicas: 2 maxReplicas: 10 metrics: - type: Resource resource: name: cpu target: type: Utilization averageUtilization: 70
bash# Create secret kubectl create secret generic deepgram-secrets \ --from-literal=api-key=$DEEPGRAM_API_KEY # Deploy kubectl apply -f k8s/
typescript// lambda/handler.ts import { createClient } from '@deepgram/sdk'; import { S3Client, GetObjectCommand } from '@aws-sdk/client-s3'; import type { S3Event } from 'aws-lambda'; const deepgram = createClient(process.env.DEEPGRAM_API_KEY!); const s3 = new S3Client({}); // Trigger: S3 upload of audio file -> Lambda -> Deepgram -> Store result export async function handler(event: S3Event) { for (const record of event.Records) { const bucket = record.s3.bucket.name; const key = decodeURIComponent(record.s3.object.key); console.log(`Processing: s3://${bucket}/${key}`); // Get audio from S3 const { Body } = await s3.send(new GetObjectCommand({ Bucket: bucket, Key: key })); const audio = Buffer.from(await Body!.transformToByteArray()); // Transcribe const { result, error } = await deepgram.listen.prerecorded.transcribeFile( audio, { model: 'nova-3', smart_format: true, diarize: true, utterances: true, } ); if (error) { console.error(`Transcription failed for ${key}:`, error.message); throw error; } console.log(`Transcribed ${key}: ${result.metadata.duration}s, ` + `${result.results.channels[0].alternatives[0].words?.length} words`); return { statusCode: 200, body: JSON.stringify({ file: key, duration: result.metadata.duration, transcript: result.results.channels[0].alternatives[0].transcript, request_id: result.metadata.request_id, }), }; } }
typescript// server.ts — Cloud Run entry point import express from 'express'; import { createClient } from '@deepgram/sdk'; const app = express(); app.use(express.json({ limit: '50mb' })); const deepgram = createClient(process.env.DEEPGRAM_API_KEY!); app.post('/transcribe', async (req, res) => { try { const { url, model = 'nova-3', diarize = false } = req.body; const { result, error } = await deepgram.listen.prerecorded.transcribeUrl( { url }, { model, smart_format: true, diarize } ); if (error) return res.status(502).json({ error: error.message }); res.json({ transcript: result.results.channels[0].alternatives[0].transcript, confidence: result.results.channels[0].alternatives[0].confidence, duration: result.metadata.duration, request_id: result.metadata.request_id, }); } catch (err: any) { res.status(500).json({ error: err.message }); } }); app.get('/health', async (req, res) => { try { const { error } = await deepgram.manage.getProjects(); res.json({ status: error ? 'degraded' : 'healthy' }); } catch { res.status(503).json({ status: 'unhealthy' }); } }); const port = process.env.PORT || 3000; app.listen(port, () => console.log(`Listening on port ${port}`));
bash# Deploy to Cloud Run gcloud run deploy deepgram-service \ --source . \ --set-env-vars DEEPGRAM_API_KEY=$(gcloud secrets versions access latest --secret deepgram-key) \ --memory 512Mi \ --timeout 300 \ --concurrency 50 \ --min-instances 1 \ --max-instances 10
bash#!/bin/bash set -euo pipefail ENV="${1:?Usage: deploy.sh <staging|production>}" echo "Deploying to $ENV..." # Build npm ci && npm run build && npm test # Build container docker build -t deepgram-service:$ENV . # Deploy based on target case $ENV in staging) kubectl --context staging apply -f k8s/ kubectl --context staging rollout status deployment/deepgram-service ;; production) kubectl --context production apply -f k8s/ kubectl --context production rollout status deployment/deepgram-service ;; esac # Post-deploy smoke test echo "Running smoke test..." ENDPOINT=$(kubectl get svc deepgram-service -o jsonpath='{.status.loadBalancer.ingress[0].ip}') curl -sf "http://$ENDPOINT/health" || { echo "SMOKE TEST FAILED"; exit 1; } echo "Deploy successful."
| Issue | Cause | Solution | |-------|-------|----------| | Container OOM | Memory limit too low | Increase to 512Mi+ | | Health check failing | Service not ready yet | Increase initialDelaySeconds | | Lambda timeout | Audio too long | Increase timeout to 300s, or use callback | | Cloud Run 429 | Too many concurrent requests | Decrease --concurrency flag | | Secret not found | K8s secret missing | Create secret before deploying |
| Case | Status | Duration (ms) | Turns | Tokens | Tool calls | ||||||||
|---|---|---|---|---|---|---|---|---|---|---|---|---|---|
| Without | With | Δ | Without | With | Δ | Without | With | Δ | Without | With | Δ | ||
case-01 | fail→fail | 18,116 | 11,072 | -39% | 1 | 1 | 0% | 3,964 | 5,331 | +34% | 0 | 0 | — |
case-02 | fail→fail | 14,702 | 13,034 | -11% | 1 | 1 | 0% | 3,237 | 5,632 | +74% | 0 | 0 | — |
case-03 | fail→fail | 18,422 | 16,323 | -11% | 1 | 1 | 0% | 4,255 | 6,605 | +55% | 0 | 0 | — |
case-04 | fail→fail | 11,216 | 6,106 | -46% | 1 | 1 | 0% | 2,269 | 3,828 | +69% | 0 | 0 | — |
case-05 | fail→pass | 10,543 | 7,870 | -25% | 1 | 1 | 0% | 2,128 | 4,145 | +95% | 0 | 0 | — |
case-06 | fail→pass | 15,234 | 10,733 | -30% | 1 | 1 | 0% | 2,795 | 4,745 | +70% | 0 | 0 | — |
case-07 | fail→pass | 11,074 | 7,098 | -36% | 1 | 1 | 0% | 2,029 | 4,022 | +98% | 0 | 0 | — |
case-08 | fail→fail | 11,793 | 8,250 | -30% | 1 | 1 | 0% | 2,135 | 4,336 | +103% | 0 | 0 | — |
case-09 | fail→fail | 11,319 | 8,353 | -26% | 1 | 1 | 0% | 2,038 | 4,221 | +107% | 0 | 0 | — |
case-10 | fail→pass | 11,414 | 3,086 | -73% | 1 | 1 | 0% | 740 | 3,142 | +325% | 0 | 0 | — |
case-11 | pass→pass | 15,807 | 10,135 | -36% | 1 | 1 | 0% | 2,861 | 4,559 | +59% | 0 | 0 | — |
case-12 | fail→fail | 9,912 | 7,770 | -22% | 1 | 1 | 0% | 1,731 | 4,154 | +140% | 0 | 0 | — |
case-13 | pass→pass | 12,303 | 6,638 | -46% | 1 | 1 | 0% | 2,409 | 4,106 | +70% | 0 | 0 | — |
case-14 | pass→pass | 15,618 | 12,478 | -20% | 1 | 1 | 0% | 2,952 | 5,133 | +74% | 0 | 0 | — |
case-15 | fail→pass | 12,256 | 3,981 | -68% | 1 | 1 | 0% | 2,270 | 3,467 | +53% | 0 | 0 | — |
case-16 | fail→fail | 6,923 | 5,312 | -23% | 1 | 1 | 0% | 1,361 | 3,688 | +171% | 0 | 0 | — |
case-17 | fail→pass | 11,929 | 4,519 | -62% | 1 | 1 | 0% | 2,229 | 3,548 | +59% | 0 | 0 | — |
case-18 | fail→fail | 10,615 | 12,039 | +13% | 1 | 1 | 0% | 2,132 | 5,068 | +138% | 0 | 0 | — |
case-19 | fail→pass | 4,291 | 2,953 | -31% | 1 | 1 | 0% | 724 | 3,226 | +346% | 0 | 0 | — |
case-20 | pass→pass | 13,882 | 10,447 | -25% | 1 | 1 | 0% | 2,477 | 4,737 | +91% | 0 | 0 | — |
case-21 | fail→pass | 10,645 | 7,568 | -29% | 1 | 1 | 0% | 1,956 | 4,014 | +105% | 0 | 0 | — |
case-22 | fail→pass | 6,360 | 1,844 | -71% | 1 | 1 | 0% | 1,113 | 3,079 | +177% | 0 | 0 | — |
case-23 | pass→pass | 21,096 | 17,098 | -19% | 1 | 1 | 0% | 4,066 | 6,385 | +57% | 0 | 0 | — |
case-24 | pass→pass | 14,175 | 20,360 | +44% | 1 | 1 | 0% | 2,435 | 6,451 | +165% | 0 | 0 | — |
case-25 | pass→pass | 14,533 | 16,226 | +12% | 1 | 1 | 0% | 3,211 | 6,237 | +94% | 0 | 0 | — |
DecimalAI ran this skill against gemini-3.6-flash twice over the same eval suite — once with the skill loaded and once without — and compared the two runs case by case. 25 cases were attempted, and 24 counted toward the lift figure. The other 1 produced results that are not comparable between the two arms, so they are excluded from the headline rather than averaged into it. The headline lift of +36 percentage points is the difference between those two pass rates over the 24 comparable cases.
Without the skill loaded, the model failed this case. With it loaded, the same prompt on the same model passed. This is one improved case from the latest verified run; every case, including any that regressed, is in the table above.
Other measured skills in the registry, with their headline benchmark lift.