From 3da61c455e73745c45664803d12e4fce8d3568b3 Mon Sep 17 00:00:00 2001 From: Daniel Onyejesi Date: Wed, 25 Mar 2026 20:35:24 +0000 Subject: [PATCH] Add ffmpeg audio conversion fallback for AWS Transcribe MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit - transcribeAWS.js: convert browser WebM/Opus → PCM 16kHz mono via ffmpeg before sending to AWS Transcribe — PCM is unambiguous and most reliable; gracefully falls back to ogg-opus if ffmpeg absent - Dockerfile: install ffmpeg (apk add ffmpeg) so Docker image works out of the box with AWS Transcribe - README: document Amazon Transcribe setup, ffmpeg requirement, Transcribe Medical specialty options, and env vars reference --- Dockerfile | 3 ++ README.md | 43 +++++++++++++++++++-- src/utils/transcribeAWS.js | 76 ++++++++++++++++++++++++++++++++------ 3 files changed, 108 insertions(+), 14 deletions(-) diff --git a/Dockerfile b/Dockerfile index fac309a..75d7c78 100644 --- a/Dockerfile +++ b/Dockerfile @@ -2,6 +2,9 @@ FROM node:20-alpine WORKDIR /app +# ffmpeg: audio conversion for AWS Transcribe (WebM → PCM) +RUN apk add --no-cache ffmpeg + COPY package.json ./ RUN npm install --omit=dev diff --git a/README.md b/README.md index 5d5e265..f85c896 100644 --- a/README.md +++ b/README.md @@ -157,12 +157,46 @@ AZURE_OPENAI_API_VERSION=2024-02-01 --- -## Whisper Transcription +## Transcription (Speech-to-Text) -Always uses OpenAI Whisper regardless of the AI provider setting: +Two providers supported. The app auto-selects AWS Transcribe when `AWS_BEDROCK_REGION` is set, otherwise falls back to OpenAI Whisper. + +### Amazon Transcribe (recommended for clinical use — HIPAA eligible) + +Uses your existing Bedrock AWS credentials. No S3 bucket required — audio streams directly to AWS. + +**Requires `ffmpeg` installed on the server** (handles audio format conversion from browser WebM to PCM). The Docker image includes ffmpeg automatically. + +```env +# Auto-enabled when AWS_BEDROCK_REGION is set. +# To force it explicitly: +TRANSCRIBE_PROVIDER=aws + +# Amazon Transcribe Medical — trained on clinical speech. +# Knows drug names, diagnoses, procedures. Recommended. +AWS_TRANSCRIBE_MEDICAL=true +AWS_TRANSCRIBE_SPECIALTY=PRIMARYCARE +# Other specialty options: CARDIOLOGY, NEUROLOGY, ONCOLOGY, RADIOLOGY, UROLOGY +``` + +Install ffmpeg on non-Docker servers: +```bash +# Ubuntu/Debian +sudo apt-get install -y ffmpeg + +# Amazon Linux / RHEL +sudo yum install -y ffmpeg + +# macOS +brew install ffmpeg +``` + +### OpenAI Whisper (default fallback) ```env OPENAI_API_KEY=sk-... +# Optionally force Whisper even when AWS is configured: +# TRANSCRIBE_PROVIDER=openai ``` --- @@ -193,7 +227,10 @@ SMTP_FROM=noreply@yourdomain.com | `AZURE_OPENAI_ENDPOINT` | If using Azure | Azure OpenAI endpoint URL | | `AZURE_OPENAI_API_KEY` | If using Azure | Azure API key | | `AZURE_DEPLOYMENT_NAME` | If using Azure | Deployment name, e.g. `gpt-4o-mini` | -| `OPENAI_API_KEY` | For transcription | OpenAI key (Whisper) | +| `OPENAI_API_KEY` | For Whisper transcription | OpenAI key (Whisper fallback) | +| `TRANSCRIBE_PROVIDER` | No | `aws` or `openai` — auto-detected from AWS config | +| `AWS_TRANSCRIBE_MEDICAL` | No | `true` to use Transcribe Medical (clinical accuracy) | +| `AWS_TRANSCRIBE_SPECIALTY` | No | `PRIMARYCARE` (default), `CARDIOLOGY`, `NEUROLOGY`, etc. | | `ELEVENLABS_API_KEY` | No | ElevenLabs TTS (optional) | | `JWT_SECRET` | **Yes** | Random 64-char string — keep secret | | `DATABASE_URL` | No | PostgreSQL URL (auto-set by docker-compose) | diff --git a/src/utils/transcribeAWS.js b/src/utils/transcribeAWS.js index 96d08a0..894f9d7 100644 --- a/src/utils/transcribeAWS.js +++ b/src/utils/transcribeAWS.js @@ -2,6 +2,10 @@ // TRANSCRIBE-AWS.JS — Amazon Transcribe Streaming // No S3 required. Audio streams directly to AWS. // +// Audio conversion: ffmpeg converts browser WebM/Opus → PCM 16kHz +// which is the most reliable format for AWS Transcribe. If ffmpeg +// is not installed, falls back to sending ogg-opus directly. +// // Env vars: // TRANSCRIBE_PROVIDER=aws — use this instead of Whisper // AWS_TRANSCRIBE_MEDICAL=true — use Transcribe Medical (better @@ -13,12 +17,50 @@ // AWS_ACCESS_KEY_ID / AWS_SECRET_ACCESS_KEY — reused credentials // ============================================================ +const { spawn } = require('child_process'); + const CHUNK_SIZE = 32768; // 32 KB per audio chunk +// Convert any browser audio (WebM, OGG, etc.) to raw PCM s16le 16kHz mono +// using ffmpeg. Returns a Buffer of raw PCM bytes. +// Throws if ffmpeg is not installed. +function convertToPCM(inputBuffer) { + return new Promise(function(resolve, reject) { + var ff = spawn('ffmpeg', [ + '-i', 'pipe:0', // read from stdin + '-f', 's16le', // raw signed 16-bit little-endian PCM + '-ar', '16000', // 16 kHz — ideal for speech recognition + '-ac', '1', // mono + '-acodec', 'pcm_s16le', + 'pipe:1' // write to stdout + ]); + + var out = []; + var errBuf = []; + + ff.stdout.on('data', function(d) { out.push(d); }); + ff.stderr.on('data', function(d) { errBuf.push(d); }); // ffmpeg logs to stderr, not an error + + ff.on('close', function(code) { + if (code !== 0) { + reject(new Error('ffmpeg conversion failed (code ' + code + '): ' + Buffer.concat(errBuf).toString().slice(-200))); + return; + } + resolve(Buffer.concat(out)); + }); + + ff.on('error', function(err) { + reject(new Error('ffmpeg not found. Install ffmpeg on the server: ' + err.message)); + }); + + ff.stdin.write(inputBuffer); + ff.stdin.end(); + }); +} + async function* makeAudioStream(buffer) { for (var i = 0; i < buffer.length; i += CHUNK_SIZE) { yield { AudioEvent: { AudioChunk: buffer.slice(i, i + CHUNK_SIZE) } }; - // Yield to event loop between chunks to avoid blocking await new Promise(function(r) { setTimeout(r, 0); }); } } @@ -35,27 +77,39 @@ async function transcribeWithAWS(audioBuffer, mimeType) { var client = new TranscribeStreamingClient({ region: region, credentials: credentials }); - // Map browser MIME type to AWS MediaEncoding - // audio/webm;codecs=opus → ogg-opus (same Opus codec, different container) - // audio/ogg;codecs=opus → ogg-opus (exact match) - // audio/wav → pcm + // Try ffmpeg conversion to PCM (most reliable with AWS Transcribe). + // Falls back to ogg-opus passthrough if ffmpeg is not installed. + var audioToSend = audioBuffer; var mediaEncoding = 'ogg-opus'; - if (mimeType && mimeType.indexOf('wav') !== -1) mediaEncoding = 'pcm'; + var sampleRate = 48000; + + try { + audioToSend = await convertToPCM(audioBuffer); + mediaEncoding = 'pcm'; + sampleRate = 16000; + console.log('[Transcribe] ffmpeg conversion OK — PCM 16kHz, ' + audioToSend.length + ' bytes'); + } catch (ffErr) { + console.warn('[Transcribe] ffmpeg unavailable, sending ogg-opus directly:', ffErr.message); + // For WAV input, try pcm anyway without ffmpeg + if (mimeType && mimeType.indexOf('wav') !== -1) { + mediaEncoding = 'pcm'; + sampleRate = 16000; + } + } var useMedical = process.env.AWS_TRANSCRIBE_MEDICAL === 'true'; var specialty = process.env.AWS_TRANSCRIBE_SPECIALTY || 'PRIMARYCARE'; - var transcript = ''; if (useMedical) { var StartMedicalStreamTranscriptionCommand = TranscribeModule.StartMedicalStreamTranscriptionCommand; var medCmd = new StartMedicalStreamTranscriptionCommand({ LanguageCode: 'en-US', - MediaSampleRateHertz: 48000, + MediaSampleRateHertz: sampleRate, MediaEncoding: mediaEncoding, Specialty: specialty, Type: 'DICTATION', - AudioStream: makeAudioStream(audioBuffer) + AudioStream: makeAudioStream(audioToSend) }); var medResp = await client.send(medCmd); for await (var event of medResp.TranscriptResultStream) { @@ -73,9 +127,9 @@ async function transcribeWithAWS(audioBuffer, mimeType) { var StartStreamTranscriptionCommand = TranscribeModule.StartStreamTranscriptionCommand; var cmd = new StartStreamTranscriptionCommand({ LanguageCode: 'en-US', - MediaSampleRateHertz: 48000, + MediaSampleRateHertz: sampleRate, MediaEncoding: mediaEncoding, - AudioStream: makeAudioStream(audioBuffer) + AudioStream: makeAudioStream(audioToSend) }); var resp = await client.send(cmd); for await (var ev of resp.TranscriptResultStream) {