pediatric-ai-scribe-v3/src/utils/ttsProvider.js
Daniel 07161c6fa8
Some checks failed
Forgejo Docker Build / Root app tests (push) Successful in 47s
Forgejo Docker Build / Build Docker image (push) Successful in 8s
Forgejo Docker Build / End-to-end (browser) (push) Failing after 6s
feat: speech models have a roster; each brings its voices, users choose across all of them
Discover lists speech models with the voices each accepts and a + Add that
puts the model on tts.roster. The Roster card lists every model with a voice
picker, Test, Make default and Remove. Test on any row (or a discovered model
not yet added) fills the test panel's voice list with that model's voices, so
Orpheus and Kokoro can be heard one voice at a time before either is chosen.

The default is a pair — PUT /config/tts/default sets tts.model and tts.voice
together and refuses a voice the model does not accept, naming the ones it
does. The generic setter no longer takes tts.model/tts.voice one at a time,
which is how a Kokoro voice got paired with Orpheus. A default that leaves
the roster stops being the default.

Users pick from the voices of every roster model, grouped by model in
Settings; the stored value is "model|voice" so read-aloud sends the voice to
the model that accepts it. A bare voice saved before there was a roster is
read as a voice of the default model. chooseTTS is the one place the pair is
decided, shared by read-aloud, the admin test and the settings options.

Co-Authored-By: Claude Fable 5.1 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_01Dv6sqaY6Vq3ChZHMem3cnU
2026-09-13 04:49:24 +02:00

262 lines
11 KiB
JavaScript

const { getLiteLLMHeaders } = require('./litellm');
function parseList(value) {
return String(value || '')
.split(',')
.map(function(item) { return item.trim(); })
.filter(Boolean);
}
// Kitten and Supertonic were retired from the gateway in favour of Kokoro, so
// their voice lists went with them. Voices are curated per family rather than
// discovered: no TTS provider exposes its voices consistently, and a list
// guessed from a model id is how a picker ends up offering a voice the provider
// will reject.
var GROQ_ORPHEUS_ENGLISH_VOICES = ['autumn', 'diana', 'hannah', 'austin', 'daniel', 'troy'];
var GROQ_ORPHEUS_ARABIC_VOICES = ['abdullah', 'fahad', 'sultan', 'lulwa', 'noura', 'aisha'];
// ── Which voices belong to which model ───────────────────────────────────
// A voice is not a property of the gateway, it is a property of the model, and
// the two were being listed side by side as if interchangeable. Picking Orpheus
// and testing it with a Kokoro voice is not a configuration, it is an error —
// every one of those calls came back 500, which is what "these settings don't
// work" was.
//
// The gateway cannot help: /model/info reports mode audio_speech for all four
// of these and carries no voice field at all. So the mapping lives here.
//
// Every list below was taken from the provider rather than from documentation.
// Groq states its own when refused: "voice must be one of the following
// voices: [...]". Fish was found by trying the OpenAI names — alloy returns
// audio and the rest are refused. Kokoro's come from the speech gateway's own
// /v1/audio/voices, which is authoritative and is preferred over this list
// whenever it answers.
// Keyed by family, not by id: getLiteLLMTTSModelFamily already maps both the
// gateway alias (groq-orpheus-english) and the upstream id
// (canopylabs/orpheus-v1-english) to one family, and a table keyed by id would
// have to repeat every list under every spelling.
var MODEL_VOICES = Object.freeze({
'groq-orpheus-english': Object.freeze(GROQ_ORPHEUS_ENGLISH_VOICES.slice()),
'groq-orpheus-arabic': Object.freeze(GROQ_ORPHEUS_ARABIC_VOICES.slice()),
'fish': Object.freeze(['alloy']),
'kokoro': Object.freeze([
'sherpa/kokoro:am_adam', 'sherpa/kokoro:am_michael', 'sherpa/kokoro:af_bella',
'sherpa/kokoro:af_nicole', 'sherpa/kokoro:bf_emma', 'sherpa/kokoro:bm_lewis'
])
});
/**
* The voices a model will actually accept, or [] when we do not know.
*
* [] is an honest answer and a useful one: the screen can say "no voice list
* for this model, type one" instead of offering another model's voices, which
* is the thing that was broken.
*/
function voicesForModel(model) {
var id = String(model || '');
// The environment wins for the model it was written for. LITELLM_TTS_VOICES
// names the voices of LITELLM_TTS_MODEL, so a deployment can change that
// model's voices without changing this file — which matters for the local
// gateway, whose list is whatever it has been built with. It answers for
// that model only: it never described any other, and treating it as a
// general list is what offered Kokoro's voices for Fish.
if (id && id === String(process.env.LITELLM_TTS_MODEL || '')) {
var configured = parseList(process.env.LITELLM_TTS_VOICES);
if (configured.length) return configured;
}
var known = MODEL_VOICES[getLiteLLMTTSModelFamily(id)];
return known ? known.slice() : [];
}
function uniqueList(values) {
var seen = new Set();
return (values || []).filter(function(value) {
if (!value || seen.has(value)) return false;
seen.add(value);
return true;
});
}
function getTTSProvider() {
var env = process.env.TTS_PROVIDER;
if (env === 'litellm') return 'litellm';
if (process.env.LITELLM_API_BASE) return 'litellm';
return 'none';
}
function getTTSEnvProvider() {
return process.env.TTS_PROVIDER || 'auto';
}
function getTTSVoiceLists() {
return {
litellm: uniqueList(parseList(process.env.LITELLM_TTS_VOICES).concat(GROQ_ORPHEUS_ENGLISH_VOICES, GROQ_ORPHEUS_ARABIC_VOICES))
};
}
function getLiteLLMTTSModelFamily(model) {
var id = String(model || '').toLowerCase();
if (id === 'local-kokoro-tts') return 'kokoro';
if (id === 'groq-orpheus-english' || id === 'canopylabs/orpheus-v1-english') return 'groq-orpheus-english';
if (id === 'groq-orpheus-arabic-saudi' || id === 'canopylabs/orpheus-arabic-saudi') return 'groq-orpheus-arabic';
if (id === 'openrouter-fish-s2.1-pro-tts' || id === 'fish-audio/s2.1-pro') return 'fish';
return 'unknown';
}
function getLiteLLMTTSRequestOptions(model) {
var family = getLiteLLMTTSModelFamily(model);
if (family === 'groq-orpheus-english' || family === 'groq-orpheus-arabic') {
return { response_format: 'wav' };
}
return {};
}
function isLiteLLMTTSVoiceCompatible(model, voice) {
if (typeof voice !== 'string' || !voice.trim()) return false;
var known = voicesForModel(model);
// A model we have a list for accepts what is on it and nothing else — that
// refusal is the whole point, and it is what stops a Kokoro voice being
// offered for Orpheus.
if (known.length) return known.indexOf(String(voice)) !== -1 || known.indexOf(String(voice).toLowerCase()) !== -1;
// A model we have no list for: anything is allowed, because refusing would
// mean refusing every voice of a model added after this code was written.
// Another model's voice is still refused — those we do know to be wrong.
var family = getLiteLLMTTSModelFamily(model);
return !Object.keys(MODEL_VOICES).some(function (other) {
return other !== family && MODEL_VOICES[other].indexOf(String(voice).toLowerCase()) !== -1;
});
}
function getLiteLLMTTSVoicesForModel(model, opts) {
opts = opts || {};
// One table, MODEL_VOICES, rather than a second copy of the same knowledge.
// The old branch fell through to LITELLM_TTS_VOICES for any model it did not
// recognise, so an unknown model was offered Kokoro's voices — which is how
// Fish came to be listed with six voices it refuses.
var voices = voicesForModel(model);
[opts.currentVoice, process.env.LITELLM_TTS_VOICE].forEach(function(voice) {
if (isLiteLLMTTSVoiceCompatible(model, voice)) voices.push(voice);
});
return uniqueList(voices.filter(function(voice) { return isLiteLLMTTSVoiceCompatible(model, voice); }));
}
function isLiteLLMTTSModel(model) {
var mode = model && model.model_info && model.model_info.mode ? String(model.model_info.mode) : '';
return mode === 'audio_speech';
}
function getLiteLLMTTSModels(models) {
return (models || [])
.filter(isLiteLLMTTSModel)
.map(function(model) { return model && (model.id || model.model_name) ? (model.id || model.model_name) : String(model || ''); });
}
function pushUniqueTTSItem(items, item) {
if (!item || !item.id) return;
if (items.some(function(existing) { return existing.id === item.id; })) return;
items.push(item);
}
/**
* The speech models on offer, each carrying the voices it accepts.
*
* Discovery used to list models and voices side by side in one flat list —
* twelve Orpheus voices, six Kokoro ones and a "configured-voice" row that
* had lost the model it belonged to, all with a Make default button. A voice
* is a property of a model, so it is listed under one: the screen adds a
* model to the roster and the voices come with it.
*/
function getLiteLLMTTSDiscoveryItems(models, opts) {
opts = opts || {};
var items = [];
getLiteLLMTTSModels(models).forEach(function(id) {
pushUniqueTTSItem(items, { id: id, name: id, source: 'gateway-api', kind: 'model', voices: voicesForModel(id) });
});
// The default and everything on the roster are still models even when the
// gateway's metadata call failed, or when the gateway stopped advertising
// one that is still routable.
[opts.currentModel].concat(opts.roster || []).forEach(function(id) {
if (!id) return;
pushUniqueTTSItem(items, { id: id, name: id, source: 'configured-model', kind: 'model', voices: voicesForModel(id) });
});
return items;
}
// ── Roster ───────────────────────────────────────────────────────────────
// A voice is only meaningful together with its model, so a stored choice —
// the admin default, a user's preference — names both. The pair travels as
// "model|voice": "|" appears in no gateway id and no voice name, unlike ":"
// which Kokoro uses (sherpa/kokoro:af_bella) and "/" which every upstream id
// does. A bare voice with no "|" is a value saved before there was a roster,
// and is read as a voice of the default model.
function voiceRef(model, voice) {
if (!model || !voice) return '';
return String(model) + '|' + String(voice);
}
function parseVoiceRef(value) {
var text = typeof value === 'string' ? value.trim() : '';
if (!text) return { model: '', voice: '' };
var at = text.indexOf('|');
if (at === -1) return { model: '', voice: text };
return { model: text.slice(0, at).trim(), voice: text.slice(at + 1).trim() };
}
/** Every voice of every roster model, in roster order, as picker options. */
function rosterVoices(roster) {
var out = [];
uniqueList(roster).forEach(function(model) {
voicesForModel(model).forEach(function(voice) {
out.push({ model: model, voice: voice, value: voiceRef(model, voice) });
});
});
return out;
}
/**
* The model and voice a request will use.
*
* One decision for the read-aloud route, the admin test and the settings
* page, so they cannot disagree. `preferred` is a voice ref (or a bare legacy
* voice) and wins when it names a roster model and a voice that model accepts;
* anything else falls through to the default pair, and the default's voice
* falls through to the first voice its model has. The one thing this never
* does is send a voice to a model that will refuse it.
*/
function chooseTTS(opts) {
opts = opts || {};
var roster = uniqueList([opts.defaultModel].concat(opts.roster || []));
var want = parseVoiceRef(opts.preferred);
if (want.voice) {
var model = want.model || opts.defaultModel || '';
if (roster.indexOf(model) !== -1 && isLiteLLMTTSVoiceCompatible(model, want.voice)) {
return { model: model, voice: want.voice };
}
}
var fallback = opts.defaultModel || '';
var voice = [opts.defaultVoice, opts.envVoice].concat(voicesForModel(fallback)).find(function(candidate) {
return isLiteLLMTTSVoiceCompatible(fallback, candidate);
}) || '';
return { model: fallback, voice: voice };
}
module.exports = {
voicesForModel,
MODEL_VOICES,
voiceRef,
parseVoiceRef,
rosterVoices,
chooseTTS,
getTTSEnvProvider,
getLiteLLMTTSDiscoveryItems,
getLiteLLMHeaders,
getLiteLLMTTSModels,
getLiteLLMTTSRequestOptions,
getLiteLLMTTSVoicesForModel,
getTTSProvider,
getTTSVoiceLists,
isLiteLLMTTSVoiceCompatible,
isLiteLLMTTSModel
};