pdf-quiz-generator/backend/app/services/tts_voices.py
Daniel 954b13e7b3 fix: a speech model is added with its voices, and Orpheus is sent where it works
An Orpheus id (groq-orpheus-english) did not start with "local-", so
generate_tts_audio sent it down the OpenAI path and every call failed. It
also could be added with no voice, and the "local-%" filters in /tts/voices
and the default lookup hid any non-local voice from learners even when it
was added and marked default.

services/tts_voices.py is the one table of which voices belong to which
model (Kokoro, Orpheus English/Arabic, Fish), the same table the scribe app
keeps. Anything the table knows, or anything local-*, goes through the
LiteLLM gateway with the options its family needs (Orpheus: wav). Adding a
bare model id creates one row per voice with friendly names, so an
administrator adds "groq-orpheus-english" and six voices appear to test one
by one; a voice from another family is refused, naming the ones that work.
Learners are offered every active voice, each saying which model serves it,
and /tts/speak answers with the media type the model actually returned.
Kitten and Supertonic tables go — those models left the gateway.

Co-Authored-By: Claude Fable 5.1 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_01Dv6sqaY6Vq3ChZHMem3cnU
2026-09-13 05:04:37 +02:00

146 lines
5.5 KiB
Python

"""Which voices belong to which LiteLLM speech model.
A voice is a property of a model, not of the gateway, and the two used to be
treated as interchangeable: a model id like ``groq-orpheus-english`` could be
added with no voice at all, and ``generate_tts_audio`` then sent it down the
OpenAI path because it did not start with ``local-``. Every one of those calls
failed, which is what "voice is not set up properly" was.
The gateway cannot help: ``/model/info`` reports ``audio_speech`` for all of
these and carries no voice field. So the mapping lives here, keyed by family,
and every list was taken from the provider rather than from documentation —
Groq names its own when refused ("voice must be one of the following voices:
[...]"), Fish accepts ``alloy`` and refuses the other OpenAI names, and
Kokoro's come from the speech gateway's ``/v1/audio/voices`` (admin.py asks it
directly and prefers that answer whenever it arrives).
The same table lives in the scribe app (src/utils/ttsProvider.js). Keep the
two in step when a model is added to the gateway.
Model ids keep the app's ``model:voice`` convention (``local-kokoro-tts:am_adam``).
"""
KOKORO_VOICES = [
("am_adam", "Kokoro Adam"), ("am_michael", "Kokoro Michael"),
("af_bella", "Kokoro Bella"), ("af_nicole", "Kokoro Nicole"),
("bf_emma", "Kokoro Emma"), ("bm_lewis", "Kokoro Lewis"),
]
ORPHEUS_ENGLISH_VOICES = [
("autumn", "Orpheus Autumn"), ("diana", "Orpheus Diana"), ("hannah", "Orpheus Hannah"),
("austin", "Orpheus Austin"), ("daniel", "Orpheus Daniel"), ("troy", "Orpheus Troy"),
]
ORPHEUS_ARABIC_VOICES = [
("abdullah", "Orpheus Abdullah"), ("fahad", "Orpheus Fahad"), ("sultan", "Orpheus Sultan"),
("lulwa", "Orpheus Lulwa"), ("noura", "Orpheus Noura"), ("aisha", "Orpheus Aisha"),
]
FISH_VOICES = [("alloy", "Fish Alloy")]
#: family → (gateway ids that belong to it, its voices, options the request needs)
FAMILIES = {
"kokoro": {
"ids": ("local-kokoro-tts",),
"voices": KOKORO_VOICES,
"request": {"response_format": "mp3"},
"media_type": "audio/mpeg",
},
"orpheus-english": {
"ids": ("groq-orpheus-english", "canopylabs/orpheus-v1-english"),
"voices": ORPHEUS_ENGLISH_VOICES,
# Groq refuses mp3 for Orpheus; wav is what it returns.
"request": {"response_format": "wav"},
"media_type": "audio/wav",
},
"orpheus-arabic": {
"ids": ("groq-orpheus-arabic-saudi", "canopylabs/orpheus-arabic-saudi"),
"voices": ORPHEUS_ARABIC_VOICES,
"request": {"response_format": "wav"},
"media_type": "audio/wav",
},
"fish": {
"ids": ("openrouter-fish-s2.1-pro-tts", "fish-audio/s2.1-pro"),
"voices": FISH_VOICES,
"request": {"response_format": "mp3"},
"media_type": "audio/mpeg",
},
}
def split(model_id: str | None) -> tuple[str, str]:
"""``local-kokoro-tts:am_adam`` → (``local-kokoro-tts``, ``am_adam``); no colon → (id, '')."""
text = (model_id or "").strip()
if ":" in text:
model, voice = text.split(":", 1)
return model.strip(), voice.strip()
return text, ""
def family(model: str | None) -> str | None:
model, _ = split(model)
key = model.lower()
for name, spec in FAMILIES.items():
if key in spec["ids"]:
return name
return None
def is_litellm_tts(model_id: str | None) -> bool:
"""Routed through the LiteLLM gateway's /v1/audio/speech.
``local-*`` is the gateway's own convention for models it serves itself;
a known family is one it proxies. Both take the same request.
"""
model, _ = split(model_id)
return model.startswith("local-") or family(model) is not None
def voices_for(model: str | None) -> list[tuple[str, str]]:
"""(voice, friendly name) pairs the model accepts, or [] when unknown."""
spec = FAMILIES.get(family(model) or "")
return list(spec["voices"]) if spec else []
def default_voice(model: str | None) -> str:
voices = voices_for(model)
if voices:
return voices[0][0]
# A local model we have no list for still needs something the gateway
# will take; alloy is what an OpenAI-shaped server defaults to.
return "alloy"
def request_options(model_id: str | None) -> dict:
spec = FAMILIES.get(family(model_id) or "")
return dict(spec["request"]) if spec else {"response_format": "mp3"}
def media_type(model_id: str | None) -> str:
spec = FAMILIES.get(family(model_id) or "")
return spec["media_type"] if spec else "audio/mpeg"
def accepts(model_id: str) -> bool:
"""A ``model:voice`` pair the model will not refuse.
A bare id whose family we know is fine (it gets its first voice); a voice
from another family is the one thing this says no to.
"""
model, voice = split(model_id)
known = voices_for(model)
if not voice or not known:
return True
return voice in {v for v, _ in known}
def expand(model_id: str, name: str | None = None) -> list[tuple[str, str]]:
"""The rows adding ``model_id`` should create: one per voice.
Adding ``groq-orpheus-english`` yields six ``groq-orpheus-english:<voice>``
rows with friendly names, so the admin's "add a model" is what populates
the voices — nobody types six ids. An id that already names a voice, or a
model with no known list, is one row as given.
"""
model, voice = split(model_id)
known = voices_for(model)
if voice or not known:
return [(model_id, name or model_id)]
return [(f"{model}:{v}", label) for v, label in known]