feat: Refine TTS retry logic to handle client-side transport failures and enhance voice management within the TTS context.
This commit is contained in:
parent
cbd8aa6240
commit
f5cd3b262a
7 changed files with 77 additions and 25 deletions
|
|
@ -13,6 +13,9 @@ import { isAuthEnabled } from '@/lib/server/auth/config';
|
||||||
import { getClientIp } from '@/lib/server/rate-limit/request-ip';
|
import { getClientIp } from '@/lib/server/rate-limit/request-ip';
|
||||||
import { getOrCreateDeviceId, setDeviceIdCookie } from '@/lib/server/rate-limit/device-id';
|
import { getOrCreateDeviceId, setDeviceIdCookie } from '@/lib/server/rate-limit/device-id';
|
||||||
|
|
||||||
|
export const runtime = 'nodejs';
|
||||||
|
export const maxDuration = 60;
|
||||||
|
|
||||||
function attachDeviceIdCookie(response: NextResponse, deviceId: string | null, didCreate: boolean) {
|
function attachDeviceIdCookie(response: NextResponse, deviceId: string | null, didCreate: boolean) {
|
||||||
if (didCreate && deviceId) {
|
if (didCreate && deviceId) {
|
||||||
setDeviceIdCookie(response, deviceId);
|
setDeviceIdCookie(response, deviceId);
|
||||||
|
|
@ -237,6 +240,10 @@ export async function POST(req: NextRequest) {
|
||||||
const openai = new OpenAI({
|
const openai = new OpenAI({
|
||||||
apiKey: openApiKey,
|
apiKey: openApiKey,
|
||||||
baseURL: openApiBaseUrl,
|
baseURL: openApiBaseUrl,
|
||||||
|
// Keep retry policy centralized in this route's fetchTTSBufferWithRetry.
|
||||||
|
maxRetries: 0,
|
||||||
|
// Keep upstream timeout below route max duration so we can return a controlled error.
|
||||||
|
timeout: Number(process.env.TTS_UPSTREAM_TIMEOUT_MS ?? 45_000),
|
||||||
});
|
});
|
||||||
|
|
||||||
const normalizedVoice = (
|
const normalizedVoice = (
|
||||||
|
|
|
||||||
|
|
@ -1,6 +1,7 @@
|
||||||
'use client';
|
'use client';
|
||||||
|
|
||||||
import { useConfig } from '@/contexts/ConfigContext';
|
import { useConfig } from '@/contexts/ConfigContext';
|
||||||
|
import { useTTS } from '@/contexts/TTSContext';
|
||||||
import { useCallback } from 'react';
|
import { useCallback } from 'react';
|
||||||
import { VoicesControlBase } from '@/components/player/VoicesControlBase';
|
import { VoicesControlBase } from '@/components/player/VoicesControlBase';
|
||||||
|
|
||||||
|
|
@ -8,7 +9,8 @@ export const VoicesControl = ({ availableVoices, setVoiceAndRestart }: {
|
||||||
availableVoices: string[];
|
availableVoices: string[];
|
||||||
setVoiceAndRestart: (voice: string) => void;
|
setVoiceAndRestart: (voice: string) => void;
|
||||||
}) => {
|
}) => {
|
||||||
const { voice, ttsModel, ttsProvider } = useConfig();
|
const { ttsModel, ttsProvider } = useConfig();
|
||||||
|
const { voice } = useTTS();
|
||||||
const onChangeVoice = useCallback((nextVoice: string) => setVoiceAndRestart(nextVoice), [setVoiceAndRestart]);
|
const onChangeVoice = useCallback((nextVoice: string) => setVoiceAndRestart(nextVoice), [setVoiceAndRestart]);
|
||||||
|
|
||||||
return (
|
return (
|
||||||
|
|
|
||||||
|
|
@ -457,11 +457,12 @@ export function EPUBProvider({ children }: { children: ReactNode }) {
|
||||||
instructions: effectiveModel === 'gpt-4o-mini-tts' ? ttsInstructions : undefined
|
instructions: effectiveModel === 'gpt-4o-mini-tts' ? ttsInstructions : undefined
|
||||||
};
|
};
|
||||||
|
|
||||||
|
// Allow one narrow client retry for transient browser->/api/tts transport failures.
|
||||||
|
// HTTP failures are not retried client-side.
|
||||||
const retryOptions: TTSRetryOptions = {
|
const retryOptions: TTSRetryOptions = {
|
||||||
maxRetries: 2,
|
maxRetries: 2,
|
||||||
initialDelay: 5000,
|
initialDelay: 300,
|
||||||
maxDelay: 10000,
|
maxDelay: 300,
|
||||||
backoffFactor: 2
|
|
||||||
};
|
};
|
||||||
|
|
||||||
const audioBuffer = await withRetry(
|
const audioBuffer = await withRetry(
|
||||||
|
|
@ -623,11 +624,12 @@ export function EPUBProvider({ children }: { children: ReactNode }) {
|
||||||
instructions: effectiveModel === 'gpt-4o-mini-tts' ? ttsInstructions : undefined
|
instructions: effectiveModel === 'gpt-4o-mini-tts' ? ttsInstructions : undefined
|
||||||
};
|
};
|
||||||
|
|
||||||
|
// Allow one narrow client retry for transient browser->/api/tts transport failures.
|
||||||
|
// HTTP failures are not retried client-side.
|
||||||
const retryOptions: TTSRetryOptions = {
|
const retryOptions: TTSRetryOptions = {
|
||||||
maxRetries: 2,
|
maxRetries: 2,
|
||||||
initialDelay: 5000,
|
initialDelay: 300,
|
||||||
maxDelay: 10000,
|
maxDelay: 300,
|
||||||
backoffFactor: 2
|
|
||||||
};
|
};
|
||||||
|
|
||||||
const audioBuffer = await withRetry(
|
const audioBuffer = await withRetry(
|
||||||
|
|
|
||||||
|
|
@ -513,11 +513,12 @@ export function PDFProvider({ children }: { children: ReactNode }) {
|
||||||
instructions: effectiveModel === 'gpt-4o-mini-tts' ? ttsInstructions : undefined
|
instructions: effectiveModel === 'gpt-4o-mini-tts' ? ttsInstructions : undefined
|
||||||
};
|
};
|
||||||
|
|
||||||
|
// Allow one narrow client retry for transient browser->/api/tts transport failures.
|
||||||
|
// HTTP failures are not retried client-side.
|
||||||
const retryOptions: TTSRetryOptions = {
|
const retryOptions: TTSRetryOptions = {
|
||||||
maxRetries: 3,
|
maxRetries: 2,
|
||||||
initialDelay: 1000,
|
initialDelay: 300,
|
||||||
maxDelay: 5000,
|
maxDelay: 300,
|
||||||
backoffFactor: 2
|
|
||||||
};
|
};
|
||||||
|
|
||||||
try {
|
try {
|
||||||
|
|
@ -686,11 +687,12 @@ export function PDFProvider({ children }: { children: ReactNode }) {
|
||||||
instructions: effectiveModel === 'gpt-4o-mini-tts' ? ttsInstructions : undefined
|
instructions: effectiveModel === 'gpt-4o-mini-tts' ? ttsInstructions : undefined
|
||||||
};
|
};
|
||||||
|
|
||||||
|
// Allow one narrow client retry for transient browser->/api/tts transport failures.
|
||||||
|
// HTTP failures are not retried client-side.
|
||||||
const retryOptions: TTSRetryOptions = {
|
const retryOptions: TTSRetryOptions = {
|
||||||
maxRetries: 3,
|
maxRetries: 2,
|
||||||
initialDelay: 1000,
|
initialDelay: 300,
|
||||||
maxDelay: 5000,
|
maxDelay: 300,
|
||||||
backoffFactor: 2
|
|
||||||
};
|
};
|
||||||
|
|
||||||
const audioBuffer: TTSAudioBuffer = await withRetry(
|
const audioBuffer: TTSAudioBuffer = await withRetry(
|
||||||
|
|
|
||||||
|
|
@ -67,6 +67,7 @@ declare global {
|
||||||
*/
|
*/
|
||||||
interface TTSContextType extends TTSPlaybackState {
|
interface TTSContextType extends TTSPlaybackState {
|
||||||
// Voice settings
|
// Voice settings
|
||||||
|
voice: string;
|
||||||
availableVoices: string[];
|
availableVoices: string[];
|
||||||
|
|
||||||
// Alignment metadata for the current sentence
|
// Alignment metadata for the current sentence
|
||||||
|
|
@ -905,23 +906,33 @@ export function TTSProvider({ children }: { children: ReactNode }): ReactElement
|
||||||
if (availableVoices.length > 0) {
|
if (availableVoices.length > 0) {
|
||||||
// Allow Kokoro multi-voice strings (e.g., "voice1(0.5)+voice2(0.5)") for any provider
|
// Allow Kokoro multi-voice strings (e.g., "voice1(0.5)+voice2(0.5)") for any provider
|
||||||
const isKokoro = isKokoroModel(configTTSModel);
|
const isKokoro = isKokoroModel(configTTSModel);
|
||||||
|
const fallbackVoice = configVoice || availableVoices[0];
|
||||||
|
|
||||||
if (isKokoro) {
|
if (isKokoro) {
|
||||||
// If Kokoro and we have any voice string (including plus/weights), don't override it.
|
// If Kokoro and we have any voice string (including plus/weights), don't override it.
|
||||||
// Only default when voice is empty.
|
// Only default when local voice is empty.
|
||||||
if (!voice) {
|
if (!voice) {
|
||||||
setVoice(availableVoices[0]);
|
setVoice(fallbackVoice);
|
||||||
}
|
}
|
||||||
return;
|
return;
|
||||||
}
|
}
|
||||||
|
|
||||||
if (!voice || !availableVoices.includes(voice)) {
|
// For non-Kokoro, only force a fallback when there is no active local voice.
|
||||||
|
// If a persisted config voice exists, keep it rather than overriding from a
|
||||||
|
// potentially stale in-flight voices response during reload.
|
||||||
|
if (!voice) {
|
||||||
|
console.log(`Voice "${voice || '(empty)'}" not set. Using "${fallbackVoice}"`);
|
||||||
|
setVoice(fallbackVoice);
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
|
||||||
|
if (!configVoice && !availableVoices.includes(voice)) {
|
||||||
console.log(`Voice "${voice || '(empty)'}" not found in available voices. Using "${availableVoices[0]}"`);
|
console.log(`Voice "${voice || '(empty)'}" not found in available voices. Using "${availableVoices[0]}"`);
|
||||||
setVoice(availableVoices[0]);
|
setVoice(availableVoices[0]);
|
||||||
// Don't save to config - just use it temporarily until user explicitly selects one
|
// Don't save to config - just use it temporarily until user explicitly selects one
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
}, [availableVoices, voice, configTTSModel]);
|
}, [availableVoices, voice, configVoice, configTTSModel]);
|
||||||
|
|
||||||
/**
|
/**
|
||||||
* Generates and plays audio for the current sentence
|
* Generates and plays audio for the current sentence
|
||||||
|
|
@ -1029,11 +1040,12 @@ export function TTSProvider({ children }: { children: ReactNode }): ReactElement
|
||||||
instructions: ttsModel === 'gpt-4o-mini-tts' ? ttsInstructions : undefined,
|
instructions: ttsModel === 'gpt-4o-mini-tts' ? ttsInstructions : undefined,
|
||||||
};
|
};
|
||||||
|
|
||||||
|
// Allow one narrow client retry for transient browser->/api/tts transport failures.
|
||||||
|
// HTTP failures are not retried client-side.
|
||||||
const retryOptions: TTSRetryOptions = {
|
const retryOptions: TTSRetryOptions = {
|
||||||
maxRetries: 3,
|
maxRetries: 2,
|
||||||
initialDelay: 1000,
|
initialDelay: 300,
|
||||||
maxDelay: 5000,
|
maxDelay: 300,
|
||||||
backoffFactor: 2
|
|
||||||
};
|
};
|
||||||
|
|
||||||
onTTSStart();
|
onTTSStart();
|
||||||
|
|
@ -1727,6 +1739,7 @@ export function TTSProvider({ children }: { children: ReactNode }): ReactElement
|
||||||
currDocPage,
|
currDocPage,
|
||||||
currDocPageNumber,
|
currDocPageNumber,
|
||||||
currDocPages,
|
currDocPages,
|
||||||
|
voice,
|
||||||
availableVoices,
|
availableVoices,
|
||||||
togglePlay,
|
togglePlay,
|
||||||
skipForward,
|
skipForward,
|
||||||
|
|
@ -1751,6 +1764,7 @@ export function TTSProvider({ children }: { children: ReactNode }): ReactElement
|
||||||
currDocPage,
|
currDocPage,
|
||||||
currDocPageNumber,
|
currDocPageNumber,
|
||||||
currDocPages,
|
currDocPages,
|
||||||
|
voice,
|
||||||
availableVoices,
|
availableVoices,
|
||||||
togglePlay,
|
togglePlay,
|
||||||
skipForward,
|
skipForward,
|
||||||
|
|
|
||||||
|
|
@ -1,6 +1,6 @@
|
||||||
'use client';
|
'use client';
|
||||||
|
|
||||||
import { useState, useCallback } from 'react';
|
import { useState, useCallback, useRef } from 'react';
|
||||||
import { getVoices } from '@/lib/client/api/audiobooks';
|
import { getVoices } from '@/lib/client/api/audiobooks';
|
||||||
|
|
||||||
const DEFAULT_VOICES = ['alloy', 'ash', 'coral', 'echo', 'fable', 'onyx', 'nova', 'sage', 'shimmer'];
|
const DEFAULT_VOICES = ['alloy', 'ash', 'coral', 'echo', 'fable', 'onyx', 'nova', 'sage', 'shimmer'];
|
||||||
|
|
@ -20,8 +20,10 @@ export function useVoiceManagement(
|
||||||
ttsModel: string | undefined
|
ttsModel: string | undefined
|
||||||
) {
|
) {
|
||||||
const [availableVoices, setAvailableVoices] = useState<string[]>([]);
|
const [availableVoices, setAvailableVoices] = useState<string[]>([]);
|
||||||
|
const fetchSeqRef = useRef(0);
|
||||||
|
|
||||||
const fetchVoices = useCallback(async () => {
|
const fetchVoices = useCallback(async () => {
|
||||||
|
const fetchSeq = ++fetchSeqRef.current;
|
||||||
try {
|
try {
|
||||||
console.log('Fetching voices...');
|
console.log('Fetching voices...');
|
||||||
const data = await getVoices({
|
const data = await getVoices({
|
||||||
|
|
@ -32,9 +34,12 @@ export function useVoiceManagement(
|
||||||
'Content-Type': 'application/json',
|
'Content-Type': 'application/json',
|
||||||
});
|
});
|
||||||
|
|
||||||
|
// Ignore stale responses from older provider/model fetches.
|
||||||
|
if (fetchSeq !== fetchSeqRef.current) return;
|
||||||
setAvailableVoices(data.voices || DEFAULT_VOICES);
|
setAvailableVoices(data.voices || DEFAULT_VOICES);
|
||||||
} catch (error) {
|
} catch (error) {
|
||||||
console.error('Error fetching voices:', error);
|
console.error('Error fetching voices:', error);
|
||||||
|
if (fetchSeq !== fetchSeqRef.current) return;
|
||||||
// Set available voices to default openai voices
|
// Set available voices to default openai voices
|
||||||
setAvailableVoices(DEFAULT_VOICES);
|
setAvailableVoices(DEFAULT_VOICES);
|
||||||
}
|
}
|
||||||
|
|
|
||||||
|
|
@ -62,6 +62,26 @@ export const withRetry = async <T>(
|
||||||
break;
|
break;
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// Narrow client retries to transport-level failures only.
|
||||||
|
// If we got an HTTP status from /api/tts, do not retry from the client.
|
||||||
|
// Server-side /api/tts already applies upstream retry logic.
|
||||||
|
const message = lastError.message.toLowerCase();
|
||||||
|
const isTransportFailure =
|
||||||
|
status === undefined &&
|
||||||
|
(
|
||||||
|
lastError instanceof TypeError ||
|
||||||
|
message.includes('failed to fetch') ||
|
||||||
|
message.includes('networkerror') ||
|
||||||
|
message.includes('network request failed') ||
|
||||||
|
message.includes('load failed') ||
|
||||||
|
message.includes('fetch failed') ||
|
||||||
|
message.includes('econnreset') ||
|
||||||
|
message.includes('etimedout')
|
||||||
|
);
|
||||||
|
if (!isTransportFailure) {
|
||||||
|
break;
|
||||||
|
}
|
||||||
|
|
||||||
if (attempt === maxRetries - 1) {
|
if (attempt === maxRetries - 1) {
|
||||||
break;
|
break;
|
||||||
}
|
}
|
||||||
|
|
|
||||||
Loading…
Reference in a new issue