/
/
1"""Helpers to render text into playable speech audio through the plugin TTS engines."""
2
3from __future__ import annotations
4
5import asyncio
6import logging
7from pathlib import Path
8from typing import TYPE_CHECKING, Any
9
10from music_assistant_models.enums import StreamType
11from music_assistant_models.errors import InvalidDataError, MusicAssistantError
12
13from music_assistant.constants import MASS_LOGGER_NAME
14
15if TYPE_CHECKING:
16 from music_assistant_models.streamdetails import StreamDetails
17
18 from music_assistant.mass import MusicAssistant
19 from music_assistant.models.plugin import TTSEngine
20
21LOGGER = logging.getLogger(f"{MASS_LOGGER_NAME}.helpers.tts")
22
23# last-resort guard so a wedged engine fails the call instead of hanging its caller.
24# Kept above the deadlines the engines apply themselves (120s in the OpenAI-compatible
25# providers), so their own, more specific error is the one that surfaces.
26TTS_QUERY_TIMEOUT_SECONDS = 180
27
28REMOTE_STREAM_SCHEMES = ("http://", "https://", "rtsp://", "rtmp://")
29
30
31async def query_tts_engine(
32 engine: TTSEngine,
33 message: str,
34 language: str | None = None,
35 timeout: float | None = None,
36 options: dict[str, Any] | None = None,
37) -> StreamDetails:
38 """
39 Render a message through a TTS engine.
40
41 :param engine: The TTS engine to speak the message.
42 :param message: The text to speak.
43 :param language: Optional language code, omit to use the engine's own default voice.
44 :param timeout: Seconds to wait for the engine, defaults to TTS_QUERY_TIMEOUT_SECONDS.
45 Lower it for a caller that is holding something up while it waits.
46 :param options: Optional integration-specific TTS options, passed through to the
47 engine as-is. Ignored by engines that have none.
48 """
49 if timeout is None:
50 timeout = TTS_QUERY_TIMEOUT_SECONDS
51 try:
52 async with asyncio.timeout(timeout) as query_timeout:
53 return await engine.provider.get_tts_message(
54 message, language=language, engine_id=engine.id, options=options
55 )
56 except TimeoutError as err:
57 # expired() tells our own cap apart from a timeout raised inside the engine
58 if not query_timeout.expired():
59 raise
60 raise MusicAssistantError(
61 f"TTS engine '{engine.uid}' did not respond within {timeout}s"
62 ) from err
63
64
65async def query_tts_engine_with_language_fallback(
66 engine: TTSEngine,
67 message: str,
68 language: str | None = None,
69 timeout: float | None = None,
70 logger: logging.Logger | None = None,
71 options: dict[str, Any] | None = None,
72) -> StreamDetails:
73 """
74 Render a message through a TTS engine, retrying without the language if it is rejected.
75
76 :param engine: The TTS engine to speak the message.
77 :param message: The text to speak.
78 :param language: Optional language code, omit to use the engine's own default voice.
79 :param timeout: Seconds to wait for the engine, defaults to TTS_QUERY_TIMEOUT_SECONDS.
80 Lower it for a caller that is holding something up while it waits.
81 :param logger: Optional logger to report a rejected language on.
82 :param options: Optional integration-specific TTS options, passed through to the
83 engine as-is. Ignored by engines that have none.
84 """
85 try:
86 return await query_tts_engine(engine, message, language, timeout, options)
87 except TimeoutError, MusicAssistantError:
88 # a timeout or our own structured failure is not a language rejection, so a
89 # language-less retry would not help and would only double the wait
90 raise
91 except Exception as err:
92 if language is None:
93 raise
94 # some engines reject a language they don't support; fall back to the engine's
95 # own default voice rather than losing the audio entirely
96 (logger or LOGGER).warning(
97 "TTS engine '%s' rejected language '%s' (%s), retrying with its default voice",
98 engine.uid,
99 language,
100 err,
101 )
102 return await query_tts_engine(engine, message, None, timeout, options)
103
104
105def resolve_tts_language(mass: MusicAssistant) -> str | None:
106 """
107 Return the language a TTS engine should speak in, as a hyphenated code like 'en-US'.
108
109 Returns None when no language is configured, leaving the engine on its own default voice.
110
111 :param mass: The Music Assistant instance holding the configured locale.
112 """
113 locale = mass.metadata.locale
114 return locale.replace("_", "-") if locale else None
115
116
117async def resolve_tts_stream_path(
118 engine: TTSEngine, stream_details: StreamDetails
119) -> tuple[str, StreamType]:
120 """
121 Return the playable path of a rendered clip and the way to stream it.
122
123 :param engine: The TTS engine that produced the clip, named in the error raised when it
124 did not return anything playable.
125 :param stream_details: The StreamDetails the engine returned.
126 """
127 path = str(stream_details.path or "").strip()
128 if path.startswith(REMOTE_STREAM_SCHEMES):
129 return path, StreamType.HTTP
130 if path and Path(path).is_absolute() and await asyncio.to_thread(Path(path).is_file):
131 return path, StreamType.LOCAL_FILE
132 raise InvalidDataError(
133 f"TTS engine '{engine.uid}' returned an unusable stream path: "
134 f"{path or '<empty>'}. StreamDetails.path must be a fetchable "
135 "http(s)/rtsp/rtmp URL or the absolute path of an existing local file."
136 )
137