/
/
1"""Various helpers for audio streaming and manipulation."""
2
3from __future__ import annotations
4
5import asyncio
6import logging
7import re
8import struct
9import urllib.parse
10from collections.abc import AsyncGenerator, Iterable, Iterator
11from contextlib import aclosing
12from io import BytesIO
13from math import isfinite
14from typing import TYPE_CHECKING, Final
15
16from music_assistant_models.enums import (
17 ContentType,
18 MediaType,
19 PlayerFeature,
20 PlayerType,
21 VolumeNormalizationMode,
22)
23from music_assistant_models.errors import InvalidDataError
24from music_assistant_models.streamdetails import MultiPartPath
25
26from music_assistant.constants import (
27 MASS_LOGGER_NAME,
28 VERBOSE_LOG_LEVEL,
29)
30from music_assistant.helpers.json import JSON_DECODE_EXCEPTIONS, json_loads
31
32from .ffmpeg import DEFAULT_MP3_BIT_RATE, get_ffmpeg_stream
33from .process import AsyncProcess, communicate
34
35if TYPE_CHECKING:
36 from music_assistant_models.media_items import AudioFormat
37 from music_assistant_models.streamdetails import StreamDetails
38
39 from music_assistant.mass import MusicAssistant
40 from music_assistant.models.player import Player
41
42LOGGER = logging.getLogger(f"{MASS_LOGGER_NAME}.helpers.audio")
43
44HTTP_HEADERS = {"User-Agent": "Lavf/60.16.100.MusicAssistant"}
45HTTP_HEADERS_ICY = {**HTTP_HEADERS, "Icy-MetaData": "1"}
46
47SLOW_PROVIDERS = ("tidal", "ytmusic", "apple_music")
48
49# Mapping of audio format identifiers to their correct IANA MIME types
50# where the format name differs from the MIME subtype.
51# Strict DLNA/UPnP devices reject non-standard MIME types (e.g. audio/mp3).
52_MIME_TYPE_OVERRIDES: Final[dict[str, str]] = {
53 "mp3": "audio/mpeg",
54}
55
56
57def get_mime_type(format_str: str) -> str:
58 """
59 Get the proper IANA MIME type for a given audio format string.
60
61 :param format_str: The audio format string (e.g. "mp3", "flac",
62 "pcm;codec=pcm;rate=44100;bitrate=16;channels=2").
63 """
64 base_format = format_str.split(";", maxsplit=1)[0]
65 if override := _MIME_TYPE_OVERRIDES.get(base_format):
66 return override
67 return f"audio/{format_str}"
68
69
70def parse_pcm_info(content_type: str) -> tuple[int, int, int]:
71 """
72 Parse PCM info from a codec/content_type string.
73
74 :param content_type: Content type string like "pcm;codec=pcm;rate=44100;bitrate=16;channels=2".
75 """
76 params = (
77 dict(urllib.parse.parse_qsl(content_type.replace(";", "&"))) if ";" in content_type else {}
78 )
79 sample_rate = int(params.get("rate", 44100))
80 sample_size = int(params.get("bitrate", 16))
81 channels = int(params.get("channels", 2))
82 return (sample_rate, sample_size, channels)
83
84
85CACHE_CATEGORY_RESOLVED_RADIO_URL: Final[int] = 100
86CACHE_PROVIDER: Final[str] = "audio"
87
88
89def iter_pcm_slices(
90 audio: bytes,
91 pcm_format: AudioFormat,
92 target_duration_ms: int = 100,
93) -> Iterator[bytes]:
94 """
95 Yield frame-aligned PCM slices of approximately ``target_duration_ms``.
96
97 Large PCM buffers (e.g. crossfade segments or full-track reads) are split
98 into fixed-size sub-chunks so that downstream consumers get predictable
99 chunk sizes for buffering, write-timeout management, and ring-buffer
100 bookkeeping.
101
102 :param audio: Raw PCM bytes to slice.
103 :param pcm_format: Format description (sample rate, bit depth, channels).
104 :param target_duration_ms: Desired slice length in milliseconds (default 100).
105 """
106 if not audio:
107 return
108 bytes_per_sample = max(1, pcm_format.bit_depth // 8)
109 frame_size = bytes_per_sample * pcm_format.channels
110 if frame_size <= 0:
111 yield audio
112 return
113 samples_per_slice = max(1, round((target_duration_ms / 1000) * pcm_format.sample_rate))
114 slice_size = max(frame_size, samples_per_slice * frame_size)
115 offset = 0
116 audio_len = len(audio)
117 while offset < audio_len:
118 end = min(audio_len, offset + slice_size)
119 # Align to frame boundary unless this is the tail of the buffer.
120 if end < audio_len:
121 aligned_end = end - (end % frame_size)
122 if aligned_end <= offset:
123 aligned_end = min(audio_len, offset + frame_size)
124 end = aligned_end
125 yield audio[offset:end]
126 offset = end
127
128
129def align_audio_to_frame_boundary(audio_data: bytes, pcm_format: AudioFormat) -> bytes:
130 """
131 Align audio data to frame boundaries by truncating incomplete frames.
132
133 :param audio_data: Raw PCM audio data to align.
134 :param pcm_format: AudioFormat of the audio data.
135 """
136 bytes_per_sample = pcm_format.bit_depth // 8
137 frame_size = bytes_per_sample * pcm_format.channels
138 valid_bytes = (len(audio_data) // frame_size) * frame_size
139 if valid_bytes != len(audio_data):
140 LOGGER.debug(
141 "Truncating %d bytes from audio buffer to align to frame boundary",
142 len(audio_data) - valid_bytes,
143 )
144 return audio_data[:valid_bytes]
145 return audio_data
146
147
148async def strip_silence(
149 audio_data: bytes,
150 pcm_format: AudioFormat,
151 reverse: bool = False,
152) -> bytes:
153 """
154 Strip silence from begin or end of pcm audio using ffmpeg.
155
156 :param audio_data: Raw PCM audio data.
157 :param pcm_format: AudioFormat of the audio data.
158 :param reverse: If True, strip from end instead of beginning.
159 """
160 args = ["ffmpeg", "-hide_banner", "-loglevel", "quiet"]
161 args += [
162 "-acodec",
163 pcm_format.content_type.name.lower(),
164 "-f",
165 pcm_format.content_type.value,
166 "-ac",
167 str(pcm_format.channels),
168 "-ar",
169 str(pcm_format.sample_rate),
170 "-i",
171 "-",
172 ]
173 if reverse:
174 args += [
175 "-af",
176 "areverse,atrim=start=0.2,silenceremove=start_periods=1"
177 ":start_silence=0.1:start_threshold=0.02,areverse",
178 ]
179 else:
180 args += [
181 "-af",
182 "atrim=start=0.2,silenceremove=start_periods=1:start_silence=0.1:start_threshold=0.02",
183 ]
184 args += ["-f", pcm_format.content_type.value, "-"]
185 _returncode, stripped_data, _stderr = await communicate(args, audio_data)
186
187 bytes_stripped = len(audio_data) - len(stripped_data)
188 if LOGGER.isEnabledFor(VERBOSE_LOG_LEVEL):
189 seconds_stripped = round(bytes_stripped / pcm_format.pcm_sample_size, 2)
190 location = "end" if reverse else "begin"
191 LOGGER.log(
192 VERBOSE_LOG_LEVEL,
193 "stripped %s seconds of silence from %s of pcm audio. bytes stripped: %s",
194 seconds_stripped,
195 location,
196 bytes_stripped,
197 )
198 return stripped_data
199
200
201def create_wave_header(
202 samplerate: int = 44100,
203 channels: int = 2,
204 bitspersample: int = 16,
205 duration: int | None = None,
206) -> bytes:
207 """Generate a wave header from given params."""
208 file = BytesIO()
209
210 # Generate format chunk
211 format_chunk_spec = b"<4sLHHLLHH"
212 format_chunk = struct.pack(
213 format_chunk_spec,
214 b"fmt ", # Chunk id
215 16, # Size of this chunk (excluding chunk id and this field)
216 1, # Audio format, 1 for PCM
217 channels, # Number of channels
218 int(samplerate), # Samplerate, 44100, 48000, etc.
219 int(samplerate * channels * (bitspersample / 8)), # Byterate
220 int(channels * (bitspersample / 8)), # Blockalign
221 bitspersample, # 16 bits for two byte samples, etc.
222 )
223 # Generate data chunk
224 # duration = 3600*6.7
225 data_chunk_spec = b"<4sL"
226 if duration is None:
227 # use max value possible
228 datasize = 4254768000 # = 6,7 hours at 44100/16
229 else:
230 # calculate from duration
231 numsamples = samplerate * duration
232 datasize = int(numsamples * channels * (bitspersample / 8))
233 data_chunk = struct.pack(
234 data_chunk_spec,
235 b"data", # Chunk id
236 int(datasize), # Chunk size (excluding chunk id and this field)
237 )
238 sum_items = [
239 # "WAVE" string following size field
240 4,
241 # "fmt " + chunk size field + chunk size
242 struct.calcsize(format_chunk_spec),
243 # Size of data chunk spec + data size
244 struct.calcsize(data_chunk_spec) + datasize,
245 ]
246 # Generate main header
247 all_chunks_size = int(sum(sum_items))
248 main_header_spec = b"<4sL4s"
249 main_header = struct.pack(main_header_spec, b"RIFF", all_chunks_size, b"WAVE")
250 # Write all the contents in
251 file.write(main_header)
252 file.write(format_chunk)
253 file.write(data_chunk)
254
255 # return file.getvalue(), all_chunks_size + 8
256 return file.getvalue()
257
258
259def create_streaming_wave_header(audio_format: AudioFormat) -> bytes:
260 """
261 Generate a wave header for a stream whose length is not known up front.
262
263 :param audio_format: The PCM format the audio behind the header is in.
264 """
265 channels = audio_format.channels
266 sample_rate = audio_format.sample_rate
267 bits_per_sample = audio_format.bit_depth
268 byte_rate = sample_rate * channels * (bits_per_sample // 8)
269 block_align = channels * (bits_per_sample // 8)
270 # RIFF size & data size both set to 0xFFFFFFFF so clients honoring the WAV
271 # length fields don't cut the stream off (create_wave_header hardcodes ~6.7h).
272 return (
273 b"RIFF"
274 + struct.pack("<L", 0xFFFFFFFF)
275 + b"WAVE"
276 + b"fmt "
277 + struct.pack(
278 "<LHHLLHH", 16, 1, channels, sample_rate, byte_rate, block_align, bits_per_sample
279 )
280 + b"data"
281 + struct.pack("<L", 0xFFFFFFFF)
282 )
283
284
285def parse_extinf_metadata(extinf_line: str) -> dict[str, str]:
286 """
287 Parse metadata from HLS EXTINF line.
288
289 Extracts structured metadata like title="...", artist="..." from EXTINF lines.
290 Common in iHeartRadio and other commercial radio HLS streams.
291
292 :param extinf_line: The EXTINF line containing metadata
293 """
294 metadata = {}
295
296 # Pattern to match key="value" pairs in the EXTINF line
297 # Handles nested quotes by matching everything until the closing quote
298 pattern = r'(\w+)="([^"]*)"'
299
300 matches = re.findall(pattern, extinf_line)
301 for key, value in matches:
302 metadata[key.lower()] = value
303
304 # Fallback: RFC 8216 plain title format `#EXTINF:<duration>,<title>`
305 if not metadata and "," in extinf_line:
306 title = extinf_line.split(",", 1)[1].strip()
307 if title:
308 metadata["title"] = title
309
310 return metadata
311
312
313def get_parts_from_position(
314 parts: list[MultiPartPath],
315 seek_position: int,
316) -> tuple[list[MultiPartPath], int]:
317 """
318 Get the remaining parts list from a timestamp.
319
320 Arguments:
321 parts: The list of parts
322 seek_position: The seeking position in seconds of the tracklist
323
324 Returns:
325 In a tuple, A list of parts, starting with the one at the requested
326 seek position and the position in seconds to seek to in the first
327 track.
328 """
329 skipped_duration = 0.0
330 for i, part in enumerate(parts):
331 if not isinstance(part, MultiPartPath):
332 raise InvalidDataError("Multi-file streamdetails requires a list of MultiPartPath")
333 if part.duration is None:
334 return parts, seek_position
335 if skipped_duration + part.duration < seek_position:
336 skipped_duration += part.duration
337 continue
338
339 position = seek_position - skipped_duration
340
341 # Seeking in some parts is inaccurate, making the seek to a chapter land on the end of
342 # the previous track. If we're within 2 second of the end, skip the current track
343 if position + 2 >= part.duration:
344 LOGGER.debug(
345 f"Skipping to the next part due to seek position being at the end: {position}",
346 )
347 if i + 1 < len(parts):
348 return parts[i + 1 :], 0
349 return parts[i:], int(position) # last part, cannot skip
350
351 return parts[i:], int(position)
352
353 raise IndexError(f"Could not find any candidate part for position {seek_position}")
354
355
356def build_concat_filelist(paths: list[str]) -> str:
357 """
358 Build the file list content for ffmpeg's concat demuxer.
359
360 :param paths: The file paths to include, in playback order.
361 """
362 lines = []
363 for path in paths:
364 # The concat demuxer uses single quotes as delimiters, so a literal quote in the
365 # path must be written as '\'' to prevent the path being truncated at the quote.
366 escaped_path = path.replace("'", "'\\''")
367 lines.append(f"file '{escaped_path}'\n")
368 return "".join(lines)
369
370
371async def realtime_pcm_pacer(
372 inner: AsyncGenerator[bytes],
373 pcm_format: AudioFormat,
374) -> AsyncGenerator[bytes]:
375 """
376 Pace a PCM byte stream at the format's native rate.
377
378 Useful for live AudioSource streams whose producer is not realtime-paced
379 (e.g. librespot's pipe backend) — without rate-limiting the consumer would
380 buffer many seconds of audio ahead of playback, making skip/next laggy.
381
382 :param inner: Source generator yielding raw PCM bytes.
383 :param pcm_format: PCM format the inner generator emits.
384 """
385 bytes_per_second = pcm_format.sample_rate * pcm_format.channels * (pcm_format.bit_depth // 8)
386 if bytes_per_second <= 0 or not pcm_format.content_type.is_pcm():
387 # non-PCM or malformed format: pass through unchanged
388 async for chunk in inner:
389 yield chunk
390 return
391 loop = asyncio.get_running_loop()
392 start_time = loop.time()
393 total_bytes = 0
394 async for chunk in inner:
395 yield chunk
396 total_bytes += len(chunk)
397 expected_elapsed = total_bytes / bytes_per_second
398 actual_elapsed = loop.time() - start_time
399 if actual_elapsed < expected_elapsed:
400 await asyncio.sleep(expected_elapsed - actual_elapsed)
401
402
403async def audio_source_silence_keepalive(
404 inner: AsyncGenerator[bytes],
405 pcm_format: AudioFormat,
406 silence_chunk_ms: int = 100,
407 idle_threshold_s: float | None = None,
408) -> AsyncGenerator[bytes]:
409 """
410 Wrap a live AudioSource PCM stream and emit silence during idle gaps.
411
412 Plugin providers exposing an AudioSource may stop yielding bytes while the
413 upstream device is paused (e.g. user paused in the Spotify app). Without
414 bytes flowing the downstream consumer (ffmpeg / the player) may disconnect.
415 This wrapper inserts ``silence_chunk_ms`` worth of zero bytes whenever the
416 inner generator hasn't produced for ``idle_threshold_s`` seconds, while
417 relaying real bytes immediately when they arrive.
418
419 Only meaningful for PCM streams — injecting raw zero bytes into a compressed
420 stream (MP3/AAC/etc.) would corrupt the bitstream. For non-PCM ``pcm_format``
421 inputs the wrapper degrades to a transparent pass-through.
422
423 :param inner: The underlying async generator yielding raw PCM bytes.
424 :param pcm_format: PCM format the inner generator emits (used to size the
425 silence chunk so it lines up to a frame boundary).
426 :param silence_chunk_ms: Duration of each silence chunk in milliseconds.
427 :param idle_threshold_s: Seconds without input before silence is inserted.
428 Defaults to the chunk duration so silence flows at realtime — critical
429 for keeping HTTP consumers (Sonos, Chromecast) connected.
430 """
431 if idle_threshold_s is None:
432 idle_threshold_s = silence_chunk_ms / 1000
433 frame_size = pcm_format.channels * (pcm_format.bit_depth // 8)
434 bytes_per_second = (
435 pcm_format.sample_rate * frame_size if pcm_format.content_type.is_pcm() else 0
436 )
437 if bytes_per_second <= 0 or frame_size <= 0:
438 # non-PCM or malformed format: pass through unchanged, no silence injection
439 async for chunk in inner:
440 yield chunk
441 return
442
443 # Round the silence chunk size DOWN to a whole-frame multiple so emitted
444 # chunks line up to PCM frame boundaries for arbitrary silence_chunk_ms /
445 # sample-rate combinations.
446 raw_silence_bytes = bytes_per_second * silence_chunk_ms // 1000
447 silence_bytes = max(frame_size, (raw_silence_bytes // frame_size) * frame_size)
448 silence_chunk = b"\x00" * silence_bytes
449 # empty bytes is the end-of-stream sentinel; real PCM frames are never empty
450 queue: asyncio.Queue[bytes] = asyncio.Queue(maxsize=8)
451
452 async def _producer() -> None:
453 # aclosing ensures inner.aclose() runs on cancellation so the underlying
454 # generator's own finally (e.g. plugin lock release, fd cleanup) fires
455 # instead of leaking until GC.
456 try:
457 async with aclosing(inner) as managed_inner:
458 async for chunk in managed_inner:
459 await queue.put(chunk)
460 finally:
461 await queue.put(b"")
462
463 producer_task = asyncio.create_task(_producer())
464 try:
465 while True:
466 try:
467 chunk = await asyncio.wait_for(queue.get(), timeout=idle_threshold_s)
468 except TimeoutError:
469 yield silence_chunk
470 continue
471 if not chunk:
472 break
473 yield chunk
474 finally:
475 producer_task.cancel()
476 try:
477 await producer_task
478 except asyncio.CancelledError:
479 pass
480 except Exception:
481 # log but don't re-raise: we're already in a finally and the
482 # downstream consumer has its own error handling for the outer stream.
483 LOGGER.exception("AudioSource producer task raised")
484
485
486async def get_silence(
487 duration: int,
488 output_format: AudioFormat,
489) -> AsyncGenerator[bytes]:
490 """Create stream of silence, encoded to format of choice."""
491 if output_format.content_type.is_pcm():
492 # pcm = just zeros
493 for _ in range(duration):
494 yield b"\0" * int(output_format.sample_rate * (output_format.bit_depth / 8) * 2)
495 return
496 if output_format.content_type == ContentType.WAV:
497 # wav silence = wave header + zero's
498 yield create_wave_header(
499 samplerate=output_format.sample_rate,
500 channels=2,
501 bitspersample=output_format.bit_depth,
502 duration=duration,
503 )
504 for _ in range(duration):
505 yield b"\0" * int(output_format.sample_rate * (output_format.bit_depth / 8) * 2)
506 return
507 # use ffmpeg for all other encodings
508 args = [
509 "ffmpeg",
510 "-hide_banner",
511 "-loglevel",
512 "quiet",
513 "-f",
514 "lavfi",
515 "-i",
516 f"anullsrc=r={output_format.sample_rate}:cl={'stereo'}",
517 "-t",
518 str(duration),
519 "-f",
520 output_format.output_format_str,
521 "-",
522 ]
523 async with AsyncProcess(args, stdout=True) as ffmpeg_proc:
524 async for chunk in ffmpeg_proc.iter_chunked():
525 yield chunk
526
527
528async def resample_pcm_audio(
529 input_audio: bytes | AsyncGenerator[bytes],
530 input_format: AudioFormat,
531 output_format: AudioFormat,
532 chunk_size: int | None = None,
533) -> AsyncGenerator[bytes]:
534 """
535 Resample PCM audio from input_format to output_format using ffmpeg.
536
537 Yields chunks of resampled audio as they become available.
538
539 :param input_audio: Raw PCM audio data or async generator of PCM chunks.
540 :param input_format: AudioFormat of the input audio.
541 :param output_format: Desired AudioFormat for the output audio.
542 :param chunk_size: Output chunk size in bytes. Defaults to 1 second of output PCM.
543 """
544 if chunk_size is None:
545 chunk_size = output_format.pcm_sample_size
546
547 async def _as_generator() -> AsyncGenerator[bytes]:
548 if isinstance(input_audio, bytes):
549 yield input_audio
550 else:
551 async for chunk in input_audio:
552 yield chunk
553
554 if input_format == output_format:
555 buffer = b""
556 async for chunk in _as_generator():
557 buffer += chunk
558 while len(buffer) >= chunk_size:
559 yield buffer[:chunk_size]
560 buffer = buffer[chunk_size:]
561 if buffer:
562 yield buffer
563 return
564
565 async for chunk in get_ffmpeg_stream(
566 audio_input=_as_generator(),
567 input_format=input_format,
568 output_format=output_format,
569 chunk_size=chunk_size,
570 ):
571 yield chunk
572
573
574def calculate_content_length(
575 fmt: AudioFormat,
576 seconds: float = 1,
577) -> int:
578 """
579 Calculate the estimated encoded size in bytes for a given format and duration.
580
581 For CBR lossy formats (MP3/AAC), the estimate is near-exact.
582 For lossless formats (FLAC), the estimate uses an empirical average
583 compression ratio and may differ from actual size by up to ~15%.
584 For uncompressed formats (PCM/WAV), the result is exact.
585
586 :param fmt: The audio format to estimate size for.
587 :param seconds: Duration in seconds.
588 """
589 pcm_size = int(fmt.sample_rate * (fmt.bit_depth / 8) * fmt.channels * seconds)
590 if fmt.content_type.is_pcm():
591 return pcm_size
592 if fmt.content_type in (ContentType.WAV, ContentType.AIFF, ContentType.DSF):
593 return pcm_size
594 if fmt.bit_rate and fmt.bit_rate < 10000:
595 return int(((fmt.bit_rate * 1000) / 8) * seconds)
596 if fmt.content_type in (ContentType.FLAC, ContentType.WAVPACK, ContentType.ALAC):
597 # FLAC compression_level 0: empirical ratio ~74.7% of PCM
598 # Source: https://z-issue.com/wp/flac-compression-level-comparison/
599 # Real-world variance: 65-85% depending on audio content.
600 return int(pcm_size * 0.747)
601 if fmt.content_type == ContentType.MP3:
602 return int(((DEFAULT_MP3_BIT_RATE * 1000) / 8) * seconds)
603 if fmt.content_type == ContentType.OGG:
604 return int((320000 / 8) * seconds)
605 if fmt.content_type in (ContentType.AAC, ContentType.M4A):
606 # CBR 256kbps as set in get_ffmpeg_args
607 return int((256000 / 8) * seconds)
608 return int((320000 / 8) * seconds)
609
610
611def get_output_format_key(fmt: AudioFormat) -> str:
612 """
613 Get a stable key representing the output encoding parameters.
614
615 :param fmt: The output audio format.
616 """
617 return f"{fmt.content_type.value}_{fmt.sample_rate}_{fmt.bit_depth}_{fmt.channels}"
618
619
620CONTENT_LENGTH_CACHE_CATEGORY = 50
621CONTENT_LENGTH_CACHE_PROVIDER = "audio"
622CONTENT_LENGTH_CACHE_EXPIRATION = 365 * 86400 # 1 year
623
624
625async def get_content_length(
626 mass: MusicAssistant,
627 uri: str,
628 output_format: AudioFormat,
629 seconds: float,
630) -> int:
631 """
632 Get the estimated encoded size, using cached actual measurement when available.
633
634 After a track has been fully streamed, its actual content size and duration
635 are cached. On subsequent plays this gives a near-exact content_length:
636 - Exact when the requested duration matches the cached duration.
637 - Very accurate when the duration differs (derived bytes-per-second).
638
639 Falls back to the static estimate from calculate_content_length() if no cache entry exists.
640
641 :param mass: The MusicAssistant instance (for cache access).
642 :param uri: The media URI (e.g. "qobuz://track/12345").
643 :param output_format: The output audio format.
644 :param seconds: Duration in seconds to estimate.
645 """
646 cache_key = f"{uri}/{get_output_format_key(output_format)}"
647 cached: dict[str, float] | None = await mass.cache.get(
648 cache_key,
649 provider=CONTENT_LENGTH_CACHE_PROVIDER,
650 category=CONTENT_LENGTH_CACHE_CATEGORY,
651 )
652 if cached is not None:
653 cached_size = cached["size"]
654 cached_duration = cached["duration"]
655 if abs(seconds - cached_duration) < 1:
656 # same duration: return the exact cached size
657 return int(cached_size)
658 # different duration: derive bytes-per-second from the cached measurement
659 return int((cached_size / cached_duration) * seconds)
660 return calculate_content_length(output_format, seconds)
661
662
663async def store_content_length_in_cache(
664 mass: MusicAssistant,
665 uri: str,
666 output_format: AudioFormat,
667 content_size: int,
668 seconds_streamed: float,
669) -> None:
670 """
671 Store the actual content size after a track has been fully streamed.
672
673 :param mass: The MusicAssistant instance (for cache access).
674 :param uri: The media URI (e.g. "qobuz://track/12345").
675 :param output_format: The output audio format used for encoding.
676 :param content_size: Total encoded bytes sent to the player.
677 :param seconds_streamed: Duration of audio streamed in seconds.
678 """
679 if seconds_streamed < 10 or content_size < 1000:
680 return
681 cache_key = f"{uri}/{get_output_format_key(output_format)}"
682 await mass.cache.set(
683 cache_key,
684 {"size": content_size, "duration": seconds_streamed},
685 expiration=CONTENT_LENGTH_CACHE_EXPIRATION,
686 provider=CONTENT_LENGTH_CACHE_PROVIDER,
687 category=CONTENT_LENGTH_CACHE_CATEGORY,
688 persistent=True,
689 )
690
691
692PROBED_DURATION_CACHE_CATEGORY = 51
693PROBED_DURATION_CACHE_PROVIDER = "audio"
694PROBED_DURATION_CACHE_EXPIRATION = 365 * 86400 # 1 year
695
696
697async def get_probed_duration(mass: MusicAssistant, uri: str) -> int | None:
698 """
699 Get the duration determined during an earlier playback of the given item, if any.
700
701 Use for items whose provider does not report a duration, such as podcast episodes
702 from a feed without itunes:duration.
703
704 :param mass: The MusicAssistant instance (for cache access).
705 :param uri: The media item URI (e.g. "overcast--1://podcast_episode/abc").
706 :return: The duration in seconds, or None if the item was never played.
707 """
708 duration: int | None = await mass.cache.get(
709 uri,
710 provider=PROBED_DURATION_CACHE_PROVIDER,
711 category=PROBED_DURATION_CACHE_CATEGORY,
712 )
713 return duration
714
715
716async def store_probed_duration(mass: MusicAssistant, uri: str, duration: int) -> None:
717 """
718 Store the duration of an item that was determined while streaming it.
719
720 A duration below a second is ignored.
721
722 :param mass: The MusicAssistant instance (for cache access).
723 :param uri: The media item URI (e.g. "overcast--1://podcast_episode/abc").
724 :param duration: The duration in seconds.
725 """
726 if duration < 1:
727 return
728 await mass.cache.set(
729 uri,
730 duration,
731 expiration=PROBED_DURATION_CACHE_EXPIRATION,
732 provider=PROBED_DURATION_CACHE_PROVIDER,
733 category=PROBED_DURATION_CACHE_CATEGORY,
734 persistent=True,
735 )
736
737
738def get_bit_rate(fmt: AudioFormat) -> int:
739 """Get the (estimated) bit rate for a given AudioFormat, if known."""
740 if fmt.bit_rate:
741 return int(fmt.bit_rate / 1000) if fmt.bit_rate >= 10000 else fmt.bit_rate
742 return int((calculate_content_length(fmt, seconds=1) / 1000) * 8)
743
744
745def resolve_output_player_ids(
746 mass: MusicAssistant,
747 player_ids: Iterable[str],
748) -> set[str]:
749 """
750 Resolve output destinations to their user-facing player identifiers.
751
752 :param mass: Music Assistant instance.
753 :param player_ids: Player or protocol-player identifiers to resolve.
754 :return: Deduplicated user-facing player identifiers.
755 """
756 resolved_ids: set[str] = set()
757 for player_id in player_ids:
758 player = mass.players.get_player(player_id)
759 resolved_ids.add(
760 player.protocol_parent_id if player and player.protocol_parent_id else player_id
761 )
762 return resolved_ids
763
764
765def is_grouping_preventing_dsp(player: Player) -> bool:
766 """
767 Check if grouping is preventing DSP from being applied to this leader/PlayerGroup.
768
769 If this returns True, no DSP should be applied to the player.
770 This function will not check if the Player is in a group, the caller should do that first.
771 """
772 # We require the caller to handle non-leader cases themselves since player.state.synced_to
773 # can be unreliable in some edge cases
774 multi_device_dsp_supported = PlayerFeature.MULTI_DEVICE_DSP in player.state.supported_features
775 child_count = len(player.state.group_members) if player.state.group_members else 0
776
777 is_multiple_devices: bool
778 if player.provider.domain == "player_group":
779 # PlayerGroups have no leader, so having a child count of 1 means
780 # the group actually contains only a single player.
781 is_multiple_devices = child_count > 1
782 elif player.state.type == PlayerType.GROUP:
783 # This is an group player external to Music Assistant.
784 is_multiple_devices = True
785 else:
786 is_multiple_devices = child_count > 0
787 return is_multiple_devices and not multi_device_dsp_supported
788
789
790def parse_loudnorm(raw_stderr: bytes | str) -> float | None:
791 """Parse Loudness measurement from ffmpeg stderr output."""
792 stderr_data = raw_stderr.decode() if isinstance(raw_stderr, bytes) else raw_stderr
793 # the report is the last thing the filter logs, and ffmpeg prints it as a block of its
794 # own below the marker line, so the object is delimited rather than on a known line.
795 # the marker carries the filter's position in the chain, which is only zero when
796 # loudnorm runs on its own
797 marker = stderr_data.rfind("[Parsed_loudnorm_")
798 if marker < 0:
799 return None
800 start = stderr_data.find("{", marker)
801 if start < 0 or (end := stderr_data.find("}", start)) < 0:
802 return None
803 try:
804 loudness_data = json_loads(stderr_data[start : end + 1])
805 measurement = float(loudness_data["input_i"])
806 except (*JSON_DECODE_EXCEPTIONS, KeyError, ValueError):
807 return None
808 # digital silence reads as -inf, which is a report that the clip has no level rather
809 # than a level to correct against
810 return measurement if isfinite(measurement) else None
811
812
813def get_normalization_mode(
814 preference: VolumeNormalizationMode,
815 volume_normalization_enabled: bool,
816 streamdetails: StreamDetails,
817) -> VolumeNormalizationMode:
818 """
819 Get the volume normalization mode for a given queue and stream.
820
821 :param preference: The configured normalization preference for the stream's media type
822 (tracks or radio), from the streams core config.
823 :param volume_normalization_enabled: Whether normalization is enabled for the queue, already
824 resolved from the per-queue setting and its global (queue controller) fallback.
825 :param streamdetails: The stream to evaluate.
826 """
827 if not volume_normalization_enabled:
828 # disabled for this queue
829 return VolumeNormalizationMode.DISABLED
830 if streamdetails.media_type == MediaType.AUDIO_SOURCE:
831 # live/realtime: upstream producer owns loudness, no measurement to converge on
832 return VolumeNormalizationMode.DISABLED
833 if streamdetails.media_type == MediaType.SOUND_EFFECT:
834 # never measured, and the dynamic fallback compresses short clips
835 return VolumeNormalizationMode.DISABLED
836 if streamdetails.target_loudness is None:
837 # no target loudness set, disable normalization
838 return VolumeNormalizationMode.DISABLED
839
840 # handle no measurement available but fallback to dynamic mode is allowed
841 if streamdetails.loudness is None and preference == VolumeNormalizationMode.FALLBACK_DYNAMIC:
842 return VolumeNormalizationMode.DYNAMIC
843
844 # handle no measurement available and no fallback allowed
845 if streamdetails.loudness is None and preference == VolumeNormalizationMode.MEASUREMENT_ONLY:
846 return VolumeNormalizationMode.DISABLED
847
848 # handle no measurement available and fallback to fixed gain is allowed
849 if streamdetails.loudness is None and preference == VolumeNormalizationMode.FALLBACK_FIXED_GAIN:
850 return VolumeNormalizationMode.FIXED_GAIN
851
852 # handle measurement available - chosen mode is measurement
853 if streamdetails.loudness is not None and preference not in (
854 VolumeNormalizationMode.DISABLED,
855 VolumeNormalizationMode.FIXED_GAIN,
856 VolumeNormalizationMode.DYNAMIC,
857 ):
858 return VolumeNormalizationMode.MEASUREMENT_ONLY
859
860 # simply return the preference
861 return preference
862