/
/
1"""Tests for the ffmpeg helper module."""
2
3from __future__ import annotations
4
5import asyncio
6import subprocess
7from array import array
8from collections.abc import AsyncGenerator, Sequence
9from math import sqrt
10from pathlib import Path
11
12import pytest
13from music_assistant_models.enums import ContentType
14from music_assistant_models.errors import AudioError
15from music_assistant_models.media_items import AudioFormat
16
17from music_assistant.helpers.dsp import ComplexFilter, ComplexFilterInput
18from music_assistant.helpers.ffmpeg import (
19 _INPUT_READ_ARGS,
20 FFMpeg,
21 FFMpegStreamInfo,
22 _build_filtergraph_args,
23 _build_overlay_mixer,
24 _get_overlay_volume_filter,
25 get_ffmpeg_args,
26 get_ffmpeg_overlay_stream,
27 get_ffmpeg_stream,
28 parse_ffmpeg_duration,
29 parse_ffmpeg_stream_info,
30)
31
32
33def test_get_ffmpeg_args_does_not_mutate_filters() -> None:
34 """Automatic resampling must not alter a caller-owned filter plan."""
35 input_format = AudioFormat(
36 content_type=ContentType.PCM_F32LE,
37 sample_rate=96000,
38 bit_depth=32,
39 channels=2,
40 )
41 output_format = AudioFormat(
42 content_type=ContentType.FLAC,
43 sample_rate=48000,
44 bit_depth=16,
45 channels=2,
46 )
47 filter_params = ["volume=-1dB"]
48
49 get_ffmpeg_args(input_format, output_format, filter_params)
50
51 assert filter_params == ["volume=-1dB"]
52
53
54def test_get_ffmpeg_args_downmixes_multichannel_for_single_channel_output() -> None:
55 """A surround source is folded to stereo before the output is narrowed to one channel."""
56 input_format = AudioFormat(
57 content_type=ContentType.PCM_F32LE,
58 sample_rate=48000,
59 bit_depth=32,
60 channels=6,
61 )
62 output_format = AudioFormat(
63 content_type=ContentType.FLAC,
64 sample_rate=48000,
65 bit_depth=16,
66 channels=1,
67 )
68
69 args = get_ffmpeg_args(input_format, output_format, ["pan=mono|c0=0.5*FL+0.5*FR"])
70
71 filter_graph = args[args.index("-af") + 1]
72 assert filter_graph.index("aformat=channel_layouts=stereo") < filter_graph.index(
73 "pan=mono|c0=0.5*FL+0.5*FR"
74 )
75
76
77def _split_at_input(args: list[str]) -> tuple[list[str], list[str]]:
78 """Split generated ffmpeg args into the part describing the input and the output."""
79 idx = args.index("-i")
80 return args[:idx], args[idx + 2 :]
81
82
83@pytest.mark.parametrize(
84 ("channels", "expected_layout"),
85 [(1, "mono"), (2, "stereo")],
86)
87def test_get_ffmpeg_args_names_layout_up_to_stereo(channels: int, expected_layout: str) -> None:
88 """Mono and stereo PCM are described by both their channel count and their layout."""
89 fmt = AudioFormat(
90 content_type=ContentType.PCM_S24LE,
91 sample_rate=48000,
92 bit_depth=24,
93 channels=channels,
94 )
95
96 input_args, output_args = _split_at_input(get_ffmpeg_args(fmt, fmt, []))
97
98 for part in (input_args, output_args):
99 assert part[part.index("-ac") + 1] == str(channels)
100 assert part[part.index("-channel_layout") + 1] == expected_layout
101
102
103def test_get_ffmpeg_args_omits_layout_above_stereo() -> None:
104 """Surround PCM is described by its channel count alone, never as a stereo layout."""
105 fmt = AudioFormat(
106 content_type=ContentType.PCM_S24LE,
107 sample_rate=48000,
108 bit_depth=24,
109 channels=6,
110 )
111
112 input_args, output_args = _split_at_input(get_ffmpeg_args(fmt, fmt, []))
113
114 for part in (input_args, output_args):
115 assert part[part.index("-ac") + 1] == "6"
116 assert "-channel_layout" not in part
117
118
119def test_multichannel_pcm_folds_down_without_stretching(tmp_path: Path) -> None:
120 """Raw surround PCM keeps its real width and length, and its rear channels survive."""
121 source = tmp_path / "surround.pcm"
122 out = tmp_path / "out.flac"
123 # only the rear channels carry a tone: a downmix that drops them yields silence
124 subprocess.run( # noqa: S603
125 [ # noqa: S607
126 "ffmpeg",
127 "-y",
128 "-f",
129 "lavfi",
130 "-i",
131 "sine=frequency=1000:duration=1:sample_rate=48000",
132 "-af",
133 "pan=5.1|BL=c0|BR=c0",
134 "-f",
135 "s16le",
136 str(source),
137 ],
138 check=True,
139 capture_output=True,
140 )
141 pcm_format = AudioFormat(
142 content_type=ContentType.PCM_S16LE,
143 sample_rate=48000,
144 bit_depth=16,
145 channels=6,
146 )
147 output_format = AudioFormat(
148 content_type=ContentType.FLAC,
149 sample_rate=48000,
150 bit_depth=16,
151 channels=2,
152 )
153 args = get_ffmpeg_args(
154 pcm_format, output_format, [], input_path=str(source), output_path=str(out)
155 )
156
157 result = subprocess.run([*args, "-y"], capture_output=True, text=True, check=False) # noqa: S603
158 assert result.returncode == 0, result.stderr
159
160 duration = subprocess.run( # noqa: S603
161 [ # noqa: S607
162 "ffprobe",
163 "-v",
164 "error",
165 "-show_entries",
166 "format=duration",
167 "-of",
168 "csv=p=0",
169 str(out),
170 ],
171 capture_output=True,
172 text=True,
173 check=True,
174 ).stdout
175 assert float(duration) == pytest.approx(1.0, abs=0.05)
176 assert _rms_db(out) > -40
177
178
179def _output_args(args: list[str]) -> list[str]:
180 """Return the output section of an ffmpeg command line (everything past the input path)."""
181 return args[args.index("-i") + 2 :]
182
183
184@pytest.mark.parametrize(
185 ("content_type", "encoder_args"),
186 [
187 (ContentType.AAC, ["-f", "adts", "-c:a", "aac", "-b:a", "256k"]),
188 (ContentType.MP3, ["-f", "mp3", "-b:a", "320k"]),
189 (ContentType.WAV, ["-ar", "44100", "-acodec", "pcm_s16le", "-f", "wav"]),
190 (
191 ContentType.FLAC,
192 ["-sample_fmt", "s16", "-ar", "44100", "-f", "flac", "-compression_level", "0"],
193 ),
194 ],
195)
196def test_get_ffmpeg_args_encoded_output_declares_channels(
197 content_type: ContentType, encoder_args: list[str]
198) -> None:
199 """Every encoded output format is handed the requested channel count."""
200 input_format = AudioFormat(
201 content_type=ContentType.PCM_S16LE, sample_rate=44100, bit_depth=16, channels=2
202 )
203 output_format = AudioFormat(content_type=content_type, sample_rate=44100, bit_depth=16)
204
205 args = get_ffmpeg_args(input_format, output_format, [])
206
207 assert _output_args(args) == ["-ac", "2", "-channel_layout", "stereo", *encoder_args, "-"]
208
209
210def test_get_ffmpeg_args_single_channel_output_declares_mono() -> None:
211 """A one channel target is declared as mono rather than stereo."""
212 fmt = AudioFormat(
213 content_type=ContentType.PCM_S16LE, sample_rate=44100, bit_depth=16, channels=1
214 )
215 output_format = AudioFormat(
216 content_type=ContentType.WAV, sample_rate=44100, bit_depth=16, channels=1
217 )
218
219 args = get_ffmpeg_args(fmt, output_format, [])
220
221 assert _output_args(args) == [
222 "-ac",
223 "1",
224 "-channel_layout",
225 "mono",
226 "-ar",
227 "44100",
228 "-acodec",
229 "pcm_s16le",
230 "-f",
231 "wav",
232 "-",
233 ]
234
235
236@pytest.mark.parametrize(
237 ("output_path", "output_content_type", "expected"),
238 [
239 ("NULL", ContentType.FLAC, ["-f", "null", "-"]),
240 ("-", ContentType.NUT, ["-vn", "-dn", "-sn", "-acodec", "copy", "-f", "nut", "-"]),
241 ],
242)
243def test_get_ffmpeg_args_passthrough_sinks_omit_channels(
244 output_path: str, output_content_type: ContentType, expected: list[str]
245) -> None:
246 """The analysis sink and the cache passthrough declare no channel count of their own."""
247 input_format = AudioFormat(
248 content_type=ContentType.PCM_S16LE, sample_rate=44100, bit_depth=16, channels=1
249 )
250 output_format = AudioFormat(
251 content_type=output_content_type, sample_rate=44100, bit_depth=16, channels=1
252 )
253
254 args = get_ffmpeg_args(input_format, output_format, [], output_path=output_path)
255
256 assert _output_args(args) == expected
257
258
259def test_get_ffmpeg_args_duplicates_mono_source_for_stereo_output() -> None:
260 """A mono source is widened by duplication, ahead of the caller's own filters."""
261 input_format = AudioFormat(
262 content_type=ContentType.PCM_S16LE, sample_rate=44100, bit_depth=16, channels=1
263 )
264 output_format = AudioFormat(
265 content_type=ContentType.FLAC, sample_rate=44100, bit_depth=16, channels=2
266 )
267
268 args = get_ffmpeg_args(input_format, output_format, ["volume=-1dB"])
269
270 assert args[args.index("-af") + 1] == "pan=stereo|c0=c0|c1=c0,volume=-1dB"
271
272
273def test_get_ffmpeg_args_mono_source_to_mono_output_is_not_widened() -> None:
274 """A mono source kept at one channel needs no channel filter at all."""
275 fmt = AudioFormat(
276 content_type=ContentType.PCM_S16LE, sample_rate=44100, bit_depth=16, channels=1
277 )
278 output_format = AudioFormat(
279 content_type=ContentType.FLAC, sample_rate=44100, bit_depth=16, channels=1
280 )
281
282 args = get_ffmpeg_args(fmt, output_format, [])
283
284 assert "-af" not in args
285
286
287# -- parse_ffmpeg_stream_info --
288
289
290def test_parse_stream_info_mp3() -> None:
291 """Lossy MP3 line yields codec/sample rate/bit rate, but no bit depth."""
292 line = "Stream #0:0: Audio: mp3, 44100 Hz, stereo, fltp, 320 kb/s"
293 info = parse_ffmpeg_stream_info(line)
294 assert info == FFMpegStreamInfo(
295 codec=ContentType.MP3,
296 sample_rate=44100,
297 bit_depth=None,
298 bit_rate=320,
299 )
300
301
302def test_parse_stream_info_aac_with_profile_and_language() -> None:
303 """AAC line with profile annotation and language tag is parsed correctly."""
304 line = "Stream #0:0(eng): Audio: aac (LC) (mp4a / 0x6134706D), 44100 Hz, stereo, fltp, 254 kb/s"
305 info = parse_ffmpeg_stream_info(line)
306 assert info is not None
307 assert info.codec == ContentType.AAC
308 assert info.sample_rate == 44100
309 assert info.bit_rate == 254
310 assert info.bit_depth is None
311
312
313def test_parse_stream_info_flac_16bit() -> None:
314 """16-bit FLAC: bit depth is inferred from the s16 sample format token."""
315 line = "Stream #0:0: Audio: flac, 44100 Hz, stereo, s16, 1024 kb/s"
316 info = parse_ffmpeg_stream_info(line)
317 assert info == FFMpegStreamInfo(
318 codec=ContentType.FLAC,
319 sample_rate=44100,
320 bit_depth=16,
321 bit_rate=1024,
322 )
323
324
325def test_parse_stream_info_flac_24bit_in_s32() -> None:
326 """24-bit FLAC is stored in s32; the explicit "(24 bit)" annotation wins."""
327 line = "Stream #0:0: Audio: flac, 96000 Hz, stereo, s32 (24 bit)"
328 info = parse_ffmpeg_stream_info(line)
329 assert info == FFMpegStreamInfo(
330 codec=ContentType.FLAC,
331 sample_rate=96000,
332 bit_depth=24,
333 bit_rate=None,
334 )
335
336
337def test_parse_stream_info_flac_24bit_hires_with_bitrate() -> None:
338 """High-resolution 24-bit FLAC at 192k with reported bit rate."""
339 line = "Stream #0:0: Audio: flac, 192000 Hz, stereo, s32 (24 bit), 5644 kb/s"
340 info = parse_ffmpeg_stream_info(line)
341 assert info == FFMpegStreamInfo(
342 codec=ContentType.FLAC,
343 sample_rate=192000,
344 bit_depth=24,
345 bit_rate=5644,
346 )
347
348
349def test_parse_stream_info_pcm_s16le() -> None:
350 """PCM stream reports codec via try_parse, sample format gives bit depth."""
351 line = "Stream #0:0: Audio: pcm_s16le, 44100 Hz, stereo, s16, 1411 kb/s"
352 info = parse_ffmpeg_stream_info(line)
353 assert info == FFMpegStreamInfo(
354 codec=ContentType.PCM_S16LE,
355 sample_rate=44100,
356 bit_depth=16,
357 bit_rate=1411,
358 )
359
360
361def test_parse_stream_info_opus_without_bitrate() -> None:
362 """Opus often omits bit rate; we still get codec and sample rate."""
363 line = "Stream #0:0: Audio: opus, 48000 Hz, stereo, fltp"
364 info = parse_ffmpeg_stream_info(line)
365 assert info == FFMpegStreamInfo(
366 codec=ContentType.OPUS,
367 sample_rate=48000,
368 bit_depth=None,
369 bit_rate=None,
370 )
371
372
373def test_parse_stream_info_alac_planar() -> None:
374 """ALAC reported with s16p (planar) sample format still yields 16-bit depth."""
375 line = "Stream #0:0: Audio: alac (alac / 0x63616C61), 44100 Hz, stereo, s16p"
376 info = parse_ffmpeg_stream_info(line)
377 assert info is not None
378 assert info.codec == ContentType.ALAC
379 assert info.sample_rate == 44100
380 assert info.bit_depth == 16
381
382
383def test_parse_stream_info_returns_none_for_non_stream_line() -> None:
384 """Non-stream log lines must return None."""
385 assert parse_ffmpeg_stream_info("Duration: 00:03:25.78, start: 0.000000") is None
386 assert parse_ffmpeg_stream_info("[error] Invalid data found") is None
387 assert parse_ffmpeg_stream_info("") is None
388
389
390def test_parse_stream_info_ignores_video_stream() -> None:
391 """Video stream lines must not be misparsed as audio."""
392 line = "Stream #0:0: Video: h264 (High), yuv420p, 1920x1080, 5000 kb/s, 25 fps"
393 assert parse_ffmpeg_stream_info(line) is None
394
395
396def test_parse_stream_info_unknown_codec_still_yields_other_fields() -> None:
397 """Unrecognised codec token returns UNKNOWN but sample rate / bit rate are still parsed."""
398 line = "Stream #0:0: Audio: somenewcodec, 48000 Hz, stereo, 192 kb/s"
399 info = parse_ffmpeg_stream_info(line)
400 assert info is not None
401 assert info.codec == ContentType.UNKNOWN
402 assert info.sample_rate == 48000
403 assert info.bit_rate == 192
404 assert info.bit_depth is None
405
406
407# -- parse_ffmpeg_duration --
408
409
410def test_parse_duration_typical() -> None:
411 """Typical 'Duration: HH:MM:SS.ms' line yields total seconds (floor)."""
412 line = "Duration: 00:03:25.78, start: 0.000000, bitrate: 320 kb/s"
413 assert parse_ffmpeg_duration(line) == 3 * 60 + 25
414
415
416def test_parse_duration_one_hour() -> None:
417 """Hours component is honoured."""
418 assert parse_ffmpeg_duration("Duration: 01:00:00.00, bitrate: 128 kb/s") == 3600
419
420
421def test_parse_duration_under_one_second() -> None:
422 """Sub-second durations round down to 0."""
423 assert parse_ffmpeg_duration("Duration: 00:00:00.50, bitrate: 128 kb/s") == 0
424
425
426def test_parse_duration_na_returns_none() -> None:
427 """Live streams report 'Duration: N/A' — must not match."""
428 assert parse_ffmpeg_duration("Duration: N/A, start: 0.000000, bitrate: N/A") is None
429
430
431def test_parse_duration_unrelated_line_returns_none() -> None:
432 """Random log lines must return None."""
433 assert parse_ffmpeg_duration("Stream #0:0: Audio: mp3, 44100 Hz") is None
434 assert parse_ffmpeg_duration("") is None
435
436
437# -- get_ffmpeg_overlay_stream (end-to-end with a real ffmpeg process) --
438
439_PCM_FORMAT = AudioFormat(
440 content_type=ContentType.PCM_S16LE, sample_rate=44100, bit_depth=16, channels=2
441)
442_BYTES_PER_SECOND = _PCM_FORMAT.pcm_sample_size # 1 second of PCM audio
443
444
445@pytest.fixture
446def overlay_file(tmp_path: Path) -> Path:
447 """Generate a 1 second mono sine-tone wav file to use as overlay source."""
448 overlay_path = tmp_path / "overlay.wav"
449 subprocess.run( # noqa: S603
450 ["ffmpeg", "-f", "lavfi", "-i", "sine=frequency=440:duration=1", str(overlay_path)], # noqa: S607
451 check=True,
452 capture_output=True,
453 )
454 return overlay_path
455
456
457@pytest.fixture
458def overlay_file_stereo(tmp_path: Path) -> Path:
459 """Generate a stereo overlay wav carrying the same tone as ``overlay_file`` on both channels."""
460 overlay_path = tmp_path / "overlay_stereo.wav"
461 subprocess.run( # noqa: S603
462 [ # noqa: S607
463 "ffmpeg",
464 "-f",
465 "lavfi",
466 "-i",
467 "sine=frequency=440:duration=1",
468 "-af",
469 "pan=stereo|c0=c0|c1=c0",
470 str(overlay_path),
471 ],
472 check=True,
473 capture_output=True,
474 )
475 return overlay_path
476
477
478@pytest.fixture
479def overlay_file_wide_stereo(tmp_path: Path) -> Path:
480 """Generate a 1 second stereo overlay wav with the tone on the left channel only."""
481 overlay_path = tmp_path / "overlay_wide.wav"
482 subprocess.run( # noqa: S603
483 [ # noqa: S607
484 "ffmpeg",
485 "-f",
486 "lavfi",
487 "-i",
488 "sine=frequency=440:duration=1",
489 "-af",
490 "pan=stereo|c0=c0",
491 str(overlay_path),
492 ],
493 check=True,
494 capture_output=True,
495 )
496 return overlay_path
497
498
499@pytest.fixture
500def overlay_file_with_silent_intro(tmp_path: Path) -> Path:
501 """Generate a 2 second overlay wav that starts with 1s of silence then a 1s tone."""
502 overlay_path = tmp_path / "overlay_silent_intro.wav"
503 subprocess.run( # noqa: S603
504 [ # noqa: S607
505 "ffmpeg",
506 "-f",
507 "lavfi",
508 "-i",
509 "sine=frequency=440:duration=1",
510 "-af",
511 "adelay=1000:all=1",
512 str(overlay_path),
513 ],
514 check=True,
515 capture_output=True,
516 )
517 return overlay_path
518
519
520async def _silence(seconds: int) -> AsyncGenerator[bytes]:
521 """Yield the given amount of seconds of PCM silence in 1-second chunks."""
522 for _ in range(seconds):
523 yield b"\x00" * _BYTES_PER_SECOND
524
525
526async def _collect_chunks(stream: AsyncGenerator[bytes]) -> list[bytes]:
527 return [chunk async for chunk in stream]
528
529
530@pytest.mark.parametrize("source_error", [RuntimeError("source failed"), BrokenPipeError()])
531async def test_ffmpeg_stream_surfaces_stdin_feeder_error(source_error: Exception) -> None:
532 """An input generator failure is surfaced after FFmpeg emits its buffered output."""
533
534 async def failing_input() -> AsyncGenerator[bytes]:
535 yield b"\x00" * _BYTES_PER_SECOND
536 raise source_error
537
538 with pytest.raises(AudioError, match="Error while feeding audio to FFmpeg") as err:
539 await _collect_chunks(
540 get_ffmpeg_stream(
541 audio_input=failing_input(),
542 input_format=AudioFormat(
543 content_type=ContentType.PCM_S16LE,
544 sample_rate=44100,
545 bit_depth=16,
546 channels=2,
547 ),
548 output_format=_PCM_FORMAT,
549 )
550 )
551
552 assert err.value.__cause__ is source_error
553
554
555async def test_ffmpeg_stream_ignores_cancelled_stdin_feeder() -> None:
556 """A cancelled input generator ends the FFmpeg stream without an error."""
557
558 async def cancelled_input() -> AsyncGenerator[bytes]:
559 yield b"\x00" * _BYTES_PER_SECOND
560 raise asyncio.CancelledError
561
562 chunks = await _collect_chunks(
563 get_ffmpeg_stream(
564 audio_input=cancelled_input(),
565 input_format=AudioFormat(
566 content_type=ContentType.PCM_S16LE,
567 sample_rate=44100,
568 bit_depth=16,
569 channels=2,
570 ),
571 output_format=_PCM_FORMAT,
572 )
573 )
574
575 assert b"".join(chunks) == b"\x00" * _BYTES_PER_SECOND
576
577
578async def test_ffmpeg_stream_ignores_early_stdin_close() -> None:
579 """FFmpeg ending its input early does not report a source failure."""
580 chunks = await _collect_chunks(
581 get_ffmpeg_stream(
582 audio_input=_silence(30),
583 input_format=_PCM_FORMAT,
584 output_format=_PCM_FORMAT,
585 extra_input_args=["-t", "0.1"],
586 )
587 )
588
589 assert chunks
590
591
592async def test_overlay_stream_surfaces_stdin_feeder_error(overlay_file: Path) -> None:
593 """A failure in the main input is surfaced after the mixed output is emitted."""
594
595 async def failing_input() -> AsyncGenerator[bytes]:
596 yield b"\x00" * _BYTES_PER_SECOND
597 raise RuntimeError("source failed")
598
599 with pytest.raises(AudioError, match="Error while feeding audio to FFmpeg") as err:
600 await _collect_chunks(
601 get_ffmpeg_overlay_stream(
602 audio_input=failing_input(),
603 overlay_input=str(overlay_file),
604 pcm_format=_PCM_FORMAT,
605 )
606 )
607
608 assert isinstance(err.value.__cause__, RuntimeError)
609
610
611def _samples(pcm: bytes, channel: int | None = None) -> Sequence[int]:
612 """Return the samples of the given PCM audio, optionally for one channel only."""
613 samples = array("h")
614 samples.frombytes(pcm)
615 return samples if channel is None else samples[channel :: _PCM_FORMAT.channels]
616
617
618def _rms(samples: Sequence[int]) -> float:
619 """Return the RMS level of the given PCM samples."""
620 return sqrt(sum(sample * sample for sample in samples) / len(samples))
621
622
623async def _mix_overlay(overlay_input: Path) -> bytes:
624 """Mix the given overlay source into 1 second of silence and return the result."""
625 return b"".join(
626 await _collect_chunks(
627 get_ffmpeg_overlay_stream(
628 audio_input=_silence(1),
629 overlay_input=str(overlay_input),
630 pcm_format=_PCM_FORMAT,
631 )
632 )
633 )
634
635
636async def test_overlay_stream_mixes_loops_and_preserves_length(overlay_file: Path) -> None:
637 """The overlay is looped and mixed in while length, format and chunking stay intact."""
638 chunks = await _collect_chunks(
639 get_ffmpeg_overlay_stream(
640 audio_input=_silence(3),
641 overlay_input=str(overlay_file),
642 pcm_format=_PCM_FORMAT,
643 chunk_size=_BYTES_PER_SECOND,
644 )
645 )
646 output = b"".join(chunks)
647 # duration=first: output length exactly matches the 3s main input
648 assert len(output) == 3 * _BYTES_PER_SECOND
649 # all chunks except the last are exactly chunk_size
650 assert all(len(chunk) == _BYTES_PER_SECOND for chunk in chunks[:-1])
651 # the main input was pure silence, so any signal proves the overlay was mixed in;
652 # signal in the third second proves the 1s overlay file was looped
653 assert any(output[:_BYTES_PER_SECOND])
654 assert any(output[2 * _BYTES_PER_SECOND :])
655
656
657async def test_overlay_stream_does_not_mutate_pcm_format(overlay_file: Path) -> None:
658 """Mixing an overlay leaves the caller's PCM format unchanged."""
659 pcm_format = AudioFormat(
660 content_type=ContentType.PCM_F32LE,
661 sample_rate=48000,
662 bit_depth=32,
663 channels=2,
664 )
665 original_format = pcm_format.to_dict()
666
667 async def silence() -> AsyncGenerator[bytes]:
668 yield b"\x00" * pcm_format.pcm_sample_size
669
670 await _collect_chunks(
671 get_ffmpeg_overlay_stream(
672 audio_input=silence(),
673 overlay_input=str(overlay_file),
674 pcm_format=pcm_format,
675 )
676 )
677
678 assert pcm_format.to_dict() == original_format
679
680
681async def test_overlay_stream_applies_volume(overlay_file: Path) -> None:
682 """Overlay volume 0% silences the overlay entirely (gain is applied)."""
683 output = b"".join(
684 await _collect_chunks(
685 get_ffmpeg_overlay_stream(
686 audio_input=_silence(1),
687 overlay_input=str(overlay_file),
688 pcm_format=_PCM_FORMAT,
689 overlay_volume=0,
690 )
691 )
692 )
693 assert len(output) == _BYTES_PER_SECOND
694 assert not any(output)
695
696
697async def test_overlay_stream_trims_leading_silence(
698 overlay_file_with_silent_intro: Path,
699) -> None:
700 """A near-silent intro on the overlay source is trimmed so it plays immediately."""
701 output = b"".join(
702 await _collect_chunks(
703 get_ffmpeg_overlay_stream(
704 audio_input=_silence(1),
705 overlay_input=str(overlay_file_with_silent_intro),
706 pcm_format=_PCM_FORMAT,
707 )
708 )
709 )
710 assert len(output) == _BYTES_PER_SECOND
711 # without trimming, the first second would be the overlay's silent intro;
712 # the trim makes the tone play from the start, so the first second has signal
713 assert any(output)
714
715
716async def test_overlay_stream_level_is_independent_of_source_channel_count(
717 overlay_file: Path, overlay_file_stereo: Path
718) -> None:
719 """A mono overlay source mixes in at the same level as an identical stereo one."""
720 mono_level = _rms(_samples(await _mix_overlay(overlay_file)))
721 stereo_level = _rms(_samples(await _mix_overlay(overlay_file_stereo)))
722 assert mono_level > 0
723 # left to FFmpeg, the mono source would be spread at 1/sqrt(2) per channel
724 # and land a factor sqrt(2) (3 dB) below the stereo one
725 assert mono_level == pytest.approx(stereo_level, rel=0.02)
726
727
728async def test_overlay_stream_preserves_stereo_image(overlay_file_wide_stereo: Path) -> None:
729 """A stereo overlay keeps its channels apart instead of being folded to dual mono."""
730 output = await _mix_overlay(overlay_file_wide_stereo)
731 # the source carries the tone on the left only, so a fold would leak it into the right
732 assert _rms(_samples(output, channel=0)) > 0
733 assert not any(_samples(output, channel=1))
734
735
736# -- overlay volume filter --
737
738
739def test_overlay_volume_filter_compensates_only_for_a_stereo_output() -> None:
740 """Only the widening to stereo costs a mono source level, so only there is it scaled up."""
741 stereo_output = _get_overlay_volume_filter(100, 2)
742 assert "nb_channels" in stereo_output
743 # a comma would read as the end of this filter in the graph
744 assert "," not in stereo_output
745 # a mono source keeps its level when it is widened further, or not at all
746 assert _get_overlay_volume_filter(100, 1) == "volume=1.0"
747 assert _get_overlay_volume_filter(60, 6) == "volume=0.6"
748
749
750def test_overlay_mixer_loops_its_source() -> None:
751 """The overlay source is looped for as long as the main stream runs."""
752 (overlay,) = _build_overlay_mixer("/sound.wav", _PCM_FORMAT, 100).inputs
753 # a local file has nothing to reconnect to
754 assert overlay.input_args == ["-stream_loop", "-1"]
755
756
757def test_overlay_mixer_reconnects_for_http_sources() -> None:
758 """An http overlay source additionally gets the reconnect options."""
759 (overlay,) = _build_overlay_mixer("http://host/sound.mp3", _PCM_FORMAT, 100).inputs
760 assert "-reconnect" in overlay.input_args
761 assert overlay.input_args[-2:] == ["-stream_loop", "-1"]
762
763
764def test_overlay_args_probe_the_main_input_and_add_no_filters() -> None:
765 """Both inputs are read under our own limits and no filter is injected on top."""
766 args = get_ffmpeg_args(
767 _PCM_FORMAT, _PCM_FORMAT, [_build_overlay_mixer("/sound.wav", _PCM_FORMAT, 100)]
768 )
769 main_input = args.index("-i")
770 # the limits must reach the main input, which is the first one, and the overlay
771 assert args.index("-probesize") < main_input
772 assert args.count("-probesize") == 2
773 assert args.index("-stream_loop") > main_input
774 # the overlay format matches the output, so nothing gets resampled or reconformed
775 assert "-af" not in args
776 assert args.count("-filter_complex") == 1
777 assert not any(arg.startswith(("pan=", "aresample=resampler=")) for arg in args)
778
779
780@pytest.mark.parametrize(
781 "extra_input_args",
782 [
783 # the concat demuxer brings its own -f, and needs the whitelist for the listed files
784 ["-safe", "0", "-f", "concat", "-i", "/list.txt"],
785 # a caller raising the probe limits relies on its own values landing last
786 ["-probesize", "65536", "-analyzeduration", "5000000"],
787 ],
788 ids=["caller-supplied-input-format", "caller-raised-probe-limits"],
789)
790def test_read_args_lead_the_main_input(extra_input_args: list[str]) -> None:
791 """Every main input is opened under our read limits, which the caller may override."""
792 args = get_ffmpeg_args(_PCM_FORMAT, _PCM_FORMAT, [], extra_input_args=extra_input_args)
793 start = args.index("-protocol_whitelist")
794 end = start + len(_INPUT_READ_ARGS)
795 assert args[start:end] == _INPUT_READ_ARGS
796 # the caller's args follow ours within the same input group, so theirs win on a conflict
797 assert args[end : end + len(extra_input_args)] == extra_input_args
798 # exactly one input either way: we add ours only when the caller brings none
799 assert args.count("-i") == 1
800
801
802# -- _log_reader_task (decode-error flood guard) --
803
804
805class _FakeStream:
806 """Stand-in for a StreamReader/StreamWriter: already closed/at EOF, nothing to drain."""
807
808 def is_closing(self) -> bool:
809 return True
810
811 def at_eof(self) -> bool:
812 return True
813
814
815class _FakeProc:
816 """Minimal stand-in for asyncio.subprocess.Process — just enough for close() to run."""
817
818 def __init__(self) -> None:
819 self.pid = 12345
820 self.returncode: int | None = None
821 self.stdin: _FakeStream | None = _FakeStream()
822 self.stdout = _FakeStream()
823
824 async def communicate(self) -> tuple[bytes, bytes]:
825 self.returncode = 0
826 return b"", b""
827
828 def send_signal(self, _sig: int) -> None:
829 pass
830
831
832class _FakeProcRacingExit(_FakeProc):
833 """A no-stdin process that has already exited, so send_signal raises ProcessLookupError."""
834
835 def __init__(self) -> None:
836 super().__init__()
837 # no stdin routes close() down the send_signal(SIGINT) branch
838 self.stdin = None
839
840 def send_signal(self, _sig: int) -> None:
841 raise ProcessLookupError("no such process")
842
843
844async def test_log_reader_reports_decode_errors_once_and_aborts() -> None:
845 """
846 Crossing the decode-error threshold logs a single line and aborts the stream.
847
848 Regression test: previously every stderr line was re-promoted to ERROR for the
849 rest of the process once 50 "Invalid data" lines were seen, flooding the log
850 with thousands of lines for a single corrupted file.
851 """
852 ffmpeg = FFMpeg(audio_input="-", input_format=_PCM_FORMAT, output_format=_PCM_FORMAT)
853
854 async def fake_stderr() -> AsyncGenerator[str]:
855 for _ in range(60):
856 yield "Invalid data found when processing input"
857 # noise that a genuinely corrupted stream keeps emitting after the threshold;
858 # none of this should reach ERROR level under the fix
859 for _ in range(20):
860 yield "Reserved bit set."
861
862 ffmpeg.iter_stderr = fake_stderr # type: ignore[method-assign]
863
864 error_lines: list[str] = []
865 ffmpeg.logger.error = lambda msg, *args: error_lines.append(msg % args if args else msg) # type: ignore[method-assign]
866
867 await ffmpeg._log_reader_task()
868 assert ffmpeg._abort_task is not None
869 await ffmpeg._abort_task
870
871 assert error_lines == ["Excessive decode errors (50+) for this stream; aborting"]
872 assert ffmpeg.closed
873
874
875async def test_log_reader_below_threshold_does_not_abort() -> None:
876 """A handful of decode errors, well under the threshold, triggers no report or abort."""
877 ffmpeg = FFMpeg(audio_input="-", input_format=_PCM_FORMAT, output_format=_PCM_FORMAT)
878
879 async def fake_stderr() -> AsyncGenerator[str]:
880 for _ in range(10):
881 yield "Invalid data found when processing input"
882 yield "Reserved bit set."
883
884 ffmpeg.iter_stderr = fake_stderr # type: ignore[method-assign]
885
886 error_lines: list[str] = []
887 ffmpeg.logger.error = lambda msg, *args: error_lines.append(msg % args if args else msg) # type: ignore[method-assign]
888
889 await ffmpeg._log_reader_task()
890
891 assert error_lines == []
892 assert ffmpeg._abort_task is None
893 assert not ffmpeg.closed
894
895
896async def test_log_reader_abort_does_not_self_deadlock() -> None:
897 """
898 The detached abort task can close() the reader's own process without a self-await.
899
900 Regression test for the deadlock this PR reintroduces abort-on-close around:
901 close() does ``await asyncio.wait_for(self._stderr_reader_task, 5)``, and
902 ``_stderr_reader_task`` here is wired to the very task running
903 ``_log_reader_task`` — the same setup ``start()`` uses in production. If the
904 abort were awaited inline from within ``_log_reader_task`` instead of via a
905 detached task, that task would be awaiting itself, which asyncio turns into
906 ``RuntimeError: Task cannot await on itself`` rather than a hang.
907 """
908 ffmpeg = FFMpeg(audio_input="-", input_format=_PCM_FORMAT, output_format=_PCM_FORMAT)
909 ffmpeg.proc = _FakeProc() # type: ignore[assignment]
910
911 async def fake_stderr() -> AsyncGenerator[str]:
912 for _ in range(50):
913 yield "Invalid data found when processing input"
914
915 ffmpeg.iter_stderr = fake_stderr # type: ignore[method-assign]
916
917 reader_task = asyncio.create_task(ffmpeg._log_reader_task())
918 ffmpeg._stderr_reader_task = reader_task
919
920 await asyncio.wait_for(reader_task, timeout=2)
921 assert ffmpeg._abort_task is not None
922 await asyncio.wait_for(ffmpeg._abort_task, timeout=2)
923
924 assert ffmpeg.closed
925
926
927async def test_abort_survives_send_signal_racing_process_exit() -> None:
928 """
929 The fire-and-forget abort must not crash if the process exits before it is signalled.
930
931 close() sends SIGINT to no-stdin processes, which raises ProcessLookupError when the
932 process already exited between the returncode check and the signal. The abort task is
933 never awaited by a caller, so that race must be swallowed inside close() rather than
934 escaping as an untracked "Task exception was never retrieved".
935 """
936 ffmpeg = FFMpeg(audio_input="-", input_format=_PCM_FORMAT, output_format=_PCM_FORMAT)
937 ffmpeg.proc = _FakeProcRacingExit() # type: ignore[assignment]
938
939 async def fake_stderr() -> AsyncGenerator[str]:
940 for _ in range(50):
941 yield "Invalid data found when processing input"
942
943 ffmpeg.iter_stderr = fake_stderr # type: ignore[method-assign]
944
945 reader_task = asyncio.create_task(ffmpeg._log_reader_task())
946 ffmpeg._stderr_reader_task = reader_task
947
948 await asyncio.wait_for(reader_task, timeout=2)
949 assert ffmpeg._abort_task is not None
950 # must complete without propagating ProcessLookupError
951 await asyncio.wait_for(ffmpeg._abort_task, timeout=2)
952
953 assert ffmpeg.closed
954
955
956# -- _build_filtergraph_args (DSP chain assembly) --
957
958
959def test_build_filtergraph_all_simple_uses_af() -> None:
960 """A chain of plain filters renders to a single -af comma chain."""
961 assert _build_filtergraph_args(["equalizer=x", "volume=3dB"]) == (
962 [],
963 ["-af", "equalizer=x,volume=3dB"],
964 )
965
966
967def test_build_filtergraph_empty_returns_no_args() -> None:
968 """An empty chain produces no ffmpeg arguments."""
969 assert _build_filtergraph_args([]) == ([], [])
970
971
972def test_build_filtergraph_single_complex_fragment() -> None:
973 """A complex fragment renders a labelled -filter_complex graph with -map."""
974 result = _build_filtergraph_args(
975 [ComplexFilter("afir=irnorm=1", [ComplexFilterInput("/ir.wav", "aresample=48000")])]
976 )
977 assert result == (
978 [*_INPUT_READ_ARGS, "-i", "/ir.wav"],
979 [
980 "-filter_complex",
981 "[1:a]aresample=48000[dsp1];[0:a][dsp1]afir=irnorm=1[dsp2]",
982 "-map",
983 "[dsp2]",
984 ],
985 )
986
987
988def test_build_filtergraph_complex_between_simple_runs() -> None:
989 """Simple runs on either side of a complex fragment weave into labelled pads."""
990 result = _build_filtergraph_args(
991 [
992 "equalizer=x",
993 ComplexFilter("afir=irnorm=1", [ComplexFilterInput("/ir.wav", "aresample=48000")]),
994 "volume=2dB",
995 ]
996 )
997 assert result == (
998 [*_INPUT_READ_ARGS, "-i", "/ir.wav"],
999 [
1000 "-filter_complex",
1001 "[0:a]equalizer=x[dsp1];[1:a]aresample=48000[dsp2];"
1002 "[dsp1][dsp2]afir=irnorm=1[dsp3];[dsp3]volume=2dB[dsp4]",
1003 "-map",
1004 "[dsp4]",
1005 ],
1006 )
1007
1008
1009def test_build_filtergraph_multiple_inputs() -> None:
1010 """A fragment with several inputs numbers them in order and feeds them to the body."""
1011 result = _build_filtergraph_args(
1012 [ComplexFilter("amerge", [ComplexFilterInput("/a.wav"), ComplexFilterInput("/b.wav")])]
1013 )
1014 assert result == (
1015 [*_INPUT_READ_ARGS, "-i", "/a.wav", *_INPUT_READ_ARGS, "-i", "/b.wav"],
1016 ["-filter_complex", "[0:a][1:a][2:a]amerge[dsp1]", "-map", "[dsp1]"],
1017 )
1018
1019
1020def test_build_filtergraph_input_args_precede_the_input() -> None:
1021 """An input's own ffmpeg options are emitted directly before its -i."""
1022 result = _build_filtergraph_args(
1023 [
1024 ComplexFilter(
1025 "amix=inputs=2",
1026 [ComplexFilterInput("/loop.wav", input_args=["-stream_loop", "-1"])],
1027 )
1028 ]
1029 )
1030 assert result == (
1031 [*_INPUT_READ_ARGS, "-stream_loop", "-1", "-i", "/loop.wav"],
1032 ["-filter_complex", "[0:a][1:a]amix=inputs=2[dsp1]", "-map", "[dsp1]"],
1033 )
1034
1035
1036def test_get_ffmpeg_args_uses_af_without_complex_filter() -> None:
1037 """Plain filter chains keep the -af path (no -filter_complex/-map)."""
1038 fmt = AudioFormat(
1039 content_type=ContentType.PCM_S16LE, sample_rate=48000, bit_depth=16, channels=2
1040 )
1041 args = get_ffmpeg_args(fmt, fmt, ["volume=-1dB"])
1042 assert "-af" in args
1043 assert "-filter_complex" not in args
1044
1045
1046def test_get_ffmpeg_args_uses_filter_complex_with_complex_filter() -> None:
1047 """A complex fragment switches the whole chain to -filter_complex with -map."""
1048 fmt = AudioFormat(
1049 content_type=ContentType.PCM_S16LE, sample_rate=48000, bit_depth=16, channels=2
1050 )
1051 args = get_ffmpeg_args(
1052 fmt, fmt, [ComplexFilter("afir=irnorm=1", [ComplexFilterInput("/ir.wav")])]
1053 )
1054 assert "-filter_complex" in args
1055 assert "-map" in args
1056 assert "-af" not in args
1057 # the impulse response is a real input, so it never passes through graph quoting
1058 assert args.count("-i") == 2
1059 assert args.index("/ir.wav") > args.index("-i")
1060
1061
1062def _rms_db(path: Path) -> float:
1063 """Return the overall RMS level of an audio file in dB via ffmpeg astats."""
1064 output = subprocess.run( # noqa: S603
1065 [ # noqa: S607
1066 "ffmpeg",
1067 "-hide_banner",
1068 "-nostats",
1069 "-i",
1070 str(path),
1071 "-af",
1072 "astats=measure_perchannel=none",
1073 "-f",
1074 "null",
1075 "-",
1076 ],
1077 capture_output=True,
1078 text=True,
1079 check=True,
1080 ).stderr
1081 for line in output.splitlines():
1082 if "RMS level dB" in line:
1083 return float(line.split("RMS level dB:")[-1])
1084 raise AssertionError("no RMS level in astats output")
1085
1086
1087def test_filtergraph_complex_runs_in_ffmpeg(tmp_path: Path) -> None:
1088 """The generated -filter_complex graph is valid and an identity IR passes audio through."""
1089 main = tmp_path / "main.wav"
1090 ir = tmp_path / "ir.wav"
1091 out = tmp_path / "out.wav"
1092 subprocess.run( # noqa: S603
1093 [ # noqa: S607
1094 "ffmpeg",
1095 "-y",
1096 "-f",
1097 "lavfi",
1098 "-i",
1099 "sine=frequency=1000:duration=1:sample_rate=48000",
1100 "-ac",
1101 "2",
1102 str(main),
1103 ],
1104 check=True,
1105 capture_output=True,
1106 )
1107 # a single-sample impulse is the identity IR: convolving with it returns the input
1108 subprocess.run( # noqa: S603
1109 [ # noqa: S607
1110 "ffmpeg",
1111 "-y",
1112 "-f",
1113 "lavfi",
1114 "-i",
1115 "aevalsrc=eq(n\\,0):d=0.01:s=48000:c=stereo",
1116 str(ir),
1117 ],
1118 check=True,
1119 capture_output=True,
1120 )
1121 input_args, filter_args = _build_filtergraph_args(
1122 [ComplexFilter("afir=irnorm=1", [ComplexFilterInput(str(ir), "aresample=48000")])]
1123 )
1124 result = subprocess.run( # noqa: S603
1125 [ # noqa: S607
1126 "ffmpeg",
1127 "-hide_banner",
1128 "-loglevel",
1129 "error",
1130 "-y",
1131 "-i",
1132 str(main),
1133 *input_args,
1134 *filter_args,
1135 str(out),
1136 ],
1137 capture_output=True,
1138 text=True,
1139 check=False,
1140 )
1141 assert result.returncode == 0, result.stderr
1142 assert out.exists()
1143 # identity IR => output level matches input level
1144 assert abs(_rms_db(out) - _rms_db(main)) < 0.5
1145
1146
1147def test_mono_source_keeps_its_level_when_widened_to_stereo(tmp_path: Path) -> None:
1148 """Widening a mono source to stereo duplicates it, where a rematrix would cost 3 dB."""
1149 main = tmp_path / "mono.wav"
1150 out = tmp_path / "stereo.wav"
1151 subprocess.run( # noqa: S603
1152 [ # noqa: S607
1153 "ffmpeg",
1154 "-y",
1155 "-f",
1156 "lavfi",
1157 "-i",
1158 "sine=frequency=1000:duration=1:sample_rate=44100",
1159 "-ac",
1160 "1",
1161 str(main),
1162 ],
1163 check=True,
1164 capture_output=True,
1165 )
1166 args = get_ffmpeg_args(
1167 AudioFormat(content_type=ContentType.WAV, sample_rate=44100, bit_depth=16, channels=1),
1168 AudioFormat(content_type=ContentType.WAV, sample_rate=44100, bit_depth=16, channels=2),
1169 [],
1170 input_path=str(main),
1171 output_path=str(out),
1172 )
1173 result = subprocess.run(args, capture_output=True, text=True, check=False) # noqa: S603
1174
1175 assert result.returncode == 0, result.stderr
1176 assert abs(_rms_db(out) - _rms_db(main)) < 0.5
1177