/
/
1"""Tests for the candidate generators' rung emission."""
2
3from __future__ import annotations
4
5import logging
6
7from music_assistant.controllers.streams.smart_fades.models import (
8 TransitionStrategy,
9 TransitionTier,
10)
11from music_assistant.controllers.streams.smart_fades.planner.candidates import (
12 _LAZY_OVERLAY_SECONDS,
13 EnergyLadderGenerator,
14 LazyOverlayGenerator,
15 TrimClosingAnchorGenerator,
16 _entry_options,
17 _vocal_duties,
18 _window_duties,
19 earns_instrumental_blend,
20)
21from music_assistant.controllers.streams.smart_fades.planner.context import (
22 TransitionContext,
23 build_transition_context,
24)
25from music_assistant.controllers.streams.smart_fades.planner.planner import SmartCrossFadePlanner
26from music_assistant.models.audio_analysis import AudioAnalysisData
27
28
29def _analysis(
30 bpm: float,
31 duration: float = 240.0,
32 grid_until: float | None = None,
33) -> AudioAnalysisData:
34 """Synthetic AudioAnalysisData with an even beat/downbeat grid, optionally truncated early."""
35 interval = 60.0 / bpm
36 count = int(duration / interval) + 1
37 beats = [i * interval for i in range(count)]
38 if grid_until is not None:
39 beats = [b for b in beats if b <= grid_until]
40 return AudioAnalysisData(
41 duration=duration,
42 bpm=bpm,
43 beats=beats,
44 downbeats=beats[::4],
45 beats_per_bar=4,
46 rms_energy=[0.8] * 1800,
47 key="C",
48 mode="minor",
49 extra_data={},
50 )
51
52
53def _instrumental_vs_vocal_ctx() -> TransitionContext:
54 """Build a transition context: outgoing instrumental, incoming vocal, both 128 BPM."""
55 beats = [i * 60 / 128 for i in range(int(180 * 128 / 60))]
56 downbeats = beats[::4]
57 aa_out = AudioAnalysisData(
58 duration=180.0,
59 bpm=128.0,
60 beats=beats,
61 downbeats=downbeats,
62 beats_per_bar=4,
63 rms_energy=[0.8] * 1800,
64 key="C",
65 mode="minor",
66 extra_data={"vocal_activity": [0.0] * 1800},
67 )
68 aa_in = AudioAnalysisData(
69 duration=180.0,
70 bpm=128.0,
71 beats=beats,
72 downbeats=downbeats,
73 beats_per_bar=4,
74 rms_energy=[0.8] * 1800,
75 key="C",
76 mode="minor",
77 extra_data={"vocal_activity": [0.9] * 1800},
78 )
79 return build_transition_context(aa_out, aa_in, 45.0, logging.getLogger("test"))
80
81
82def _big_trim_gap_ctx() -> TransitionContext:
83 """
84 Build a context whose energy anchor lands early, stranding audible tail behind it.
85
86 rms_energy holds at 0.9 for the first 70% of the buffer, drops to a
87 still-audible 0.25 until 95%, then to silence - no vocal data, so the
88 gap can only be closed by an energy-path generator.
89 """
90 beats = [i * 60 / 128 for i in range(int(45 * 128 / 60))]
91 downbeats = beats[::4]
92 rms_energy = [0.9] * 1260 + [0.25] * 450 + [0.0] * 90
93 aa_out = AudioAnalysisData(
94 duration=45.0,
95 bpm=128.0,
96 beats=beats,
97 downbeats=downbeats,
98 beats_per_bar=4,
99 rms_energy=rms_energy,
100 key="C",
101 mode="minor",
102 extra_data={},
103 )
104 aa_in = AudioAnalysisData(
105 duration=45.0,
106 bpm=128.0,
107 beats=beats,
108 downbeats=downbeats,
109 beats_per_bar=4,
110 rms_energy=[0.8] * 1800,
111 key="C",
112 mode="minor",
113 extra_data={},
114 )
115 ctx = build_transition_context(aa_out, aa_in, 45.0, logging.getLogger("test"))
116 assert ctx.audio_end - ctx.default_anchor >= 8.0
117 return ctx
118
119
120def _small_trim_gap_ctx() -> TransitionContext:
121 """Build a context with flat rms_energy, so the energy anchor already sits at the audible end."""
122 beats = [i * 60 / 128 for i in range(int(45 * 128 / 60))]
123 downbeats = beats[::4]
124 aa_out = AudioAnalysisData(
125 duration=45.0,
126 bpm=128.0,
127 beats=beats,
128 downbeats=downbeats,
129 beats_per_bar=4,
130 rms_energy=[0.8] * 1800,
131 key="C",
132 mode="minor",
133 extra_data={},
134 )
135 aa_in = AudioAnalysisData(
136 duration=45.0,
137 bpm=128.0,
138 beats=beats,
139 downbeats=downbeats,
140 beats_per_bar=4,
141 rms_energy=[0.8] * 1800,
142 key="C",
143 mode="minor",
144 extra_data={},
145 )
146 return build_transition_context(aa_out, aa_in, 45.0, logging.getLogger("test"))
147
148
149def test_energy_ladder_emits_only_plain_rungs() -> None:
150 """A one-instrumental/one-vocal pair gets the plain ladder, no 16-bar spec."""
151 instrumental_vs_vocal_ctx = _instrumental_vs_vocal_ctx()
152 specs = list(EnergyLadderGenerator().generate(instrumental_vs_vocal_ctx))
153 assert specs
154 assert all(spec.bars <= 8 for spec in specs)
155
156
157def test_trim_closing_ladder_emitted_for_big_trim_gap() -> None:
158 """An instrumental tail with a large audible gap past the energy anchor gets late-anchored rungs."""
159 ctx = _big_trim_gap_ctx()
160 specs = list(TrimClosingAnchorGenerator().generate(ctx))
161 assert specs
162 for spec in specs:
163 assert spec.anchor_s is not None
164 assert spec.anchor_s > ctx.default_anchor
165 assert spec.anchor_s <= ctx.audio_end
166 # the ladder is walked, not just one rung
167 assert len({spec.bars for spec in specs}) >= 2
168 # every rung shares the single late anchor, including the longest one
169 assert 8 in {spec.bars for spec in specs}
170
171
172def test_trim_closing_not_emitted_for_small_gap() -> None:
173 """A tail whose energy anchor already sits near the audible end emits nothing."""
174 specs = list(TrimClosingAnchorGenerator().generate(_small_trim_gap_ctx()))
175 assert specs == []
176
177
178def _late_blendable_only_ctx() -> TransitionContext:
179 """
180 Build a context whose early window is too sparse to blend but the late one qualifies.
181
182 A full 4/4 grid runs to the end of both 124 BPM decks with matching keys,
183 so the tier at the audible end is FULL_BLEND. The outgoing rms_energy
184 stays loud for only a few bars into the buffer before dropping to a
185 still-audible level and then real silence, so the early mix-out anchor
186 lands with too few downbeats behind it for the early window's tier check
187 to pass, while the audible tail runs on for many more bars past it.
188 """
189 beats = [i * 60 / 124 for i in range(int(240 * 124 / 60))]
190 downbeats = beats[::4]
191 rms_energy = [0.9] * 1550 + [0.25] * (1730 - 1550) + [0.0] * (1800 - 1730)
192 aa_out = AudioAnalysisData(
193 duration=240.0,
194 bpm=124.0,
195 beats=beats,
196 downbeats=downbeats,
197 beats_per_bar=4,
198 rms_energy=rms_energy,
199 key="A",
200 mode="minor",
201 extra_data={},
202 )
203 aa_in = AudioAnalysisData(
204 duration=240.0,
205 bpm=124.0,
206 beats=beats,
207 downbeats=downbeats,
208 beats_per_bar=4,
209 rms_energy=[0.8] * 1800,
210 key="A",
211 mode="minor",
212 extra_data={},
213 )
214 ctx = build_transition_context(aa_out, aa_in, 45.0, logging.getLogger("test"))
215 assert ctx.audio_end - ctx.default_anchor >= 8.0
216 early_downbeats = [d for d in ctx.outgoing.downbeats if d <= ctx.default_anchor]
217 assert len(early_downbeats) < 8
218 return ctx
219
220
221def test_trim_closing_ladder_uses_the_tier_at_its_own_anchor() -> None:
222 """A grid that only becomes blendable at the audible end still earns the long rungs."""
223 ctx = _late_blendable_only_ctx()
224 assert ctx.tier is TransitionTier.QUICK_FADE # the early window has too few downbeats
225 specs = list(TrimClosingAnchorGenerator().generate(ctx))
226 assert specs
227 assert max(spec.bars for spec in specs) == 8
228 assert all(spec.tier is not TransitionTier.QUICK_FADE for spec in specs)
229 assert all(spec.ideal_bars == 8 for spec in specs)
230
231
232def _ctx_with_late_natural_entry() -> TransitionContext:
233 """Build a context where B grooves late: its natural entry lands deep in the 45s head."""
234 aa_out = _analysis(bpm=124.0, duration=200.0)
235 aa_in = _analysis(bpm=124.0, duration=45.0)
236 aa_in.rms_energy = [0.05] * 720 + [0.9] * 1080
237 ctx = build_transition_context(aa_out, aa_in, 45.0, logging.getLogger("test"))
238 assert ctx.natural_entry > 10.0
239 return ctx
240
241
242def _ambient_unblendable_ctx() -> tuple[AudioAnalysisData, AudioAnalysisData]:
243 """
244 Build an outgoing/incoming pair whose grid is unusable but both decks are ambient.
245
246 The outgoing downbeat grid dies at 10s (rubato tail, like the 3.2 sparse-tail
247 fixture); its energy stays quiet-but-audible out to ~43s before real silence,
248 stranding a large gap past the energy anchor. Both decks carry a validated
249 all-zero vocal timeline, so both duties read 0.0 (ambient).
250 """
251 grid_beats = [i * 60 / 128 for i in range(int(10.0 * 128 / 60) + 1)]
252 rms_energy = [0.9] * 1260 + [0.25] * 450 + [0.0] * 90
253 aa_out = AudioAnalysisData(
254 duration=45.0,
255 bpm=128.0,
256 beats=grid_beats,
257 downbeats=grid_beats[::4],
258 beats_per_bar=4,
259 rms_energy=rms_energy,
260 key="C",
261 mode="minor",
262 extra_data={"vocal_activity": [0.0] * 1800},
263 )
264 full_beats = [i * 60 / 128 for i in range(int(45 * 128 / 60))]
265 aa_in = AudioAnalysisData(
266 duration=45.0,
267 bpm=128.0,
268 beats=full_beats,
269 downbeats=full_beats[::4],
270 beats_per_bar=4,
271 rms_energy=[0.8] * 1800,
272 key="C",
273 mode="minor",
274 extra_data={"vocal_activity": [0.0] * 1800},
275 )
276 return aa_out, aa_in
277
278
279def _vocal_unblendable_ctx() -> TransitionContext:
280 """Build the same ambient pair, but with the incoming deck fully sung: never qualifies."""
281 aa_out, aa_in = _ambient_unblendable_ctx()
282 aa_in.extra_data = {"vocal_activity": [0.9] * 1800}
283 return build_transition_context(aa_out, aa_in, 45.0, logging.getLogger("test"))
284
285
286def _clean_full_blend_ctx() -> TransitionContext:
287 """Build a context with a full, evenly-spaced grid: earns the ordinary full-blend tier."""
288 aa_out = _analysis(bpm=124.0, duration=200.0)
289 aa_in = _analysis(bpm=124.0, duration=200.0)
290 return build_transition_context(aa_out, aa_in, 45.0, logging.getLogger("test"))
291
292
293def test_lazy_overlay_wins_for_both_ambient_unblendable_pair() -> None:
294 """Quiet-tail + ambient incoming: the long overlay replaces the 2-bar rescue."""
295 ctx_out_aa, ctx_in_aa = _ambient_unblendable_ctx()
296 plan = SmartCrossFadePlanner(logging.getLogger("test")).plan(ctx_out_aa, ctx_in_aa, 45.0)
297 assert plan.metrics.strategy is TransitionStrategy.LAZY_OVERLAY
298 assert plan.crossfade_duration >= 12.0
299 assert plan.fadein_trim_start is None # B keeps its intro
300
301
302def test_lazy_overlay_not_emitted_for_vocal_material() -> None:
303 """A singing deck never gets the unphrased long overlay."""
304 specs = list(LazyOverlayGenerator().generate(_vocal_unblendable_ctx()))
305 assert specs == []
306
307
308def test_lazy_overlay_not_emitted_when_grid_blendable() -> None:
309 """A clean, blendable pair never falls back to the unphrased overlay."""
310 specs = list(LazyOverlayGenerator().generate(_clean_full_blend_ctx()))
311 assert specs == []
312
313
314def test_lazy_overlay_beats_trim_closing_on_a_qualifying_pair() -> None:
315 """
316 The overlay must win the tie against trim-closing's equally-cheap short rungs.
317
318 Both generators anchor near the audible end with ~zero trim on this
319 context, so this exercises the actual tie-break (generator order), not
320 just an absence of competition.
321 """
322 aa_out, aa_in = _ambient_unblendable_ctx()
323 ctx = build_transition_context(aa_out, aa_in, 45.0, logging.getLogger("test"))
324 # trim-closing must actually compete here, or this proves nothing
325 assert list(TrimClosingAnchorGenerator().generate(ctx))
326
327 plan = SmartCrossFadePlanner(logging.getLogger("test")).plan(aa_out, aa_in, 45.0)
328 assert plan.metrics.strategy is TransitionStrategy.LAZY_OVERLAY
329
330
331def _lazy_gate_outgoing() -> AudioAnalysisData:
332 """
333 Outgoing analysis shared by the lazy-gate vocal-window fixtures.
334
335 Same shape as ``_late_blendable_only_ctx``'s outgoing deck: a full 4/4
336 grid at 124 BPM, but the early mix-out anchor leaves fewer than 8
337 downbeats before it, so the pair reaches QUICK_FADE and the lazy gate.
338 The vocal timeline is all-zero, so the outgoing side never contributes duty.
339 """
340 beats = [i * 60 / 124 for i in range(int(240 * 124 / 60))]
341 downbeats = beats[::4]
342 rms_energy = [0.9] * 1550 + [0.25] * (1730 - 1550) + [0.0] * (1800 - 1730)
343 return AudioAnalysisData(
344 duration=240.0,
345 bpm=124.0,
346 beats=beats,
347 downbeats=downbeats,
348 beats_per_bar=4,
349 rms_energy=rms_energy,
350 key="A",
351 mode="minor",
352 extra_data={"vocal_activity": [0.0] * 1800},
353 )
354
355
356def _lazy_gate_incoming(vocal_run: tuple[float, float]) -> AudioAnalysisData:
357 """Incoming analysis for the lazy-gate fixtures: a 45s head with vocal only over ``vocal_run``."""
358 beats = [i * 60 / 124 for i in range(int(45 * 124 / 60))]
359 vocal_activity = [0.0] * 1800
360 frame_duration = 45.0 / 1800
361 start_bin = int(vocal_run[0] / frame_duration)
362 end_bin = int(vocal_run[1] / frame_duration)
363 for i in range(start_bin, end_bin):
364 vocal_activity[i] = 0.95
365 return AudioAnalysisData(
366 duration=45.0,
367 bpm=124.0,
368 beats=beats,
369 downbeats=beats[::4],
370 beats_per_bar=4,
371 rms_energy=[0.8] * 1800,
372 key="A",
373 mode="minor",
374 extra_data={"vocal_activity": vocal_activity},
375 )
376
377
378def _front_loaded_vocal_ctx() -> TransitionContext:
379 """
380 Build a lazy-gate context where B's vocal sits inside the overlay's first 16s.
381
382 B's vocal run covers media 4.0-7.2s: ~3.2s of a 16s overlay (~0.20 duty)
383 but only ~0.07 over the full 45s head, so the whole-window gate would
384 pass it while the windowed gate correctly blocks it.
385 """
386 ctx = build_transition_context(
387 _lazy_gate_outgoing(), _lazy_gate_incoming((4.0, 7.2)), 45.0, logging.getLogger("test")
388 )
389 assert ctx.tier is TransitionTier.QUICK_FADE
390 whole = _vocal_duties(ctx)
391 assert whole is not None
392 assert whole[1] <= 0.10
393 windowed = _window_duties(ctx, _LAZY_OVERLAY_SECONDS)
394 assert windowed is not None
395 assert windowed[1] > 0.10
396 return ctx
397
398
399def _late_vocal_ctx() -> TransitionContext:
400 """
401 Build a lazy-gate context where B's vocal sits entirely outside the overlay's first 16s.
402
403 B's vocal run covers media 20.0-30.0s: 0.0 duty inside a 16s overlay but
404 ~0.22 over the full 45s head, so the whole-window gate wrongly blocks it
405 while the windowed gate correctly allows it.
406 """
407 ctx = build_transition_context(
408 _lazy_gate_outgoing(), _lazy_gate_incoming((20.0, 30.0)), 45.0, logging.getLogger("test")
409 )
410 assert ctx.tier is TransitionTier.QUICK_FADE
411 whole = _vocal_duties(ctx)
412 assert whole is not None
413 assert whole[1] > 0.10
414 windowed = _window_duties(ctx, _LAZY_OVERLAY_SECONDS)
415 assert windowed is not None
416 assert windowed[1] <= 0.10
417 return ctx
418
419
420def test_lazy_overlay_denied_when_vocals_sit_inside_the_overlay() -> None:
421 """Vocals concentrated in B's first 16s block the overlay even when its 45s duty is low."""
422 ctx = _front_loaded_vocal_ctx()
423 assert list(LazyOverlayGenerator().generate(ctx)) == []
424
425
426def test_lazy_overlay_allowed_when_vocals_sit_outside_the_overlay() -> None:
427 """Vocals late in B's head leave the overlay window ambient, so the overlay still fires."""
428 ctx = _late_vocal_ctx()
429 specs = list(LazyOverlayGenerator().generate(ctx))
430 assert len(specs) == 1
431 assert specs[0].strategy is TransitionStrategy.LAZY_OVERLAY
432
433
434def test_instrumental_blend_gate_unchanged_by_window_duties() -> None:
435 """The both-instrumental 16-bar gate keeps reading whole-window duty."""
436 ctx = _front_loaded_vocal_ctx()
437 assert _vocal_duties(ctx) is not None
438 # the 16-bar gate's own verdict must not move when the lazy gate narrows its window
439 assert earns_instrumental_blend(ctx) is False
440
441
442def test_short_rungs_offer_intro_keeping_entry() -> None:
443 """At 1-2 bars an entry at 0.0 (keep B's intro) is offered alongside the natural entry."""
444 ctx = _ctx_with_late_natural_entry()
445 options = _entry_options(ctx, 2)
446 assert 0.0 in options
447 assert ctx.natural_entry in options
448 # 0.0 must precede the natural entry: the selector ties break to the
449 # earlier candidate, so order decides which one a tie actually prefers
450 assert options.index(0.0) < options.index(ctx.natural_entry)
451 assert 0.0 not in _entry_options(ctx, 8)
452