/
/
1"""
2Smart Fades - candidate value objects and the timed-candidate factory.
3
4A ``CandidateSpec`` is a generator's declared intent (which tier rung, which
5anchor/entry, which generator produced it) before any plan exists; a
6``Candidate`` is that spec paired with its timed ``TransitionPlan`` and
7computed ``PlanMetrics``, ready for policies to score. The factory builds
8TIMED candidates only - anchor, overlap timing, tempo ramp, trims and metrics.
9EQ is deliberately absent: scoring never needs it, so the winner-only
10``PlanAssembler`` applies it after selection.
11
12Every ``build()`` derives its anchored tail fresh from the immutable
13``TransitionContext``, which is what replaces the old planner's
14restore-pristine/re-anchor scratchpad: two builds of the same spec are
15guaranteed to yield identical candidates.
16"""
17
18from __future__ import annotations
19
20from abc import ABC, abstractmethod
21from dataclasses import dataclass, replace
22from typing import TYPE_CHECKING
23
24from music_assistant.constants import VERBOSE_LOG_LEVEL
25from music_assistant.controllers.streams.smart_fades.helpers import (
26 MIN_EFFECTIVE_FADE_BUFFER,
27 SMART_CROSSFADE_DURATION,
28 compute_gradual_tempo_steps,
29 generate_synthetic_timestamps,
30)
31from music_assistant.controllers.streams.smart_fades.models import (
32 FadeOutTrim,
33 PlanMetrics,
34 TempoPlan,
35 TransitionPlan,
36 TransitionStrategy,
37 TransitionTier,
38)
39from music_assistant.controllers.streams.smart_fades.structure import point_in_mask
40from music_assistant.controllers.streams.smart_fades.vocal import (
41 collision_metrics,
42 mask_saturated,
43 merge_windows,
44)
45
46from .context import TIME_STRETCH_BPM_PERCENTAGE_THRESHOLD, choose_tier
47
48if TYPE_CHECKING:
49 import logging
50 from collections.abc import Iterable
51
52 import numpy as np
53 import numpy.typing as npt
54
55 from music_assistant.controllers.streams.smart_fades.models import BandProfile
56
57 from .context import TransitionContext
58
59# Overlap length per tier, in bars of the outgoing grid (research: real DJ
60# transitions cluster at 32 beats = 8 bars; the doubled 16-bar blend is
61# earned only when FireRed shows both decks near-instrumental, where the
62# long exposure carries no vocal-collision risk)
63_FULL_BLEND_BARS: int = 8
64_INSTRUMENTAL_BLEND_BARS: int = 16
65# vocal duty (unpadded mask coverage) at or under this on BOTH decks
66# qualifies as near-instrumental
67_INSTRUMENTAL_DUTY_MAX: float = 0.05
68_TEMPO_BLEND_BARS: int = 8
69# QUICK_FADE bars by BPM incompatibility: (max diff %, bars); beyond -> 1 bar
70_QUICK_FADE_LADDER: tuple[tuple[float, int], ...] = ((12.0, 4), (20.0, 2))
71# phrase-aligned rung set every ladder walks, largest first
72RUNG_LADDER: tuple[int, ...] = (16, 8, 4, 2, 1)
73
74# Deep-trim guard: a bar is protected when its low band is silent but its
75# voice/melody bands are active (cutting there beheads a sung intro)
76_TRIM_GUARD_LOW_FLOOR: float = 0.25
77_TRIM_GUARD_VOICE_FLOOR: float = 0.4
78
79# Trim-closing anchors engage only when the audible end sits this much past
80# the energy anchor; below it the default anchor's trim is already acceptable
81_TRIM_CLOSING_MIN_GAP_S: float = 8.0
82
83# Lazy-overlay length: a long unphrased equal-power blend, not a rung on any ladder
84_LAZY_OVERLAY_SECONDS: float = 16.0
85# both decks at or under this in-window vocal duty qualify as ambient
86_LAZY_DUTY_MAX: float = 0.10
87
88
89@dataclass(frozen=True, slots=True)
90class CandidateSpec:
91 """A candidate's declared shape: tier rung, anchor/entry choice, and generator provenance."""
92
93 tier: TransitionTier
94 bars: int
95 # buffer-local; None = pristine audible end
96 anchor_s: float | None
97 # None = natural entry
98 entry_s: float | None
99 strategy: TransitionStrategy = TransitionStrategy.ENERGY_ALIGNED
100 # generator name, for scoreboard + tie-break
101 source: str = ""
102 # the tier ladder's top rung; 0 = same as bars
103 ideal_bars: int = 0
104
105
106@dataclass(frozen=True, slots=True)
107class Candidate:
108 """One fully-built candidate: its spec, timed plan, and computed metrics."""
109
110 spec: CandidateSpec
111 # timed plan, eq_plan neutral until assembly
112 plan: TransitionPlan
113 metrics: PlanMetrics
114 # the tier ladder's top rung for this context
115 ideal_bars: int
116
117
118def earns_instrumental_blend(ctx: TransitionContext) -> bool:
119 """Whether verified near-instrumental decks earn the doubled full-blend overlap."""
120 duties = _vocal_duties(ctx)
121 if duties is None:
122 return False
123 out_duty, in_duty = duties
124 return out_duty <= _INSTRUMENTAL_DUTY_MAX and in_duty <= _INSTRUMENTAL_DUTY_MAX
125
126
127def bars_ladder(ctx: TransitionContext, tier: TransitionTier) -> list[int]:
128 """Candidate bar counts to try for a tier, largest first (shorter rungs fit smaller buffers)."""
129 if tier is TransitionTier.QUICK_FADE:
130 # a mismatched meter has no shared bar grid to blend across; cap short
131 # regardless of how close the tempos happen to be
132 ladder = ((0.0, 2),) if ctx.cross_meter else _QUICK_FADE_LADDER
133 ideal = next((bars for limit, bars in ladder if ctx.bpm_diff_percent <= limit), 1)
134 elif tier is TransitionTier.TEMPO_BLEND:
135 ideal = _TEMPO_BLEND_BARS
136 elif earns_instrumental_blend(ctx):
137 ideal = _INSTRUMENTAL_BLEND_BARS
138 else:
139 ideal = _FULL_BLEND_BARS
140 return [bars for bars in RUNG_LADDER if bars <= ideal]
141
142
143class CandidateGenerator(ABC):
144 """One source of candidate specs; generators only emit, the factory validates feasibility."""
145
146 name: str
147
148 @abstractmethod
149 def generate(self, ctx: TransitionContext) -> Iterable[CandidateSpec]:
150 """Emit this generator's candidate specs for one transition, best first."""
151
152
153class EnergyLadderGenerator(CandidateGenerator):
154 """Emits the tier's bar-count ladder at each viable anchor, across entry options."""
155
156 name = "energy-ladder"
157
158 def generate(self, ctx: TransitionContext) -> Iterable[CandidateSpec]:
159 """Emit one spec per (rung, entry option) at the default, full-band and pinned anchors."""
160 ladder = bars_ladder(ctx, ctx.tier)
161 ideal = ladder[0]
162 # the default anchor is already kick-folded; a pure full-band variant
163 # would let a longer blend win past the kick die-out, defeating the
164 # researched kick handover, so it is deliberately not emitted (trim-closing
165 # may still anchor later when >=8s of audible tail would otherwise be stranded)
166 anchors: list[float | None] = [None]
167 a_pin = _fade_onset_pin(ctx)
168 if a_pin < ctx.default_anchor:
169 anchors.append(a_pin)
170 for anchor in anchors:
171 for bars in ladder:
172 # the default anchor keeps the factory-chosen entry (grid
173 # alignment + rolling intro), exactly like the old energy
174 # candidate; explicit entry options are pin-scoped, like the
175 # old remediation rungs they came from
176 entries = [None] if anchor is None else _entry_options(ctx, bars)
177 for entry in entries:
178 yield CandidateSpec(
179 tier=ctx.tier,
180 bars=bars,
181 anchor_s=anchor,
182 entry_s=entry,
183 source=self.name,
184 ideal_bars=ideal,
185 )
186
187
188class CodaAnchorGenerator(CandidateGenerator):
189 """Emits candidates anchored inside the outgoing track's validated coda/outro zone."""
190
191 name = "coda-anchor"
192
193 def generate(self, ctx: TransitionContext) -> Iterable[CandidateSpec]:
194 """Emit the zone's fitting rung(s) at the coda anchor, or nothing without a valid zone."""
195 # coda shifting is a vocal-collision remediation: without a vocal
196 # timeline the zone was validated against an empty mask and the old
197 # planner never coda-shifted, so stay inactive on the energy-only path
198 if ctx.vocal_out_placement is None:
199 return
200 if ctx.coda_zone is None or ctx.outgoing_profile is None:
201 return
202 zone = ctx.coda_zone
203 bar_starts = ctx.outgoing_profile.bar_starts
204 in_zone = bar_starts[(bar_starts >= zone.start_s) & (bar_starts <= zone.end_s)]
205 if not len(in_zone):
206 return
207 anchor = float(in_zone[-1]) - ctx.buffer_offset
208 if anchor < MIN_EFFECTIVE_FADE_BUFFER:
209 return
210 bar_a = ctx.outgoing.beats_per_bar * 60.0 / ctx.outgoing.bpm
211 zone_bars = int((zone.end_s - zone.start_s) / bar_a)
212 top_rung = next((n for n in (16, 8, 4, 2) if n <= zone_bars), None)
213 if top_rung is None:
214 return
215 for bars in dict.fromkeys((top_rung, 2)):
216 for entry in _entry_options(ctx, bars):
217 yield CandidateSpec(
218 tier=ctx.tier,
219 bars=bars,
220 anchor_s=anchor,
221 entry_s=entry,
222 source=self.name,
223 ideal_bars=top_rung,
224 )
225
226
227class ProtectiveAnchorGenerator(CandidateGenerator):
228 """Emits the tier's ladder anchored to keep A's last outgoing vocal phrase intact."""
229
230 name = "protective-anchor"
231
232 def generate(self, ctx: TransitionContext) -> Iterable[CandidateSpec]:
233 """Emit every ladder rung at the nearest anchor that doesn't truncate A's last phrase."""
234 if ctx.vocal_out_placement is None or not ctx.vocal_out_placement.windows:
235 return
236 target = _outgoing_vocal_end(ctx)
237 anchor = _nearest_protective_anchor(ctx, target)
238 ladder = bars_ladder(ctx, ctx.tier)
239 ideal = ladder[0]
240 bar_seconds = ctx.outgoing.beats_per_bar * 60.0 / ctx.outgoing.bpm
241 for bars in ladder:
242 # the old protection had two re-anchor triggers: cover the last
243 # vocal phrase, and close a short fade's audible-trim gap; emit
244 # both anchors so each stays reachable as a candidate
245 anchors = [anchor]
246 trim_target = ctx.audio_end - bars * bar_seconds
247 if trim_target > anchor:
248 trim_anchor = _nearest_protective_anchor(ctx, trim_target, prefer_earliest=False)
249 if trim_anchor not in anchors:
250 anchors.append(trim_anchor)
251 # the factory-chosen (beat-aligned) entry leads, as the old
252 # protection rebuild pinned it; explicit groove/natural options
253 # follow so a remediated entry stays reachable too
254 for rung_anchor in anchors:
255 for entry in [None, *_entry_options(ctx, bars)]:
256 yield CandidateSpec(
257 tier=ctx.tier,
258 bars=bars,
259 anchor_s=rung_anchor,
260 entry_s=entry,
261 source=self.name,
262 ideal_bars=ideal,
263 )
264
265
266class VocalOnsetEntryGenerator(CandidateGenerator):
267 """Emits an entry that lands B's first vocal onset exactly at the overlap end."""
268
269 name = "vocal-onset-entry"
270
271 def generate(self, ctx: TransitionContext) -> Iterable[CandidateSpec]:
272 """Emit the vocal-onset-aligned entry at the tier's ideal rung, or nothing when illegal."""
273 if ctx.vocal_in_placement is None or not ctx.vocal_in_placement.windows:
274 return
275 if mask_saturated(ctx.vocal_in_placement, float(SMART_CROSSFADE_DURATION)):
276 return
277 ideal = bars_ladder(ctx, ctx.tier)[0]
278 bar_b = ctx.incoming.beats_per_bar * 60.0 / ctx.incoming.bpm
279 entry = ctx.vocal_in_placement.windows[0][0] - ideal * bar_b
280 if entry < 0.0 or point_in_mask(ctx.vocal_in_placement, entry):
281 return
282 a_pin = _fade_onset_pin(ctx)
283 anchors: list[float | None] = [a_pin if a_pin < ctx.default_anchor else None]
284 # the old planner's protection rebuild preserved a remediated entry at
285 # the protective anchor; emitting the combination keeps that reachable
286 if ctx.vocal_out_placement is not None and ctx.vocal_out_placement.windows:
287 protective = _nearest_protective_anchor(
288 ctx, min(ctx.vocal_out_placement.last_end(), ctx.audio_end)
289 )
290 if protective not in anchors and protective != ctx.default_anchor:
291 anchors.append(protective)
292 for anchor in anchors:
293 yield CandidateSpec(
294 tier=ctx.tier,
295 bars=ideal,
296 anchor_s=anchor,
297 entry_s=entry,
298 source=self.name,
299 ideal_bars=ideal,
300 )
301
302
303class RescueAnchorGenerator(CandidateGenerator):
304 """Emits a modest, late-anchored rung as a last resort before the emergency handoff."""
305
306 name = "rescue-anchor"
307
308 def generate(self, ctx: TransitionContext) -> Iterable[CandidateSpec]:
309 """Emit 1-2 bar rungs anchored as late as the tail allows, never past A's own vocal end."""
310 full_ladder = bars_ladder(ctx, ctx.tier)
311 bar_seconds = ctx.outgoing.beats_per_bar * 60.0 / ctx.outgoing.bpm
312 last_vocal_end = _outgoing_vocal_end(ctx)
313 for bars in [rung for rung in full_ladder if rung <= 2]:
314 target = min(ctx.audio_end, max(ctx.audio_end - bars * bar_seconds, last_vocal_end))
315 anchor = _nearest_protective_anchor(ctx, target, prefer_earliest=False)
316 for entry in [None, *_entry_options(ctx, bars)]:
317 yield CandidateSpec(
318 tier=ctx.tier,
319 bars=bars,
320 anchor_s=anchor,
321 entry_s=entry,
322 source=self.name,
323 ideal_bars=full_ladder[0],
324 )
325
326
327class TrimClosingAnchorGenerator(CandidateGenerator):
328 """Emits the tier's ladder at that anchor, at the audible end, when a large tail is stranded."""
329
330 name = "trim-closing-anchor"
331
332 def generate(self, ctx: TransitionContext) -> Iterable[CandidateSpec]:
333 """Emit every ladder rung at the audible end, or nothing when the trim gap is small."""
334 anchor = ctx.audio_end
335 if anchor - ctx.default_anchor < _TRIM_CLOSING_MIN_GAP_S:
336 return
337 # ctx.tier is decided at the early anchor; the grid can be blendable at the audible end
338 _, tier = choose_tier(ctx.outgoing, ctx.incoming, anchor)
339 ladder = bars_ladder(ctx, tier)
340 for bars in ladder:
341 yield CandidateSpec(
342 tier=tier,
343 bars=bars,
344 anchor_s=anchor,
345 entry_s=None,
346 source=self.name,
347 ideal_bars=ladder[0],
348 )
349
350
351class LazyOverlayGenerator(CandidateGenerator):
352 """Emits one long unphrased overlay when the grid is unusable but both decks are ambient."""
353
354 name = "lazy-overlay"
355
356 def generate(self, ctx: TransitionContext) -> Iterable[CandidateSpec]:
357 """Emit the overlay spec, or nothing when the pair doesn't qualify."""
358 if ctx.tier is not TransitionTier.QUICK_FADE or ctx.cross_meter:
359 return
360 if ctx.bpm_diff_percent > TIME_STRETCH_BPM_PERCENTAGE_THRESHOLD:
361 return
362 duties = _window_duties(ctx, _LAZY_OVERLAY_SECONDS)
363 if duties is None or duties[0] > _LAZY_DUTY_MAX or duties[1] > _LAZY_DUTY_MAX:
364 return
365 yield CandidateSpec(
366 tier=ctx.tier,
367 bars=1,
368 anchor_s=ctx.audio_end,
369 entry_s=None,
370 strategy=TransitionStrategy.LAZY_OVERLAY,
371 source=self.name,
372 ideal_bars=1,
373 )
374
375
376def default_generators() -> tuple[CandidateGenerator, ...]:
377 """Return the standard generator set, in preference order (best first)."""
378 return (
379 EnergyLadderGenerator(),
380 CodaAnchorGenerator(),
381 ProtectiveAnchorGenerator(),
382 VocalOnsetEntryGenerator(),
383 LazyOverlayGenerator(),
384 TrimClosingAnchorGenerator(),
385 )
386
387
388class CandidateFactory:
389 """Builds timed candidates from specs, purely over the transition context."""
390
391 def __init__(self, ctx: TransitionContext, logger: logging.Logger) -> None:
392 """Initialize the factory for one transition."""
393 self._ctx = ctx
394 self._logger = logger
395
396 def build(self, spec: CandidateSpec) -> Candidate | None:
397 """
398 Build one complete timed candidate for a spec, or ``None`` when it is infeasible.
399
400 Every timing, tempo and trim decision is derived fresh from the
401 context and the spec's anchor - a candidate never inherits state from
402 a previously built one. Infeasible means the spec's bar count needs
403 more room than the incoming buffer has, or its entry leaves no legal
404 alignment; a 1-bar spec never fails this way, matching the plan floor.
405 The returned candidate's spec reflects what was actually built: a
406 re-anchored tail can downgrade the tier and cap the bar count.
407 """
408 if spec.strategy is TransitionStrategy.LAZY_OVERLAY:
409 return self._build_lazy_overlay(spec)
410 tail = self._anchored_tail(spec.anchor_s)
411 # a re-anchored tail can downgrade the tier (shorter/irregular grid); the
412 # requested bar count still reflects the old tier, so cap it at the new
413 # tier's largest rung or a long overlap ships without its tempo ramp
414 _, tier = choose_tier(self._ctx.outgoing, self._ctx.incoming, tail.effective_end)
415 bars_cap = bars_ladder(self._ctx, tier)[0]
416 bars = min(spec.bars, bars_cap)
417
418 fadein_start_pos = (
419 spec.entry_s if spec.entry_s is not None else self._choose_fadein_entry(tail, bars)
420 )
421 if bars > 1 and fadein_start_pos is None:
422 self._logger.log(
423 VERBOSE_LOG_LEVEL,
424 "dropping spec source=%s tier=%s bars=%d anchor=%s: no beat-aligned incoming entry",
425 spec.source,
426 tier,
427 bars,
428 spec.anchor_s,
429 )
430 return None
431 crossfade_duration = self._calculate_crossfade_duration(tail, bars)
432
433 tempo_plan = self._choose_tempo_ramp(tier, tail, crossfade_duration)
434 crossfade_duration, fadein_trim_start = self._lock_in_timing(
435 tail, crossfade_duration, fadein_start_pos, tempo_plan
436 )
437 if bars > 1 and fadein_start_pos is not None and fadein_trim_start is None:
438 self._logger.log(
439 VERBOSE_LOG_LEVEL,
440 "dropping spec source=%s tier=%s bars=%d anchor=%s: "
441 "no legal timing lock for the pinned entry",
442 spec.source,
443 tier,
444 bars,
445 spec.anchor_s,
446 )
447 return None
448 # Rolling-intro alignment: on a full blend with no pinned entry, deepen B's
449 # trim so its groove entry lands at the overlap END (B's intro runs under A,
450 # its drop hits where A's music dies). A sung run that no legal cut clears
451 # is infeasible so the caller's ladder drops to a shorter overlap instead —
452 # except at the 1-bar floor, which must always yield a candidate: there the
453 # un-deepened trim ships as-is.
454 if tier is TransitionTier.FULL_BLEND and spec.entry_s is None:
455 feasible, aligned = self._align_rolling_intro(crossfade_duration, fadein_trim_start)
456 if not feasible:
457 if bars > 1:
458 self._logger.log(
459 VERBOSE_LOG_LEVEL,
460 "dropping spec source=%s tier=%s bars=%d anchor=%s: "
461 "rolling intro: no legal cut clears the sung run",
462 spec.source,
463 tier,
464 bars,
465 spec.anchor_s,
466 )
467 return None
468 else:
469 fadein_trim_start = aligned
470
471 plan = TransitionPlan(
472 tier=tier,
473 fade_out_window=tail.effective_end,
474 crossfade_duration=crossfade_duration,
475 tempo_plan=tempo_plan,
476 fadeout_trim=tail.fadeout_trim,
477 fadein_trim_start=fadein_trim_start,
478 )
479 built_spec = replace(spec, tier=tier, bars=bars)
480 return Candidate(
481 spec=built_spec,
482 plan=plan,
483 metrics=self._score(built_spec, plan),
484 ideal_bars=spec.ideal_bars or spec.bars,
485 )
486
487 def score(self, spec: CandidateSpec, plan: TransitionPlan) -> PlanMetrics:
488 """Score an arbitrary (spec, plan) pair against this context, for a plan edited post-build."""
489 return self._score(spec, plan)
490
491 @property
492 def _bpm_ratio(self) -> float:
493 """Tempo ratio between the incoming and outgoing track."""
494 return self._ctx.incoming.bpm / self._ctx.outgoing.bpm
495
496 def _anchored_tail(self, anchor_s: float | None) -> _AnchoredTail:
497 """Derive the tail state for an anchor, never later than the RMS-audible boundary."""
498 import numpy as np # noqa: PLC0415
499
500 ctx = self._ctx
501 anchor = anchor_s if anchor_s is not None else ctx.default_anchor
502 effective_end = min(anchor, ctx.audio_end)
503 # same sub-half-second slack rule as the tail cue: the rendered stream
504 # still ends at the buffer end, so the anchor must follow it
505 fadeout_trim: FadeOutTrim | None
506 if effective_end >= ctx.buffer_duration - 0.5:
507 effective_end = ctx.buffer_duration
508 fadeout_trim = None
509 else:
510 fadeout_trim = FadeOutTrim(
511 end_pos=effective_end,
512 trimmed_seconds=ctx.buffer_duration - effective_end,
513 )
514 protective = np.asarray(ctx.protective_downbeats, dtype=np.float32)
515 return _AnchoredTail(
516 effective_end=effective_end,
517 fadeout_trim=fadeout_trim,
518 beats=ctx.outgoing.beats[ctx.outgoing.beats <= effective_end],
519 downbeats=ctx.outgoing.downbeats[ctx.outgoing.downbeats <= effective_end],
520 # protective downbeats reach all the way to audio_end, so they cover
521 # any position an anchor could have chosen
522 extrapolated_downbeats=protective[protective <= effective_end],
523 )
524
525 def _choose_fadein_entry(self, tail: _AnchoredTail, crossfade_bars: int) -> float | None:
526 """Choose where the incoming track enters, aligned to its beat grid."""
527
528 def calculate_beat_positions(
529 fade_out_beats: npt.NDArray[np.float32],
530 fade_in_beats: npt.NDArray[np.float32],
531 num_beats: int,
532 ) -> float | None:
533 """Calculate start positions from beat arrays."""
534 if len(fade_out_beats) < num_beats or len(fade_in_beats) < num_beats:
535 return None
536
537 fade_in_slice = fade_in_beats[:num_beats]
538 return float(fade_in_slice[0])
539
540 # Try downbeats first for most musical timing
541 downbeat_positions = calculate_beat_positions(
542 tail.extrapolated_downbeats, self._ctx.incoming.downbeats, crossfade_bars
543 )
544 if downbeat_positions is not None:
545 return downbeat_positions
546
547 # Try regular beats if downbeats insufficient
548 required_beats = crossfade_bars * self._ctx.incoming.beats_per_bar
549 beat_positions = calculate_beat_positions(
550 tail.beats, self._ctx.incoming.beats, required_beats
551 )
552 if beat_positions is not None:
553 return beat_positions
554
555 # Fallback: No beat alignment possible
556 self._logger.log(VERBOSE_LOG_LEVEL, "No beat alignment possible (insufficient beats)")
557 return None
558
559 def _calculate_crossfade_duration(self, tail: _AnchoredTail, crossfade_bars: int) -> float:
560 """Calculate the crossfade duration for a bar count, capped to the audible tail."""
561 downbeats = tail.downbeats
562 bar_seconds = self._ctx.outgoing.beats_per_bar * 60.0 / self._ctx.outgoing.bpm
563 # the downbeat span assumes the anchor sits (near) the last downbeat;
564 # when the grid dies early (unsnapped anchor), the anchor gap would be
565 # added to EVERY rung — a "1-bar" quick fade could span the whole gap
566 if (
567 len(downbeats) > crossfade_bars
568 and tail.effective_end - float(downbeats[-1]) < bar_seconds
569 ):
570 # the real span between downbeats honors the track's own tempo/rubato
571 # more precisely than a constant-BPM estimate
572 musical_duration = float(
573 tail.effective_end - downbeats[len(downbeats) - 1 - crossfade_bars]
574 )
575 else:
576 seconds_per_beat = 60.0 / self._ctx.incoming.bpm
577 musical_duration = crossfade_bars * self._ctx.incoming.beats_per_bar * seconds_per_beat
578
579 # Cap at the audible fade-out room so crossfade_start never goes negative
580 # downstream (effective_end <= SMART_CROSSFADE_DURATION always)
581 actual_duration = min(musical_duration, tail.effective_end)
582
583 if musical_duration > actual_duration:
584 self._logger.log(
585 VERBOSE_LOG_LEVEL,
586 "Constraining crossfade duration from %.1fs to %.1fs (audible tail limit)",
587 musical_duration,
588 actual_duration,
589 )
590
591 return actual_duration
592
593 def _choose_tempo_ramp(
594 self, tier: TransitionTier, tail: _AnchoredTail, crossfade_duration: float
595 ) -> TempoPlan:
596 """Choose the gradual tempo ramp that beatmatches the outgoing track, if any."""
597 if tier is TransitionTier.QUICK_FADE:
598 return TempoPlan()
599 if not 0.1 < self._ctx.bpm_diff_percent <= TIME_STRETCH_BPM_PERCENTAGE_THRESHOLD:
600 return TempoPlan()
601 return TempoPlan(steps=self._compute_tempo_steps(tail, crossfade_duration))
602
603 def _compute_tempo_steps(
604 self, tail: _AnchoredTail, crossfade_duration: float
605 ) -> list[tuple[float, float]]:
606 """Compute the gradual tempo ramp in the 10s window before the crossfade."""
607 stretch_duration = 10.0
608 crossfade_start = tail.effective_end - crossfade_duration
609 # A crossfade consuming the whole audible tail leaves no room for a
610 # pre-fade tempo ramp
611 if crossfade_start <= 0:
612 return []
613 stretch_start = max(0.0, crossfade_start - stretch_duration)
614 stretch_end = crossfade_start
615
616 # Collect timing points within the stretch window
617 beats = tail.beats
618 beat_mask = (beats >= stretch_start) & (beats <= stretch_end)
619 db_mask = (tail.extrapolated_downbeats >= stretch_start) & (
620 tail.extrapolated_downbeats <= stretch_end
621 )
622 window_beats = beats[beat_mask] - stretch_start
623 window_downbeats = tail.extrapolated_downbeats[db_mask] - stretch_start
624
625 # >3% BPM diff: beat-level stepping (more steps = smoother)
626 # <=3%: downbeat-level stepping, fall back to beats if too few
627 if self._ctx.bpm_diff_percent > 3.0:
628 stretch_timestamps = window_beats
629 elif len(window_downbeats) >= 2:
630 stretch_timestamps = window_downbeats
631 else:
632 stretch_timestamps = window_beats
633
634 # Fall back to synthetic timestamps when < 2 real timestamps
635 if len(stretch_timestamps) < 2:
636 stretch_timestamps = generate_synthetic_timestamps(
637 stretch_end - stretch_start,
638 self._ctx.outgoing.bpm,
639 beats_per_bar=self._ctx.outgoing.beats_per_bar,
640 )
641
642 tempo_steps = compute_gradual_tempo_steps(
643 start_ratio=1.0,
644 end_ratio=self._bpm_ratio,
645 downbeats=stretch_timestamps,
646 )
647 if not tempo_steps:
648 tempo_steps = [(0.0, self._bpm_ratio)]
649
650 # Shift timestamps back to buffer-relative coordinates for FFmpeg
651 return [(ts + stretch_start, ratio) for ts, ratio in tempo_steps]
652
653 def _lock_in_timing(
654 self,
655 tail: _AnchoredTail,
656 crossfade_duration: float,
657 fadein_start_pos: float | None,
658 tempo_plan: TempoPlan,
659 ) -> tuple[float, float | None]:
660 """
661 Lock the overlap timing: confirm the fade-in entry and snap to downbeats.
662
663 Returns the final crossfade duration (downbeat-snapped and compensated
664 for time-stretch compression) and the fade-in trim position, or ``None``
665 when beat alignment is skipped.
666
667 :param tail: The anchored tail the candidate is built on.
668 :param crossfade_duration: Draft crossfade duration in seconds.
669 :param fadein_start_pos: Chosen entry point in the incoming track, if any.
670 :param tempo_plan: The tempo ramp chosen for this transition.
671 """
672 # Adjust crossfade duration to align with outgoing track's downbeats.
673 # When stretching, only consider downbeats after the stretch window
674 # to ensure the outgoing track has reached the target tempo.
675 crossfade_start = tail.effective_end - crossfade_duration
676 crossfade_duration = self._adjust_crossfade_to_downbeats(
677 tail,
678 crossfade_duration=crossfade_duration,
679 fadein_start_pos=fadein_start_pos,
680 min_downbeat_pos=crossfade_start if tempo_plan else 0.0,
681 render_ratio=self._bpm_ratio if tempo_plan else 1.0,
682 )
683
684 # Compensate crossfade duration for time-stretch compression.
685 # Gate on the tempo plan (not stretch eligibility) so a guard-skipped
686 # stretch doesn't apply a compensation for a stretch that never ran.
687 if tempo_plan:
688 crossfade_duration = crossfade_duration / self._bpm_ratio
689
690 fadein_trim_start: float | None = None
691 if (
692 fadein_start_pos is not None
693 and fadein_start_pos + crossfade_duration <= SMART_CROSSFADE_DURATION
694 ):
695 fadein_trim_start = fadein_start_pos
696 else:
697 self._logger.log(
698 VERBOSE_LOG_LEVEL,
699 "Skipping beat alignment: not enough audio after trim (%s + %.1fs > %.1fs)",
700 fadein_start_pos,
701 crossfade_duration,
702 SMART_CROSSFADE_DURATION,
703 )
704
705 return crossfade_duration, fadein_trim_start
706
707 def _adjust_crossfade_to_downbeats(
708 self,
709 tail: _AnchoredTail,
710 crossfade_duration: float,
711 fadein_start_pos: float | None,
712 min_downbeat_pos: float = 0.0,
713 render_ratio: float = 1.0,
714 ) -> float:
715 """Adjust crossfade duration to align with outgoing track's downbeats."""
716 # If we don't have downbeats or beat alignment is disabled, return original duration
717 if len(tail.extrapolated_downbeats) == 0 or fadein_start_pos is None:
718 return crossfade_duration
719
720 # Calculate where the crossfade would start in the buffer
721 ideal_start_pos = tail.effective_end - crossfade_duration
722
723 self._logger.log(
724 VERBOSE_LOG_LEVEL,
725 "Downbeat adjustment - ideal_start=%.2fs (effective_end=%.1fs - crossfade=%.2fs), "
726 "fadein_start=%.2fs",
727 ideal_start_pos,
728 tail.effective_end,
729 crossfade_duration,
730 fadein_start_pos,
731 )
732
733 # Find the closest downbeats (earlier and later)
734 earlier_downbeat = None
735 later_downbeat = None
736
737 for downbeat in tail.extrapolated_downbeats:
738 if downbeat < min_downbeat_pos:
739 continue
740 if downbeat <= ideal_start_pos:
741 earlier_downbeat = downbeat
742 elif downbeat > ideal_start_pos and later_downbeat is None:
743 later_downbeat = downbeat
744 break
745
746 # Try earlier downbeat first (longer crossfade)
747 if earlier_downbeat is not None:
748 adjusted_duration = float(tail.effective_end - earlier_downbeat)
749 if fadein_start_pos + adjusted_duration / render_ratio <= SMART_CROSSFADE_DURATION:
750 if abs(adjusted_duration - crossfade_duration) > 0.1:
751 self._logger.log(
752 VERBOSE_LOG_LEVEL,
753 "Adjusted crossfade duration from %.2fs to %.2fs to align with "
754 "downbeat at %.2fs (earlier)",
755 crossfade_duration,
756 adjusted_duration,
757 earlier_downbeat,
758 )
759 return adjusted_duration
760
761 # Try later downbeat (shorter crossfade)
762 if later_downbeat is not None:
763 adjusted_duration = float(tail.effective_end - later_downbeat)
764 if fadein_start_pos + adjusted_duration / render_ratio <= SMART_CROSSFADE_DURATION:
765 if abs(adjusted_duration - crossfade_duration) > 0.1:
766 self._logger.log(
767 VERBOSE_LOG_LEVEL,
768 "Adjusted crossfade duration from %.2fs to %.2fs to align with "
769 "downbeat at %.2fs (later)",
770 crossfade_duration,
771 adjusted_duration,
772 later_downbeat,
773 )
774 return adjusted_duration
775
776 # If no suitable downbeat found, return original duration
777 self._logger.log(
778 VERBOSE_LOG_LEVEL,
779 "Could not adjust crossfade duration to downbeats, using original %.2fs",
780 crossfade_duration,
781 )
782 return crossfade_duration
783
784 def _align_rolling_intro(
785 self, crossfade_duration: float, fadein_trim_start: float | None
786 ) -> tuple[bool, float | None]:
787 """
788 Deepen B's trim when the overlap can't cover its intro (else B's groove lands on dead air).
789
790 Returns ``(feasible, trim)``: ``(False, None)`` when a sung run leaves
791 no legal cut anywhere in the buffer, else the (possibly deepened) trim.
792 """
793 import numpy as np # noqa: PLC0415
794
795 entry = self._ctx.natural_entry
796 trim = fadein_trim_start or 0.0
797 if entry <= 0.0 or entry - trim <= crossfade_duration:
798 return True, fadein_trim_start
799 if entry > SMART_CROSSFADE_DURATION:
800 # groove enters beyond the buffered head — unreachable defensively
801 return True, fadein_trim_start
802 deep_trim = entry - crossfade_duration
803 downbeats = self._ctx.incoming.downbeats
804 if len(downbeats):
805 deep_trim = float(downbeats[np.argmin(np.abs(downbeats - deep_trim))])
806 guarded = self._guard_deep_trim(deep_trim, crossfade_duration)
807 if guarded is None:
808 self._logger.log(
809 VERBOSE_LOG_LEVEL,
810 "Rolling intro: no legal cut clears the sung run within the buffer; "
811 "dropping to a shorter overlap instead",
812 )
813 return False, None
814 self._logger.log(
815 VERBOSE_LOG_LEVEL,
816 "Rolling intro: trimming %.1fs of pre-groove intro to keep the handover anchored",
817 guarded,
818 )
819 return True, guarded
820
821 def _guard_deep_trim(self, deep_trim: float, crossfade_duration: float) -> float | None:
822 """Push a deep-trim cut off a protected run; ``None`` when no legal cut fits the buffer."""
823 import numpy as np # noqa: PLC0415
824
825 # the cut is a minimum, so only search later: first unprotected downbeat at/after wins
826 if self._ctx.incoming_profile is None:
827 return deep_trim
828 bar_starts = self._ctx.incoming_profile.bar_starts
829 protected = self._protected_bars(self._ctx.incoming_profile)
830 start_idx = int(np.searchsorted(bar_starts, deep_trim - 1e-6))
831 if start_idx >= len(bar_starts) or not protected[start_idx]:
832 return deep_trim
833 for i in range(start_idx, len(protected)):
834 if protected[i]:
835 continue
836 candidate = float(bar_starts[i])
837 if candidate + crossfade_duration <= SMART_CROSSFADE_DURATION:
838 return candidate
839 break
840 return None
841
842 @staticmethod
843 def _protected_bars(profile: BandProfile) -> npt.NDArray[np.bool_]:
844 """Mark each bar of ``profile`` as protected: low-silent but voice/melody-active."""
845 low = profile.bar_power["low"]
846 low_mid = profile.bar_power["low_mid"]
847 mid = profile.bar_power["mid"]
848 low_silent = low < _TRIM_GUARD_LOW_FLOOR * profile.reference["low"]
849 voice_active = (low_mid >= _TRIM_GUARD_VOICE_FLOOR * profile.reference["low_mid"]) | (
850 mid >= _TRIM_GUARD_VOICE_FLOOR * profile.reference["mid"]
851 )
852 return low_silent & voice_active
853
854 def _build_lazy_overlay(self, spec: CandidateSpec) -> Candidate:
855 """Build the unphrased long-overlay candidate: anchored at the audible end, no alignment."""
856 tail = self._anchored_tail(spec.anchor_s)
857 plan = TransitionPlan(
858 tier=spec.tier,
859 fade_out_window=tail.effective_end,
860 crossfade_duration=min(_LAZY_OVERLAY_SECONDS, tail.effective_end),
861 tempo_plan=TempoPlan(),
862 fadeout_trim=tail.fadeout_trim,
863 fadein_trim_start=None,
864 )
865 return Candidate(
866 spec=spec, plan=plan, metrics=self._score(spec, plan), ideal_bars=spec.ideal_bars
867 )
868
869 def _score(self, spec: CandidateSpec, plan: TransitionPlan) -> PlanMetrics:
870 """Score a candidate: trims, retained vocal time, downbeat alignment, collision."""
871 ctx = self._ctx
872 audible_outgoing_trim = max(0.0, ctx.audio_end - plan.fade_out_window)
873 anchor_on_downbeat = self._is_on_downbeat(plan.fade_out_window)
874 # deliberate extension over the old planner (which never scored the
875 # energy-only path): policies need real trim/downbeat facts on every
876 # candidate; each vocal-dependent field needs only its own deck's mask
877 outgoing_vocal_fade_seconds = 0.0
878 collision_seconds = weighted_collision = 0.0
879 if ctx.vocal_out_scoring is not None:
880 outgoing_windows = self._rendered_outgoing_windows(plan)
881 in_fade = [
882 (max(0.0, left), min(plan.crossfade_duration, right))
883 for left, right in outgoing_windows
884 if right > 0.0 and left < plan.crossfade_duration
885 ]
886 outgoing_vocal_fade_seconds = sum(
887 right - left for left, right in merge_windows(in_fade)
888 )
889 if ctx.vocal_in_scoring is not None:
890 collision_seconds, weighted_collision = collision_metrics(
891 outgoing_windows,
892 self._rendered_incoming_windows(plan),
893 plan.crossfade_duration,
894 )
895 return PlanMetrics(
896 strategy=spec.strategy,
897 audible_outgoing_trim=audible_outgoing_trim,
898 outgoing_vocal_fade_seconds=outgoing_vocal_fade_seconds,
899 anchor_on_downbeat=anchor_on_downbeat,
900 collision_seconds=collision_seconds,
901 weighted_collision_seconds=weighted_collision,
902 )
903
904 def _rendered_outgoing_windows(self, plan: TransitionPlan) -> list[tuple[float, float]]:
905 """Map the outgoing (unpadded) vocal scoring mask into rendered crossfade-local seconds."""
906 assert self._ctx.vocal_out_scoring is not None # narrowed by the caller
907 rendered_anchor = self._rendered_time(plan, plan.fade_out_window)
908 rendered_start = rendered_anchor - plan.crossfade_duration
909 return [
910 (
911 self._rendered_time(plan, left) - rendered_start,
912 self._rendered_time(plan, right) - rendered_start,
913 )
914 for left, right in self._ctx.vocal_out_scoring.windows
915 ]
916
917 def _rendered_incoming_windows(self, plan: TransitionPlan) -> list[tuple[float, float]]:
918 """Map the incoming (unpadded) scoring mask into the plan's fadein-trim-relative seconds."""
919 assert self._ctx.vocal_in_scoring is not None # narrowed by the caller
920 trim = plan.fadein_trim_start or 0.0
921 return [(left - trim, right - trim) for left, right in self._ctx.vocal_in_scoring.windows]
922
923 @staticmethod
924 def _rendered_time(plan: TransitionPlan, input_time: float) -> float:
925 """Map a buffer-local outgoing input-time position to its rendered-stream position."""
926 clamped = max(0.0, min(input_time, plan.fade_out_window))
927 return clamped - plan.tempo_plan.savings_until(clamped)
928
929 def _is_on_downbeat(self, position: float, tolerance: float = 0.05) -> bool:
930 """Whether a buffer-local position sits within tolerance of an outgoing downbeat."""
931 import numpy as np # noqa: PLC0415
932
933 downbeats = np.asarray(self._ctx.protective_downbeats, dtype=np.float32)
934 return bool(len(downbeats) and np.min(np.abs(downbeats - position)) <= tolerance)
935
936
937@dataclass(frozen=True, slots=True)
938class _AnchoredTail:
939 """The outgoing tail derived for one anchor: masked grids, trim, and effective end."""
940
941 effective_end: float
942 fadeout_trim: FadeOutTrim | None
943 beats: npt.NDArray[np.float32]
944 downbeats: npt.NDArray[np.float32]
945 extrapolated_downbeats: npt.NDArray[np.float32]
946
947
948def _vocal_duties(ctx: TransitionContext) -> tuple[float, float] | None:
949 """Outgoing/incoming vocal duty fractions the instrumental-blend/lazy-overlay gates key on."""
950 if ctx.vocal_out_scoring is None or ctx.vocal_in_scoring is None:
951 return None
952 out_duty = sum(right - left for left, right in ctx.vocal_out_scoring.windows) / max(
953 ctx.audio_end, 0.001
954 )
955 in_duty = sum(right - left for left, right in ctx.vocal_in_scoring.windows) / float(
956 SMART_CROSSFADE_DURATION
957 )
958 return out_duty, in_duty
959
960
961def _window_duties(ctx: TransitionContext, seconds: float) -> tuple[float, float] | None:
962 """
963 Vocal duty per deck over the window an unphrased overlay of ``seconds`` actually spans.
964
965 :param ctx: The transition context.
966 :param seconds: Requested overlay length; the outgoing window is the last
967 ``seconds`` before the audible end, the incoming window its first ``seconds``.
968 """
969 if ctx.vocal_out_scoring is None or ctx.vocal_in_scoring is None:
970 return None
971 # mirrors the anchored tail: a sub-half-second gap to the buffer end is not trimmed
972 effective_end = ctx.audio_end
973 if effective_end >= ctx.buffer_duration - 0.5:
974 effective_end = ctx.buffer_duration
975 span = min(seconds, effective_end)
976 if span <= 0.0:
977 return None
978 out_secs = sum(
979 max(0.0, min(right, effective_end) - max(left, effective_end - span))
980 for left, right in ctx.vocal_out_scoring.windows
981 )
982 in_secs = sum(
983 max(0.0, min(right, span) - max(left, 0.0)) for left, right in ctx.vocal_in_scoring.windows
984 )
985 return out_secs / span, in_secs / span
986
987
988def _fade_onset_pin(ctx: TransitionContext) -> float:
989 """Buffer-local anchor pin ahead of a detected mastered fade, else the default anchor."""
990 if ctx.fade_onset is None:
991 return ctx.default_anchor
992 # never pin inside A's own vocal window: exiting mid-phrase is the exact
993 # defect this pin fixes, so the vocal end floors the pin
994 lead_end = ctx.vocal_out_placement.last_end() if ctx.vocal_out_placement else 0.0
995 onset_local = max(ctx.fade_onset, lead_end)
996 if onset_local < MIN_EFFECTIVE_FADE_BUFFER:
997 return ctx.default_anchor
998 return min(ctx.default_anchor, onset_local)
999
1000
1001def _entry_options(ctx: TransitionContext, bars: int) -> list[float]:
1002 """B entries for a rung: groove alignment, intro-keeping 0.0 (bars<=2), then natural entry."""
1003 import numpy as np # noqa: PLC0415
1004
1005 bar_b = ctx.incoming.beats_per_bar * 60.0 / ctx.incoming.bpm
1006 options: list[float] = []
1007 deep = ctx.natural_entry - bars * bar_b
1008 if deep > 0.0 and len(ctx.incoming.downbeats):
1009 downbeats = ctx.incoming.downbeats
1010 snapped = float(downbeats[np.argmin(np.abs(downbeats - deep))])
1011 in_mask = ctx.vocal_in_placement is not None and point_in_mask(
1012 ctx.vocal_in_placement, snapped
1013 )
1014 if snapped >= 0.0 and not in_mask:
1015 options.append(snapped)
1016 # at 1-2 bars the overlap is a handover, not a blend: keeping B's whole
1017 # intro (zero trim) lets it build naturally after A's tail rides out; it
1018 # goes before the natural entry so a scoring tie prefers keeping the intro
1019 if bars <= 2 and ctx.natural_entry > 0.0 and 0.0 not in options:
1020 options.append(0.0)
1021 if ctx.natural_entry not in options:
1022 options.append(ctx.natural_entry)
1023 return options
1024
1025
1026def _outgoing_vocal_end(ctx: TransitionContext) -> float:
1027 """Return the last outgoing vocal end within the audible tail, or 0.0 without vocal data."""
1028 mask = ctx.vocal_out_placement
1029 if mask is None or not mask.windows:
1030 return 0.0
1031 return min(mask.last_end(), ctx.audio_end)
1032
1033
1034def _nearest_protective_anchor(
1035 ctx: TransitionContext, target: float, *, prefer_earliest: bool = True
1036) -> float:
1037 """
1038 Protective downbeat at/after target within the RMS-audible boundary.
1039
1040 :param ctx: The transition context.
1041 :param target: Earliest buffer-local position the anchor may take.
1042 :param prefer_earliest: Use the first qualifying downbeat (protecting a
1043 vocal needs only just enough extra room) instead of the last (closing
1044 an audible-trim gap as tightly as possible).
1045 """
1046 candidates = [
1047 downbeat for downbeat in ctx.protective_downbeats if target <= downbeat <= ctx.audio_end
1048 ]
1049 if candidates:
1050 return candidates[0] if prefer_earliest else candidates[-1]
1051 return min(target, ctx.audio_end)
1052