/
/
1"""Podcastfeed -> Mass."""
2
3import logging
4from datetime import UTC, datetime
5from io import BytesIO
6from math import isfinite
7from typing import TYPE_CHECKING, Any
8
9import podcastparser
10from aiohttp.client import ClientError, ClientTimeout
11from music_assistant_models.enums import ContentType, ImageType, LinkType, MediaType
12from music_assistant_models.errors import MediaNotFoundError
13from music_assistant_models.media_items import (
14 AudioFormat,
15 ItemMapping,
16 MediaItemChapter,
17 MediaItemImage,
18 MediaItemLink,
19 Podcast,
20 PodcastEpisode,
21 ProviderMapping,
22 UniqueList,
23)
24
25if TYPE_CHECKING:
26 import aiohttp
27
28 from music_assistant.mass import MusicAssistant
29
30LOGGER = logging.getLogger(__name__)
31
32# best-effort enrichment must never stall episode resolution, so cap the chapter fetch
33_CHAPTERS_FETCH_TIMEOUT = ClientTimeout(total=10)
34
35# defaults for the parsed-feed cache shared by the podcast providers
36CACHE_CATEGORY_PODCAST_FEED = 0
37PODCAST_FEED_CACHE_EXPIRATION = 24 * 3600
38
39
40async def get_podcastparser_dict(
41 *, session: aiohttp.ClientSession, feed_url: str, max_episodes: int = 0
42) -> dict[str, Any]:
43 """
44 Get feed parsed by podcastparser by providing the url.
45
46 max_episodes = 0 does not limit the returned episodes.
47 """
48 feed_data: bytes | None = None
49 # without user agent, some feeds can not be retrieved
50 # https://github.com/music-assistant/support/issues/3596
51 # but, reports on discord show, that also the opposite may be true
52 for headers in [{"User-Agent": "Mozilla/5.0"}, {}]:
53 # raises ClientError on status failure
54 # ClientError is the base class of all possible Error, i.e. not authorized,
55 # url doesn't exist etc. The body is read inside the context manager, so a
56 # connection is never left open when a feed fails midway.
57 try:
58 async with session.get(feed_url, headers=headers, raise_for_status=True) as response:
59 feed_data = await response.read()
60 except ClientError:
61 continue
62 break
63 if feed_data is None:
64 # we did not get a single acceptable response
65 raise MediaNotFoundError(
66 f"Did not get acceptable response while trying to access {feed_url}."
67 )
68 feed_stream = BytesIO(feed_data)
69 try:
70 return podcastparser.parse(feed_url, feed_stream, max_episodes=max_episodes) # type: ignore[no-any-return]
71 except podcastparser.FeedParseError:
72 raise MediaNotFoundError(f"The url at {feed_url} returns invalid RSS data.")
73
74
75async def get_cached_podcast(
76 *,
77 mass: MusicAssistant,
78 provider_instance_id: str,
79 feed_url: str,
80 max_episodes: int = 0,
81 cache_category: int = CACHE_CATEGORY_PODCAST_FEED,
82 cache_expiration: int = PODCAST_FEED_CACHE_EXPIRATION,
83) -> dict[str, Any]:
84 """
85 Return a podcast's parsed feed, retrieving and caching it when not cached yet.
86
87 :param mass: The MusicAssistant instance holding the cache.
88 :param provider_instance_id: Provider instance the cache entry belongs to.
89 :param feed_url: The podcast's feed url, also used as the cache key.
90 :param max_episodes: Maximum number of episodes to parse, 0 for unlimited.
91 :param cache_category: Cache category to store the parsed feed under.
92 :param cache_expiration: Time in seconds the cached feed stays valid.
93 :raises MediaNotFoundError: If the feed could not be retrieved or parsed.
94 """
95 parsed_feed = await mass.cache.get(
96 key=feed_url,
97 provider=provider_instance_id,
98 category=cache_category,
99 default=None,
100 )
101 if parsed_feed is None:
102 return await refresh_cached_podcast(
103 mass=mass,
104 provider_instance_id=provider_instance_id,
105 feed_url=feed_url,
106 max_episodes=max_episodes,
107 cache_category=cache_category,
108 cache_expiration=cache_expiration,
109 )
110 # this is a dictionary from podcastparser
111 return parsed_feed # type: ignore[no-any-return]
112
113
114async def refresh_cached_podcast(
115 *,
116 mass: MusicAssistant,
117 provider_instance_id: str,
118 feed_url: str,
119 max_episodes: int = 0,
120 cache_category: int = CACHE_CATEGORY_PODCAST_FEED,
121 cache_expiration: int = PODCAST_FEED_CACHE_EXPIRATION,
122) -> dict[str, Any]:
123 """
124 Retrieve a podcast's feed and store it in the cache, replacing any cached copy.
125
126 Use this on the library sync path: the sync must always refresh the cached feed,
127 regardless of whether a (still valid) cache entry exists.
128
129 :param mass: The MusicAssistant instance holding the cache.
130 :param provider_instance_id: Provider instance the cache entry belongs to.
131 :param feed_url: The podcast's feed url, also used as the cache key.
132 :param max_episodes: Maximum number of episodes to parse, 0 for unlimited.
133 :param cache_category: Cache category to store the parsed feed under.
134 :param cache_expiration: Time in seconds the cached feed stays valid.
135 :raises MediaNotFoundError: If the feed could not be retrieved or parsed.
136 """
137 parsed_feed = await get_podcastparser_dict(
138 session=mass.http_session, feed_url=feed_url, max_episodes=max_episodes
139 )
140 await mass.cache.set(
141 key=feed_url,
142 provider=provider_instance_id,
143 category=cache_category,
144 data=parsed_feed,
145 expiration=cache_expiration,
146 )
147 return parsed_feed
148
149
150def parse_podcast(
151 *,
152 feed_url: str,
153 parsed_feed: dict[str, Any],
154 instance_id: str,
155 domain: str,
156 mass_item_id: str | None = None,
157) -> Podcast:
158 """
159 Podcast -> Mass Podcast.
160
161 The item_id is the feed url by default, or the optional mass_item_id instead.
162 """
163 publisher = parsed_feed.get("author") or parsed_feed.get("itunes_author", "NO_AUTHOR")
164 item_id = feed_url if mass_item_id is None else mass_item_id
165 mass_podcast = Podcast(
166 item_id=item_id,
167 name=parsed_feed.get("title", "NO_TITLE"),
168 publisher=publisher,
169 provider=instance_id,
170 uri=parsed_feed.get("link"),
171 provider_mappings={
172 ProviderMapping(
173 item_id=item_id,
174 provider_domain=domain,
175 provider_instance=instance_id,
176 )
177 },
178 )
179 genres: list[str] = []
180 if _genres := parsed_feed.get("itunes_categories"):
181 for _sub_genre in _genres:
182 if isinstance(_sub_genre, list):
183 genres.extend(x for x in _sub_genre if isinstance(x, str))
184 elif isinstance(_sub_genre, str):
185 genres.append(_sub_genre)
186
187 mass_podcast.metadata.genres = set(genres)
188 mass_podcast.metadata.description = parsed_feed.get("description", "")
189 mass_podcast.metadata.explicit = parsed_feed.get("explicit", False)
190 language = parsed_feed.get("language")
191 if language is not None:
192 mass_podcast.metadata.languages = UniqueList([language])
193 episodes = parsed_feed.get("episodes", [])
194 mass_podcast.total_episodes = len(episodes)
195 podcast_cover = parsed_feed.get("cover_url")
196 if podcast_cover is not None:
197 mass_podcast.metadata.images = UniqueList(
198 [
199 MediaItemImage(
200 type=ImageType.THUMB,
201 path=podcast_cover,
202 provider=instance_id,
203 remotely_accessible=True,
204 )
205 ]
206 )
207 return mass_podcast
208
209
210def get_stream_url_from_episode(*, episode: dict[str, Any]) -> str | None:
211 """
212 Give the url of the episode's playable enclosure, or None if the episode has none.
213
214 Prefers the first audio or video enclosure, falling back to any non-image one.
215
216 :param episode: A single episode dict as returned by podcastparser.
217 """
218 fallback: str | None = None
219 for enclosure in episode.get("enclosures", []):
220 url = enclosure.get("url")
221 if not url:
222 continue
223 mime_type = str(enclosure.get("mime_type") or "")
224 if mime_type.startswith(("audio/", "video/")):
225 return str(url)
226 # feeds do declare bogus mime types (e.g. type="file") for real audio, so anything
227 # that is not an image stays a candidate in case no audio enclosure is declared
228 if fallback is None and not mime_type.startswith("image/"):
229 fallback = str(url)
230 return fallback
231
232
233def get_stream_url_and_guid_from_episode(*, episode: dict[str, Any]) -> tuple[str, str | None]:
234 """Give episode's stream url and guid, if it exists."""
235 stream_url = get_stream_url_from_episode(episode=episode)
236 if stream_url is None:
237 raise ValueError("Episode has no playable enclosure")
238 guid = episode.get("guid")
239 if guid is not None:
240 # The media's item_id is {prov_podcast_id} {guid_or_stream_url}
241 # see parse_podcast_episode.
242 # However, the guid must not contain a space, otherwise it is invalid.
243 # We cannot check, if it is a proper guid (uuid.UUID4(...)), as some podcast feeds
244 # do not follow the standard.
245 guid = None if len(guid.split(" ")) > 1 else guid
246 return stream_url, guid
247
248
249def find_episode_stream_url(*, parsed_feed: dict[str, Any], guid_or_stream_url: str) -> str | None:
250 """
251 Return the stream url of the episode identified by the item_id's episode part.
252
253 :param parsed_feed: The podcastparser dict of the feed holding the episode.
254 :param guid_or_stream_url: Episode part of the item_id, see parse_podcast_episode.
255 """
256 for episode in parsed_feed.get("episodes", []):
257 try:
258 stream_url, guid = get_stream_url_and_guid_from_episode(episode=episode)
259 except ValueError:
260 # episode without a playable enclosure carries no stream; skip it instead of
261 # aborting the lookup for the (potentially later) requested episode
262 continue
263 # only a guid rejected as unusable (None) falls back to the stream url, so an
264 # empty guid resolves the same way it was turned into an item_id
265 if guid_or_stream_url == (stream_url if guid is None else guid):
266 return stream_url
267 return None
268
269
270def parse_podcast_episode(
271 *,
272 episode: dict[str, Any],
273 prov_podcast_id: str,
274 episode_cnt: int,
275 podcast_cover: str | None = None,
276 podcast_name: str | None = None,
277 instance_id: str,
278 domain: str,
279 mass_item_id: str | None = None,
280) -> PodcastEpisode | None:
281 """
282 Podcast Episode -> Mass Podcast Episode.
283
284 The item_id is {prov_podcast_id} {guid_or_stream_url} by default, or the optional mass_item_id
285 instead. The podcast_cover is used, if the episode should not have its own cover. The
286 podcast_name names the parent podcast reference (falls back to the episode title if unset).
287
288 The function returns None, if the episode enclosure is missing, i.e. there is no stream
289 information present.
290 """
291 episode_duration = episode.get("total_time", 0.0)
292 episode_title = episode.get("title", "NO_EPISODE_TITLE")
293 episode_cover = episode.get("episode_art_url", podcast_cover)
294
295 # this is unix epoch in s, and 0 if unknown
296 episode_published: int | None = episode.get("published")
297 if episode_published == 0:
298 episode_published = None
299
300 # prefer the explicit itunes:episode number for ordering (parity with podcast_index);
301 # fall back to the feed enumeration order when the feed omits it
302 episode_number = episode.get("number")
303 episode_position = (
304 episode_number if isinstance(episode_number, int) and episode_number > 0 else episode_cnt
305 )
306
307 try:
308 stream_url, guid = get_stream_url_and_guid_from_episode(episode=episode)
309 except ValueError:
310 # we are missing the episode enclosure or stream information
311 return None
312 # We treat a guid as invalid if contains a space.
313 guid_or_stream_url = guid if guid is not None and len(guid.split(" ")) == 1 else stream_url
314
315 # Default episode id. A guid is preferred as identification.
316 episode_id = f"{prov_podcast_id} {guid_or_stream_url}" if mass_item_id is None else mass_item_id
317 mass_episode = PodcastEpisode(
318 item_id=episode_id,
319 provider=instance_id,
320 name=episode_title,
321 duration=int(episode_duration),
322 position=episode_position,
323 podcast=ItemMapping(
324 item_id=prov_podcast_id,
325 provider=instance_id,
326 name=podcast_name or episode_title,
327 media_type=MediaType.PODCAST,
328 ),
329 provider_mappings={
330 ProviderMapping(
331 item_id=episode_id,
332 provider_domain=domain,
333 provider_instance=instance_id,
334 audio_format=AudioFormat(
335 content_type=ContentType.try_parse(stream_url),
336 ),
337 url=stream_url,
338 )
339 },
340 )
341 if episode_published is not None:
342 mass_episode.metadata.release_date = datetime.fromtimestamp(episode_published, tz=UTC)
343
344 # description (podcastparser normalizes this to plain text, defaulting to "")
345 if description := episode.get("description"):
346 mass_episode.metadata.description = description
347
348 # explicit flag (itunes:explicit); only set when the feed actually declared it
349 explicit = episode.get("explicit")
350 if explicit is not None:
351 mass_episode.metadata.explicit = bool(explicit)
352
353 # episode webpage (the item <link>)
354 if link := episode.get("link"):
355 mass_episode.metadata.links = {MediaItemLink(type=LinkType.WEBSITE, url=link)}
356
357 # hosts/guests (podcast:person), mapped to performer names
358 if performers := parse_podcast_persons(episode.get("persons")):
359 mass_episode.metadata.performers = set(performers)
360
361 # inline chapters (Podlove Simple Chapters, parsed by podcastparser)
362 if chapters := episode.get("chapters"):
363 _chapters: list[MediaItemChapter] = []
364 for chapter in chapters:
365 if not isinstance(chapter, dict):
366 continue
367 title = chapter.get("title")
368 start = chapter.get("start")
369 # start may legitimately be 0 (opening chapter), so test against None
370 if title and start is not None:
371 _chapters.append(
372 MediaItemChapter(position=len(_chapters) + 1, name=title, start=start)
373 )
374 if _chapters:
375 mass_episode.metadata.chapters = _chapters
376
377 # cover image
378 if episode_cover is not None:
379 mass_episode.metadata.images = UniqueList(
380 [
381 MediaItemImage(
382 type=ImageType.THUMB,
383 path=episode_cover,
384 provider=instance_id,
385 remotely_accessible=True,
386 )
387 ]
388 )
389
390 return mass_episode
391
392
393def parse_podcast_persons(persons: Any) -> list[str]:
394 """
395 Extract performer names from a Podcasting 2.0 ``podcast:person`` collection.
396
397 Accepts the persons list as produced by podcastparser (feeds) or by the Podcast
398 Index API. Returns the display names de-duplicated (case-insensitive) in feed
399 order; any non-list input yields no names, so callers can pass a raw
400 ``.get("persons")`` without guarding.
401
402 :param persons: The raw persons collection, or any value (non-lists yield []).
403 """
404 names: list[str] = []
405 seen: set[str] = set()
406 if not isinstance(persons, list):
407 return names
408 for person in persons:
409 name = person.get("name") if isinstance(person, dict) else person
410 if not isinstance(name, str) or not (name := name.strip()):
411 continue
412 key = name.casefold()
413 if key in seen:
414 continue
415 seen.add(key)
416 names.append(name)
417 return names
418
419
420def _coerce_seconds(value: Any) -> float | None:
421 """Coerce a chapter time value to float seconds, or None if not parseable."""
422 if value is None:
423 return None
424 try:
425 seconds = float(value)
426 except TypeError, ValueError:
427 return None
428 return seconds if isfinite(seconds) else None
429
430
431def parse_chapters_from_json(data: dict[str, Any]) -> list[MediaItemChapter]:
432 """
433 Parse a Podcasting 2.0 ``podcast:chapters`` JSON document into chapters.
434
435 Spec: https://github.com/Podcastindex-org/podcast-namespace/blob/main/docs/1.0.md#chapters
436 Entries without a usable ``startTime``/``title`` are skipped, as are entries
437 explicitly hidden from the table of contents (``toc: false``).
438 """
439 chapters: list[MediaItemChapter] = []
440 raw_chapters = data.get("chapters")
441 if not isinstance(raw_chapters, list):
442 return chapters
443 for raw_chapter in raw_chapters:
444 if not isinstance(raw_chapter, dict):
445 continue
446 if raw_chapter.get("toc") is False:
447 continue
448 title = raw_chapter.get("title")
449 start = _coerce_seconds(raw_chapter.get("startTime"))
450 if not isinstance(title, str) or not title or start is None:
451 continue
452 chapters.append(
453 MediaItemChapter(
454 position=len(chapters) + 1,
455 name=title,
456 start=start,
457 end=_coerce_seconds(raw_chapter.get("endTime")),
458 )
459 )
460 return chapters
461
462
463async def enrich_episode_chapters(
464 *,
465 session: aiohttp.ClientSession,
466 chapters_json_url: str | None,
467 mass_episode: PodcastEpisode,
468) -> None:
469 """
470 Attach ``podcast:chapters`` (external JSON) to an episode lacking inline chapters.
471
472 Transport-agnostic: the caller supplies the chapters JSON URL directly (e.g.
473 podcastparser's ``chapters_json_url`` or the Podcast Index API ``chaptersUrl``),
474 so any podcast provider can reuse it. Chapter data is supplementary: any
475 fetch/parse failure is logged and ignored so it can never break episode
476 resolution or playback. Intended for the single-episode path only, to avoid a
477 network request per episode when listing a whole podcast.
478
479 :param session: The aiohttp session used for the best-effort fetch.
480 :param chapters_json_url: The Podcasting 2.0 chapters JSON URL, or None to skip.
481 :param mass_episode: The episode to enrich; untouched if it already has chapters.
482 """
483 if mass_episode.metadata.chapters:
484 return
485 url = chapters_json_url
486 if not url:
487 return
488 # send a browser UA: some podcast hosts/CDNs reject non-browser agents
489 # (music-assistant/support#3596). Chapter data is supplementary, so unlike
490 # get_podcastparser_dict we make a single best-effort attempt with no UA-less retry.
491 # TimeoutError (raised on total-timeout) is not a ClientError, so catch it explicitly
492 # to keep this enrichment best-effort. The parse stays inside the try so a malformed
493 # document can never break episode resolution either.
494 try:
495 async with session.get(
496 url,
497 headers={"User-Agent": "Mozilla/5.0"},
498 raise_for_status=True,
499 timeout=_CHAPTERS_FETCH_TIMEOUT,
500 ) as response:
501 data = await response.json(content_type=None)
502 if isinstance(data, dict) and (chapters := parse_chapters_from_json(data)):
503 mass_episode.metadata.chapters = chapters
504 except (ClientError, TimeoutError, ValueError, TypeError) as err:
505 LOGGER.warning("Failed to fetch podcast chapters from %s: %s", url, err)
506