/
/
1"""Internet Archive music provider implementation."""
2
3from __future__ import annotations
4
5import contextlib
6import re
7from collections.abc import AsyncGenerator
8from typing import TYPE_CHECKING, Any
9
10import aiohttp
11from music_assistant_models.enums import MediaType, ProviderFeature
12from music_assistant_models.errors import InvalidDataError, MediaNotFoundError
13from music_assistant_models.media_items import (
14 Album,
15 Artist,
16 Audiobook,
17 MediaItemChapter,
18 Podcast,
19 PodcastEpisode,
20 ProviderMapping,
21 SearchResults,
22 Track,
23)
24from music_assistant_models.unique_list import UniqueList
25
26from music_assistant.constants import UNKNOWN_ARTIST
27from music_assistant.controllers.cache import use_cache
28from music_assistant.helpers.throttle_retry import ThrottlerManager, throttle_with_retries
29from music_assistant.models.music_provider import MusicProvider
30
31from .helpers import InternetArchiveClient, clean_text, extract_year, parse_duration
32from .parsers import (
33 add_item_image,
34 artist_exists,
35 create_artist,
36 create_provider_mapping,
37 create_title_from_identifier,
38 doc_to_album,
39 doc_to_audiobook,
40 doc_to_podcast,
41 doc_to_track,
42 is_audiobook_content,
43 is_likely_album,
44 is_podcast_content,
45)
46from .streaming import InternetArchiveStreaming
47
48if TYPE_CHECKING:
49 from music_assistant_models.config_entries import ConfigEntry, ProviderConfig
50 from music_assistant_models.provider import ProviderManifest
51 from music_assistant_models.streamdetails import StreamDetails
52
53 from music_assistant import MusicAssistant
54
55
56class InternetArchiveProvider(MusicProvider):
57 """Implementation of Internet Archive music provider."""
58
59 def __init__(
60 self,
61 mass: MusicAssistant,
62 manifest: ProviderManifest,
63 config: ProviderConfig,
64 supported_features: set[ProviderFeature],
65 ) -> None:
66 """Initialize the provider."""
67 super().__init__(mass, manifest, config, supported_features)
68 self.throttler = ThrottlerManager(
69 rate_limit=10, period=60, retry_attempts=5, initial_backoff=5
70 )
71 self.client = InternetArchiveClient(mass)
72 self.streaming = InternetArchiveStreaming(self)
73
74 @property
75 def max_concurrent_streams(self) -> None:
76 """Allow unlimited concurrent upstream source streams."""
77 return None
78
79 async def get_config_entries(self) -> tuple[ConfigEntry, ...]:
80 """Return Config entries to configure this provider."""
81 return ()
82
83 @property
84 def is_streaming_provider(self) -> bool:
85 """Return True if provider is a streaming provider."""
86 return True
87
88 @throttle_with_retries
89 async def _get_json(self, url: str, params: dict[str, Any] | None = None) -> dict[str, Any]:
90 """Make a GET request and return JSON response with throttling."""
91 return await self.client._get_json(url, params)
92
93 @throttle_with_retries
94 async def _search(self, **kwargs: Any) -> dict[str, Any]:
95 """Throttled search wrapper."""
96 return await self.client.search(**kwargs)
97
98 @throttle_with_retries
99 async def _get_metadata(self, identifier: str) -> dict[str, Any]:
100 """Throttled metadata wrapper."""
101 return await self.client.get_metadata(identifier)
102
103 @use_cache(expiration=86400 * 30) # 30 days - file listings are static
104 @throttle_with_retries
105 async def _get_audio_files(self, identifier: str) -> list[dict[str, Any]]:
106 """Throttled audio files wrapper."""
107 return await self.client.get_audio_files(identifier)
108
109 @use_cache(86400 * 7) # 7 days
110 async def search(
111 self,
112 search_query: str,
113 media_types: list[MediaType],
114 limit: int = 5,
115 ) -> SearchResults:
116 """
117 Perform search on Internet Archive.
118
119 Uses multiple search strategies to maximize result coverage with
120 proper result accumulation and broader search patterns.
121
122 Args:
123 search_query: The search term to look for
124 media_types: List of media types to search for
125 limit: Maximum number of results to return per media type
126
127 Returns:
128 SearchResults object containing found items
129 """
130 if not search_query.strip():
131 return SearchResults()
132
133 # Adjust search intensity based on what's being requested
134 rows_per_strategy = min(limit * 2, 16) if len(media_types) > 1 else min(limit * 2, 100)
135
136 # Collect results in separate lists
137 tracks: list[Track] = []
138 albums: list[Album] = []
139 artists: list[Artist] = []
140 audiobooks: list[Audiobook] = []
141 podcasts: list[Podcast] = []
142
143 # Track processed identifiers to avoid duplicates across strategies
144 processed_ids: set[str] = set()
145
146 # Build search strategies based on requested media types
147 search_strategies = []
148
149 # For music searches: focus on title and creator
150 if any(mt in media_types for mt in [MediaType.TRACK, MediaType.ALBUM, MediaType.ARTIST]):
151 search_strategies.extend(
152 [
153 (f"creator:({search_query}) AND mediatype:audio", "downloads desc"),
154 (f"title:({search_query}) AND mediatype:audio", "downloads desc"),
155 (f"subject:({search_query}) AND mediatype:audio", "downloads desc"),
156 ]
157 )
158
159 # For audiobooks: search within audiobook collections, still limit to audio
160 if MediaType.AUDIOBOOK in media_types:
161 audiobook_query = f"{search_query} AND collection:(librivoxaudio OR audio_bookspoetry) AND mediatype:audio"
162 search_strategies.append((audiobook_query, "downloads desc"))
163
164 # For podcasts: search within podcast collections
165 if MediaType.PODCAST in media_types:
166 podcast_query = f"{search_query} AND collection:podcasts AND mediatype:audio"
167 search_strategies.append((podcast_query, "downloads desc"))
168
169 for strategy_idx, (strategy_query, sort_order) in enumerate(search_strategies):
170 self.logger.debug("Trying search strategy %d: %s", strategy_idx + 1, strategy_query)
171
172 try:
173 search_response = await self._search(
174 query=strategy_query,
175 rows=rows_per_strategy,
176 sort=sort_order,
177 )
178
179 response_data = search_response.get("response", {})
180 docs = response_data.get("docs", [])
181 self.logger.debug(
182 "Strategy %d '%s' found %d raw results",
183 strategy_idx + 1,
184 strategy_query,
185 len(docs),
186 )
187
188 # Process results and extract different media types
189 strategy_processed = 0
190 strategy_skipped = 0
191
192 for doc in docs:
193 try:
194 identifier = doc.get("identifier")
195 if not identifier or identifier in processed_ids:
196 strategy_skipped += 1
197 continue
198
199 # Track this identifier to avoid duplicates
200 processed_ids.add(identifier)
201
202 await self._process_search_result(
203 doc, tracks, albums, artists, audiobooks, podcasts, media_types
204 )
205 strategy_processed += 1
206
207 # Check if we have enough results across all types
208 if self._has_sufficient_results(
209 tracks, albums, artists, audiobooks, podcasts, media_types, limit
210 ):
211 self.logger.debug(
212 "Sufficient results found after strategy %d, stopping search",
213 strategy_idx + 1,
214 )
215 break
216
217 except (InvalidDataError, KeyError) as err:
218 self.logger.debug("Skipping invalid search result: %s", err)
219 strategy_skipped += 1
220 continue
221
222 self.logger.debug(
223 "Strategy %d '%s': processed %d new items, skipped %d items. "
224 "Running totals - tracks: %d, albums: %d, artists: %d, "
225 "audiobooks: %d, podcasts: %d",
226 strategy_idx + 1,
227 strategy_query,
228 strategy_processed,
229 strategy_skipped,
230 len(tracks),
231 len(albums),
232 len(artists),
233 len(audiobooks),
234 len(podcasts),
235 )
236
237 # If we have sufficient results, stop trying more strategies
238 if self._has_sufficient_results(
239 tracks, albums, artists, audiobooks, podcasts, media_types, limit
240 ):
241 break
242
243 except Exception as err:
244 self.logger.warning("Search strategy %d failed: %s", strategy_idx + 1, err)
245 continue
246
247 # Log final results for debugging
248 self.logger.debug(
249 "Search for '%s' completed. Final results - tracks: %d, albums: %d, "
250 "artists: %d, audiobooks: %d, podcasts: %d (processed %d unique items)",
251 search_query,
252 len(tracks),
253 len(albums),
254 len(artists),
255 len(audiobooks),
256 len(podcasts),
257 len(processed_ids),
258 )
259
260 return SearchResults(
261 tracks=tracks[:limit] if MediaType.TRACK in media_types else [],
262 albums=albums[:limit] if MediaType.ALBUM in media_types else [],
263 artists=artists[:limit] if MediaType.ARTIST in media_types else [],
264 audiobooks=audiobooks[:limit] if MediaType.AUDIOBOOK in media_types else [],
265 podcasts=podcasts[:limit] if MediaType.PODCAST in media_types else [],
266 )
267
268 def _has_sufficient_results(
269 self,
270 tracks: list[Track],
271 albums: list[Album],
272 artists: list[Artist],
273 audiobooks: list[Audiobook],
274 podcasts: list[Podcast],
275 media_types: list[MediaType],
276 limit: int,
277 ) -> bool:
278 """Check if we have sufficient results for all requested media types."""
279 return (
280 (MediaType.TRACK not in media_types or len(tracks) >= limit)
281 and (MediaType.ALBUM not in media_types or len(albums) >= limit)
282 and (MediaType.ARTIST not in media_types or len(artists) >= limit)
283 and (MediaType.AUDIOBOOK not in media_types or len(audiobooks) >= limit)
284 and (MediaType.PODCAST not in media_types or len(podcasts) >= limit)
285 )
286
287 async def _process_search_result(
288 self,
289 doc: dict[str, Any],
290 tracks: list[Track],
291 albums: list[Album],
292 artists: list[Artist],
293 audiobooks: list[Audiobook],
294 podcasts: list[Podcast],
295 media_types: list[MediaType],
296 ) -> None:
297 """
298 Process a single search result document from Internet Archive.
299
300 Determines the appropriate media type and creates corresponding objects.
301 Uses improved heuristics to classify items as tracks, albums, or audiobooks.
302 """
303 identifier = doc.get("identifier")
304 if not identifier:
305 raise InvalidDataError("Missing identifier in search result")
306
307 title = clean_text(doc.get("title"))
308 creator = clean_text(doc.get("creator"))
309
310 # Be lenient - allow items without title if they have identifier
311 if not title and not identifier:
312 raise InvalidDataError("Missing both title and identifier in search result")
313
314 # Use identifier as fallback title if needed
315 if not title:
316 title = create_title_from_identifier(identifier)
317
318 # Determine what type of item this is
319 mediatype = doc.get("mediatype", "")
320 collection = doc.get("collection", [])
321 if isinstance(collection, str):
322 collection = [collection]
323
324 # Check if this is audiobook content using improved detection
325 if is_audiobook_content(doc) and MediaType.AUDIOBOOK in media_types:
326 audiobook = doc_to_audiobook(
327 doc, self.domain, self.instance_id, self.client.get_item_url
328 )
329 if audiobook:
330 audiobooks.append(audiobook)
331 return # Don't process as other media types
332
333 # Check if this is podcast content
334 if is_podcast_content(doc) and MediaType.PODCAST in media_types:
335 podcast = doc_to_podcast(doc, self.domain, self.instance_id, self.client.get_item_url)
336 if podcast:
337 podcasts.append(podcast)
338 return # Don't process as other media types
339
340 # For etree items, usually each item is an album (concert)
341 if mediatype == "etree" or "etree" in collection:
342 if MediaType.ALBUM in media_types:
343 album = doc_to_album(doc, self.domain, self.instance_id, self.client.get_item_url)
344 if album:
345 albums.append(album)
346
347 if MediaType.ARTIST in media_types and creator:
348 artist = create_artist(creator, self.domain, self.instance_id)
349 if artist and not artist_exists(artist, artists):
350 artists.append(artist)
351
352 elif mediatype == "audio":
353 # Use heuristics to determine album vs track without expensive API calls
354 if is_likely_album(doc):
355 if MediaType.ALBUM in media_types:
356 album = doc_to_album(
357 doc, self.domain, self.instance_id, self.client.get_item_url
358 )
359 if album:
360 albums.append(album)
361 elif MediaType.TRACK in media_types:
362 track = doc_to_track(doc, self.domain, self.instance_id, self.client.get_item_url)
363 if track:
364 tracks.append(track)
365
366 if MediaType.ARTIST in media_types and creator:
367 artist = create_artist(creator, self.domain, self.instance_id)
368 if artist and not artist_exists(artist, artists):
369 artists.append(artist)
370
371 @use_cache(expiration=86400 * 60) # Cache for 60 days - artist "tracks" change infrequently
372 async def get_track(self, prov_track_id: str) -> Track:
373 """Get full track details by id."""
374 metadata = await self._get_metadata(prov_track_id)
375 item_metadata = metadata.get("metadata", {})
376
377 title = clean_text(item_metadata.get("title"))
378 creator = clean_text(item_metadata.get("creator"))
379
380 if not title:
381 raise MediaNotFoundError(f"Track {prov_track_id} not found or invalid")
382
383 track = Track(
384 item_id=prov_track_id,
385 provider=self.instance_id,
386 name=title,
387 provider_mappings={
388 create_provider_mapping(
389 prov_track_id, self.domain, self.instance_id, self.client.get_item_url
390 )
391 },
392 )
393
394 # Add artist
395 if creator:
396 track.artists = UniqueList([create_artist(creator, self.domain, self.instance_id)])
397 else:
398 track.artists = UniqueList(
399 [create_artist(UNKNOWN_ARTIST, self.domain, self.instance_id)]
400 )
401
402 # Add duration from first audio file
403 try:
404 audio_files = await self._get_audio_files(prov_track_id)
405 if audio_files and audio_files[0].get("length"):
406 duration = parse_duration(audio_files[0]["length"])
407 if duration:
408 track.duration = duration
409 except (TimeoutError, aiohttp.ClientError) as err:
410 self.logger.debug("Network error getting duration for track %s: %s", prov_track_id, err)
411 except (KeyError, ValueError, TypeError) as err:
412 self.logger.debug("Could not parse duration for track %s: %s", prov_track_id, err)
413
414 # Add metadata
415 if description := clean_text(item_metadata.get("description")):
416 track.metadata.description = description
417
418 # Add thumbnail
419 add_item_image(track, prov_track_id, self.instance_id)
420
421 return track
422
423 @use_cache(expiration=86400 * 60) # Cache for 60 days - album catalogs change infrequently
424 async def get_album(self, prov_album_id: str) -> Album:
425 """Get full album details by id."""
426 metadata = await self._get_metadata(prov_album_id)
427 item_metadata = metadata.get("metadata", {})
428
429 title = clean_text(item_metadata.get("title"))
430 creator = clean_text(item_metadata.get("creator"))
431
432 if not title:
433 raise MediaNotFoundError(f"Album {prov_album_id} not found or invalid")
434
435 album = Album(
436 item_id=prov_album_id,
437 provider=self.instance_id,
438 name=title,
439 provider_mappings={
440 create_provider_mapping(
441 prov_album_id, self.domain, self.instance_id, self.client.get_item_url
442 )
443 },
444 )
445
446 # Add artist
447 if creator:
448 album.artists = UniqueList([create_artist(creator, self.domain, self.instance_id)])
449 else:
450 album.artists = UniqueList(
451 [create_artist(UNKNOWN_ARTIST, self.domain, self.instance_id)]
452 )
453
454 # Add metadata
455 if date := extract_year(item_metadata.get("date")):
456 album.year = date
457
458 if description := clean_text(item_metadata.get("description")):
459 album.metadata.description = description
460
461 # Add thumbnail
462 add_item_image(album, prov_album_id, self.instance_id)
463
464 return album
465
466 @use_cache(expiration=86400 * 60) # Cache for 60 days - artist catalogs change infrequently
467 async def get_artist(self, prov_artist_id: str) -> Artist:
468 """
469 Get full artist details by id.
470
471 Args:
472 prov_artist_id: Provider-specific artist identifier (artist name)
473
474 Returns:
475 Artist object
476 """
477 # Artist IDs are just the creator names
478 return Artist(
479 item_id=prov_artist_id,
480 provider=self.instance_id,
481 name=prov_artist_id,
482 provider_mappings={
483 ProviderMapping(
484 item_id=prov_artist_id,
485 provider_domain=self.domain,
486 provider_instance=self.instance_id,
487 )
488 },
489 )
490
491 @use_cache(expiration=86400 * 30) # Cache for 30 days - audiobook catalogs change infrequently
492 async def get_audiobook(self, prov_audiobook_id: str) -> Audiobook:
493 """Get full audiobook details by id."""
494 metadata = await self._get_metadata(prov_audiobook_id)
495 item_metadata = metadata.get("metadata", {})
496
497 title = clean_text(item_metadata.get("title"))
498 creator = clean_text(item_metadata.get("creator"))
499
500 if not title:
501 raise MediaNotFoundError(f"Audiobook {prov_audiobook_id} not found or invalid")
502
503 audiobook = Audiobook(
504 item_id=prov_audiobook_id,
505 provider=self.instance_id,
506 name=title,
507 provider_mappings={
508 create_provider_mapping(
509 prov_audiobook_id, self.domain, self.instance_id, self.client.get_item_url
510 )
511 },
512 )
513
514 # Add author/narrator
515 if creator:
516 author_list = [creator]
517 audiobook.authors = UniqueList(author_list)
518
519 # Add metadata
520 if description := clean_text(item_metadata.get("description")):
521 audiobook.metadata.description = description
522
523 # Add thumbnail
524 add_item_image(audiobook, prov_audiobook_id, self.instance_id)
525
526 # Calculate duration and chapters
527 try:
528 total_duration, chapters = await self._calculate_audiobook_duration_and_chapters(
529 prov_audiobook_id
530 )
531 audiobook.duration = total_duration
532 if len(chapters) > 1:
533 audiobook.metadata.chapters = chapters
534
535 except Exception as err:
536 self.logger.warning(
537 f"Could not process audio files for audiobook {prov_audiobook_id}: {err}"
538 )
539 audiobook.duration = 0
540 audiobook.metadata.chapters = []
541
542 return audiobook
543
544 async def get_album_tracks(self, prov_album_id: str) -> list[Track]:
545 """Get album tracks for given album id."""
546 metadata = await self._get_metadata(prov_album_id)
547 item_metadata = metadata.get("metadata", {})
548 audio_files = await self._get_audio_files(prov_album_id)
549 tracks = []
550
551 # Pre-create album artist to avoid duplicates
552 album_artist = clean_text(item_metadata.get("creator"))
553 album_artist_normalized = album_artist.lower() if album_artist else ""
554 album_artist_obj = None
555 if album_artist:
556 album_artist_obj = create_artist(album_artist, self.domain, self.instance_id)
557 else:
558 album_artist_obj = create_artist(UNKNOWN_ARTIST, self.domain, self.instance_id)
559
560 for i, file_info in enumerate(audio_files, 1):
561 filename = file_info.get("name", "")
562
563 # Use file's title if available, otherwise clean up filename
564 track_name = file_info.get("title", filename)
565 if not track_name or track_name == filename:
566 track_name = filename.rsplit(".", 1)[0] if "." in filename else filename
567
568 # Try to extract track number from file metadata first, then filename
569 track_number = self._extract_track_number(file_info, track_name, i)
570
571 track = Track(
572 item_id=f"{prov_album_id}#{filename}",
573 provider=self.instance_id,
574 name=track_name,
575 track_number=track_number,
576 provider_mappings={
577 ProviderMapping(
578 item_id=f"{prov_album_id}#{filename}",
579 provider_domain=self.domain,
580 provider_instance=self.instance_id,
581 url=self.client.get_download_url(prov_album_id, filename),
582 available=True,
583 )
584 },
585 )
586
587 # Add file-specific artist if available, otherwise use album artist
588 file_artist = file_info.get("artist") or file_info.get("creator")
589 if file_artist:
590 file_artist_cleaned = clean_text(file_artist)
591 file_artist_normalized = file_artist_cleaned.lower()
592 # Check if this is the same as album artist to avoid duplicates (case-insensitive)
593 if album_artist_normalized and file_artist_normalized == album_artist_normalized:
594 track.artists = UniqueList([album_artist_obj])
595 else:
596 track.artists = UniqueList(
597 [create_artist(file_artist_cleaned, self.domain, self.instance_id)]
598 )
599 else:
600 # Use pre-created album artist object
601 track.artists = UniqueList([album_artist_obj])
602
603 # Add duration if available
604 if duration_str := file_info.get("length"):
605 if duration := parse_duration(duration_str):
606 track.duration = duration
607
608 # Add genre if available
609 if genre := file_info.get("genre"):
610 track.metadata.genres = {clean_text(genre)}
611
612 tracks.append(track)
613
614 return tracks
615
616 def _extract_track_number(
617 self, file_info: dict[str, Any], track_name: str, fallback: int
618 ) -> int:
619 """Extract track number from file metadata or filename."""
620 track_number = None
621
622 if "track" in file_info:
623 with contextlib.suppress(ValueError, AttributeError):
624 track_number = int(str(file_info["track"]).split("/")[0])
625
626 if track_number is None:
627 # Fallback to filename parsing
628 track_num_match = re.search(r"^(\d+)[\s\-_.]*(.+)", track_name)
629 track_number = int(track_num_match.group(1)) if track_num_match else fallback
630
631 return track_number
632
633 @use_cache(
634 expiration=86400 * 30, allow_expired_cache=True
635 ) # Cache for 30 days - artist catalogs change infrequently
636 async def get_artist_albums(self, prov_artist_id: str) -> list[Album]:
637 """
638 Get albums for a specific artist.
639
640 Uses metadata heuristics to determine likely albums without expensive
641 API calls for better performance.
642
643 Args:
644 prov_artist_id: Provider-specific artist identifier (artist name)
645
646 Returns:
647 List of Album objects by the artist
648 """
649 albums: list[Album] = []
650 page = 0
651 page_size = 200 # IA's maximum
652
653 while len(albums) < 1000: # Reasonable upper limit
654 search_response = await self._search(
655 query=f'creator:"{prov_artist_id}" AND (format:"VBR MP3" OR format:"FLAC" \
656 OR format:"Ogg Vorbis")',
657 sort="downloads desc",
658 rows=page_size,
659 page=page,
660 )
661
662 docs = search_response.get("response", {}).get("docs", [])
663 if not docs:
664 break
665
666 for doc in docs:
667 try:
668 # Use metadata heuristics instead of expensive API calls
669 # to determine if item is an album
670 if is_likely_album(doc):
671 album = doc_to_album(
672 doc, self.domain, self.instance_id, self.client.get_item_url
673 )
674 if album:
675 albums.append(album)
676 except (KeyError, ValueError, TypeError) as err:
677 self.logger.debug(
678 "Skipping invalid album for artist %s: %s", prov_artist_id, err
679 )
680 continue
681 except (TimeoutError, aiohttp.ClientError) as err:
682 self.logger.debug(
683 "Network error processing album for artist %s: %s", prov_artist_id, err
684 )
685 continue
686 except Exception as err:
687 self.logger.exception(
688 "Unexpected error processing album for artist %s: %s", prov_artist_id, err
689 )
690 continue
691 page += 1
692 return albums
693
694 @use_cache(expiration=86400 * 7, allow_expired_cache=True) # Cache for 1 week
695 async def get_artist_toptracks(self, prov_artist_id: str) -> list[Track]:
696 """
697 Get top tracks for a specific artist.
698
699 Uses the same search as get_artist_albums but filters for single tracks.
700
701 Args:
702 prov_artist_id: Provider-specific artist identifier (artist name)
703
704 Returns:
705 List of Track objects representing the artist's top tracks
706 """
707 tracks = []
708 search_response = await self._search(
709 query=(
710 f'creator:"{prov_artist_id}" AND '
711 f'(format:"VBR MP3" OR format:"FLAC" OR format:"Ogg Vorbis")'
712 ),
713 rows=25, # Limit for "top" tracks
714 sort="downloads desc",
715 )
716
717 response_data = search_response.get("response", {})
718 docs = response_data.get("docs", [])
719
720 for doc in docs:
721 try:
722 # Only include items that are NOT classified as albums
723 if not is_likely_album(doc):
724 track = doc_to_track(
725 doc, self.domain, self.instance_id, self.client.get_item_url
726 )
727 if track:
728 tracks.append(track)
729 except (KeyError, ValueError, TypeError) as err:
730 self.logger.debug("Skipping invalid track for artist %s: %s", prov_artist_id, err)
731 continue
732 except (TimeoutError, aiohttp.ClientError) as err:
733 self.logger.debug(
734 "Network error processing track for artist %s: %s", prov_artist_id, err
735 )
736 continue
737 except Exception as err:
738 self.logger.exception(
739 "Unexpected error processing track for artist %s: %s", prov_artist_id, err
740 )
741 continue
742
743 if len(tracks) >= 25:
744 break
745
746 return tracks
747
748 async def get_stream_details(self, item_id: str, media_type: MediaType) -> StreamDetails:
749 """
750 Get streamdetails for a track or audiobook.
751
752 Delegates to the streaming handler for proper multi-file support.
753
754 Args:
755 item_id: Provider-specific item identifier
756 media_type: The type of media being requested
757
758 Returns:
759 StreamDetails object configured for the specific item type
760
761 Raises:
762 MediaNotFoundError: If no audio files are found for the item
763 """
764 return await self.streaming.get_stream_details(item_id, media_type)
765
766 async def _calculate_audiobook_duration_and_chapters(
767 self, item_id: str
768 ) -> tuple[int, list[MediaItemChapter]]:
769 """Calculate duration and chapters for audiobooks."""
770 audio_files = await self._get_audio_files(item_id)
771 total_duration = 0
772 chapters = []
773 current_position = 0.0
774
775 for i, file_info in enumerate(audio_files, 1):
776 chapter_duration = parse_duration(file_info.get("length", "0")) or 0
777 total_duration += chapter_duration
778
779 chapter_name = file_info.get("title") or file_info.get("name", f"Chapter {i}")
780 chapter = MediaItemChapter(
781 position=i,
782 name=clean_text(chapter_name),
783 start=current_position,
784 end=current_position + chapter_duration if chapter_duration > 0 else None,
785 )
786 chapters.append(chapter)
787 current_position += chapter_duration
788
789 return total_duration, chapters
790
791 async def get_audio_stream(
792 self, streamdetails: StreamDetails, seek_position: int = 0
793 ) -> AsyncGenerator[bytes]:
794 """Get audio stream from Internet Archive."""
795 # Use sock_read=None to allow long audiobook chapters to stream fully
796 timeout = aiohttp.ClientTimeout(sock_read=None, total=None)
797
798 if streamdetails.media_type == MediaType.AUDIOBOOK and isinstance(streamdetails.data, dict):
799 chapter_urls = streamdetails.data.get("chapters", [])
800 chapters_data = streamdetails.data.get("chapters_data", [])
801
802 # Calculate which chapter to start from based on seek_position
803 seek_position_ms = seek_position * 1000
804 start_chapter = 0
805
806 if seek_position > 0 and chapters_data:
807 accumulated_duration_ms = 0
808
809 for i, chapter_data in enumerate(chapters_data):
810 chapter_duration_ms = (
811 parse_duration(chapter_data.get("length", "0")) or 0
812 ) * 1000
813
814 if accumulated_duration_ms + chapter_duration_ms > seek_position_ms:
815 start_chapter = i
816 break
817 accumulated_duration_ms += chapter_duration_ms
818
819 # Stream chapters starting from calculated position
820 chapters_yielded = False
821 for i in range(start_chapter, len(chapter_urls)):
822 chapter_url = chapter_urls[i]
823
824 try:
825 async with self.mass.http_session.get(chapter_url, timeout=timeout) as response:
826 response.raise_for_status()
827 async for chunk in response.content.iter_chunked(8192):
828 chapters_yielded = True
829 yield chunk
830 except Exception as e:
831 self.logger.error(f"Chapter {i + 1} streaming failed: {e}")
832 continue
833
834 # If no chapters succeeded, raise an error instead of silent failure
835 if not chapters_yielded:
836 raise MediaNotFoundError(
837 f"Failed to stream any chapters for audiobook {streamdetails.item_id}"
838 )
839
840 else:
841 # Handle single files
842 audio_files = await self._get_audio_files(streamdetails.item_id)
843 if audio_files:
844 download_url = self.client.get_download_url(
845 streamdetails.item_id, audio_files[0]["name"]
846 )
847 async with self.mass.http_session.get(download_url, timeout=timeout) as response:
848 response.raise_for_status()
849 async for chunk in response.content.iter_chunked(8192):
850 yield chunk
851
852 @use_cache(expiration=86400 * 7) # Cache for 1 week
853 async def get_podcast(self, prov_podcast_id: str) -> Podcast:
854 """Get full podcast details by id."""
855 metadata = await self._get_metadata(prov_podcast_id)
856 item_metadata = metadata.get("metadata", {})
857
858 title = clean_text(item_metadata.get("title"))
859 creator = clean_text(item_metadata.get("creator"))
860
861 if not title:
862 raise MediaNotFoundError(f"Podcast {prov_podcast_id} not found or invalid")
863
864 podcast = Podcast(
865 item_id=prov_podcast_id,
866 provider=self.instance_id,
867 name=title,
868 provider_mappings={
869 create_provider_mapping(
870 prov_podcast_id, self.domain, self.instance_id, self.client.get_item_url
871 )
872 },
873 )
874
875 # Add publisher/creator
876 if creator:
877 podcast.publisher = creator
878
879 # Add metadata
880 if description := clean_text(item_metadata.get("description")):
881 podcast.metadata.description = description
882
883 # Add thumbnail
884 add_item_image(podcast, prov_podcast_id, self.instance_id)
885
886 # Calculate total episodes
887 try:
888 audio_files = await self._get_audio_files(prov_podcast_id)
889 podcast.total_episodes = len(audio_files)
890 except Exception as err:
891 self.logger.warning(f"Could not get episode count for podcast {prov_podcast_id}: {err}")
892 podcast.total_episodes = None
893
894 return podcast
895
896 async def get_podcast_episodes(self, prov_podcast_id: str) -> AsyncGenerator[PodcastEpisode]:
897 """Get podcast episodes for given podcast id."""
898 metadata = await self._get_metadata(prov_podcast_id)
899 item_metadata = metadata.get("metadata", {})
900 audio_files = await self._get_audio_files(prov_podcast_id)
901
902 # Create podcast reference for episodes
903 podcast = Podcast(
904 item_id=prov_podcast_id,
905 provider=self.instance_id,
906 name=clean_text(item_metadata.get("title", prov_podcast_id)),
907 provider_mappings={
908 create_provider_mapping(
909 prov_podcast_id, self.domain, self.instance_id, self.client.get_item_url
910 )
911 },
912 )
913
914 for i, file_info in enumerate(audio_files, 1):
915 filename = file_info.get("name", "")
916
917 # Use file's title if available, otherwise clean up filename
918 episode_name = file_info.get("title", filename)
919 if not episode_name or episode_name == filename:
920 episode_name = filename.rsplit(".", 1)[0] if "." in filename else filename
921
922 # Try to extract episode number from file metadata first, then filename
923 episode_number = self._extract_track_number(file_info, episode_name, i)
924
925 episode = PodcastEpisode(
926 item_id=f"{prov_podcast_id}#{filename}",
927 provider=self.instance_id,
928 name=episode_name,
929 position=episode_number,
930 podcast=podcast,
931 provider_mappings={
932 ProviderMapping(
933 item_id=f"{prov_podcast_id}#{filename}",
934 provider_domain=self.domain,
935 provider_instance=self.instance_id,
936 url=self.client.get_download_url(prov_podcast_id, filename),
937 available=True,
938 )
939 },
940 )
941
942 # Add duration if available
943 if duration_str := file_info.get("length"):
944 if duration := parse_duration(duration_str):
945 episode.duration = duration
946
947 # Add episode metadata
948 if description := file_info.get("description"):
949 episode.metadata.description = clean_text(description)
950
951 yield episode
952
953 async def get_podcast_episode(self, prov_episode_id: str) -> PodcastEpisode:
954 """Get single podcast episode by id."""
955 if "#" not in prov_episode_id:
956 raise MediaNotFoundError(f"Invalid episode ID format: {prov_episode_id}")
957
958 podcast_id, _ = prov_episode_id.split("#", 1)
959
960 async for episode in self.get_podcast_episodes(podcast_id):
961 if episode.item_id == prov_episode_id:
962 return episode
963
964 raise MediaNotFoundError(f"Episode {prov_episode_id} not found")
965