/
/
1"""Internet Archive music provider implementation."""
2
3from __future__ import annotations
4
5import contextlib
6import re
7from collections.abc import AsyncGenerator
8from typing import TYPE_CHECKING, Any
9
10import aiohttp
11from music_assistant_models.enums import MediaType, ProviderFeature
12from music_assistant_models.errors import InvalidDataError, MediaNotFoundError
13from music_assistant_models.media_items import (
14 Album,
15 Artist,
16 Audiobook,
17 MediaItemChapter,
18 Podcast,
19 PodcastEpisode,
20 ProviderMapping,
21 SearchResults,
22 Track,
23)
24from music_assistant_models.unique_list import UniqueList
25
26from music_assistant.constants import UNKNOWN_ARTIST
27from music_assistant.controllers.cache import use_cache
28from music_assistant.helpers.throttle_retry import ThrottlerManager, throttle_with_retries
29from music_assistant.models.music_provider import MusicProvider
30
31from .helpers import InternetArchiveClient, clean_text, extract_year, parse_duration
32from .parsers import (
33 add_item_image,
34 artist_exists,
35 create_artist,
36 create_provider_mapping,
37 create_title_from_identifier,
38 doc_to_album,
39 doc_to_audiobook,
40 doc_to_podcast,
41 doc_to_track,
42 is_audiobook_content,
43 is_likely_album,
44 is_podcast_content,
45)
46from .streaming import InternetArchiveStreaming
47
48if TYPE_CHECKING:
49 from music_assistant_models.config_entries import ConfigEntry, ProviderConfig
50 from music_assistant_models.provider import ProviderManifest
51 from music_assistant_models.streamdetails import StreamDetails
52
53 from music_assistant import MusicAssistant
54
55
56class InternetArchiveProvider(MusicProvider):
57 """Implementation of Internet Archive music provider."""
58
59 def __init__(
60 self,
61 mass: MusicAssistant,
62 manifest: ProviderManifest,
63 config: ProviderConfig,
64 supported_features: set[ProviderFeature],
65 ) -> None:
66 """Initialize the provider."""
67 super().__init__(mass, manifest, config, supported_features)
68 self.throttler = ThrottlerManager(
69 rate_limit=10, period=60, retry_attempts=5, initial_backoff=5
70 )
71 self.client = InternetArchiveClient(mass)
72 self.streaming = InternetArchiveStreaming(self)
73
74 async def get_config_entries(self) -> tuple[ConfigEntry, ...]:
75 """Return Config entries to configure this provider."""
76 return ()
77
78 @property
79 def is_streaming_provider(self) -> bool:
80 """Return True if provider is a streaming provider."""
81 return True
82
83 @throttle_with_retries
84 async def _get_json(self, url: str, params: dict[str, Any] | None = None) -> dict[str, Any]:
85 """Make a GET request and return JSON response with throttling."""
86 return await self.client._get_json(url, params)
87
88 @throttle_with_retries
89 async def _search(self, **kwargs: Any) -> dict[str, Any]:
90 """Throttled search wrapper."""
91 return await self.client.search(**kwargs)
92
93 @throttle_with_retries
94 async def _get_metadata(self, identifier: str) -> dict[str, Any]:
95 """Throttled metadata wrapper."""
96 return await self.client.get_metadata(identifier)
97
98 @use_cache(expiration=86400 * 30) # 30 days - file listings are static
99 @throttle_with_retries
100 async def _get_audio_files(self, identifier: str) -> list[dict[str, Any]]:
101 """Throttled audio files wrapper."""
102 return await self.client.get_audio_files(identifier)
103
104 @use_cache(86400 * 7) # 7 days
105 async def search(
106 self,
107 search_query: str,
108 media_types: list[MediaType],
109 limit: int = 5,
110 ) -> SearchResults:
111 """
112 Perform search on Internet Archive.
113
114 Uses multiple search strategies to maximize result coverage with
115 proper result accumulation and broader search patterns.
116
117 Args:
118 search_query: The search term to look for
119 media_types: List of media types to search for
120 limit: Maximum number of results to return per media type
121
122 Returns:
123 SearchResults object containing found items
124 """
125 if not search_query.strip():
126 return SearchResults()
127
128 # Adjust search intensity based on what's being requested
129 rows_per_strategy = min(limit * 2, 16) if len(media_types) > 1 else min(limit * 2, 100)
130
131 # Collect results in separate lists
132 tracks: list[Track] = []
133 albums: list[Album] = []
134 artists: list[Artist] = []
135 audiobooks: list[Audiobook] = []
136 podcasts: list[Podcast] = []
137
138 # Track processed identifiers to avoid duplicates across strategies
139 processed_ids: set[str] = set()
140
141 # Build search strategies based on requested media types
142 search_strategies = []
143
144 # For music searches: focus on title and creator
145 if any(mt in media_types for mt in [MediaType.TRACK, MediaType.ALBUM, MediaType.ARTIST]):
146 search_strategies.extend(
147 [
148 (f"creator:({search_query}) AND mediatype:audio", "downloads desc"),
149 (f"title:({search_query}) AND mediatype:audio", "downloads desc"),
150 (f"subject:({search_query}) AND mediatype:audio", "downloads desc"),
151 ]
152 )
153
154 # For audiobooks: search within audiobook collections, still limit to audio
155 if MediaType.AUDIOBOOK in media_types:
156 audiobook_query = f"{search_query} AND collection:(librivoxaudio OR audio_bookspoetry) AND mediatype:audio"
157 search_strategies.append((audiobook_query, "downloads desc"))
158
159 # For podcasts: search within podcast collections
160 if MediaType.PODCAST in media_types:
161 podcast_query = f"{search_query} AND collection:podcasts AND mediatype:audio"
162 search_strategies.append((podcast_query, "downloads desc"))
163
164 for strategy_idx, (strategy_query, sort_order) in enumerate(search_strategies):
165 self.logger.debug("Trying search strategy %d: %s", strategy_idx + 1, strategy_query)
166
167 try:
168 search_response = await self._search(
169 query=strategy_query,
170 rows=rows_per_strategy,
171 sort=sort_order,
172 )
173
174 response_data = search_response.get("response", {})
175 docs = response_data.get("docs", [])
176 self.logger.debug(
177 "Strategy %d '%s' found %d raw results",
178 strategy_idx + 1,
179 strategy_query,
180 len(docs),
181 )
182
183 # Process results and extract different media types
184 strategy_processed = 0
185 strategy_skipped = 0
186
187 for doc in docs:
188 try:
189 identifier = doc.get("identifier")
190 if not identifier or identifier in processed_ids:
191 strategy_skipped += 1
192 continue
193
194 # Track this identifier to avoid duplicates
195 processed_ids.add(identifier)
196
197 await self._process_search_result(
198 doc, tracks, albums, artists, audiobooks, podcasts, media_types
199 )
200 strategy_processed += 1
201
202 # Check if we have enough results across all types
203 if self._has_sufficient_results(
204 tracks, albums, artists, audiobooks, podcasts, media_types, limit
205 ):
206 self.logger.debug(
207 "Sufficient results found after strategy %d, stopping search",
208 strategy_idx + 1,
209 )
210 break
211
212 except (InvalidDataError, KeyError) as err:
213 self.logger.debug("Skipping invalid search result: %s", err)
214 strategy_skipped += 1
215 continue
216
217 self.logger.debug(
218 "Strategy %d '%s': processed %d new items, skipped %d items. "
219 "Running totals - tracks: %d, albums: %d, artists: %d, "
220 "audiobooks: %d, podcasts: %d",
221 strategy_idx + 1,
222 strategy_query,
223 strategy_processed,
224 strategy_skipped,
225 len(tracks),
226 len(albums),
227 len(artists),
228 len(audiobooks),
229 len(podcasts),
230 )
231
232 # If we have sufficient results, stop trying more strategies
233 if self._has_sufficient_results(
234 tracks, albums, artists, audiobooks, podcasts, media_types, limit
235 ):
236 break
237
238 except Exception as err:
239 self.logger.warning("Search strategy %d failed: %s", strategy_idx + 1, err)
240 continue
241
242 # Log final results for debugging
243 self.logger.debug(
244 "Search for '%s' completed. Final results - tracks: %d, albums: %d, "
245 "artists: %d, audiobooks: %d, podcasts: %d (processed %d unique items)",
246 search_query,
247 len(tracks),
248 len(albums),
249 len(artists),
250 len(audiobooks),
251 len(podcasts),
252 len(processed_ids),
253 )
254
255 return SearchResults(
256 tracks=tracks[:limit] if MediaType.TRACK in media_types else [],
257 albums=albums[:limit] if MediaType.ALBUM in media_types else [],
258 artists=artists[:limit] if MediaType.ARTIST in media_types else [],
259 audiobooks=audiobooks[:limit] if MediaType.AUDIOBOOK in media_types else [],
260 podcasts=podcasts[:limit] if MediaType.PODCAST in media_types else [],
261 )
262
263 def _has_sufficient_results(
264 self,
265 tracks: list[Track],
266 albums: list[Album],
267 artists: list[Artist],
268 audiobooks: list[Audiobook],
269 podcasts: list[Podcast],
270 media_types: list[MediaType],
271 limit: int,
272 ) -> bool:
273 """Check if we have sufficient results for all requested media types."""
274 return (
275 (MediaType.TRACK not in media_types or len(tracks) >= limit)
276 and (MediaType.ALBUM not in media_types or len(albums) >= limit)
277 and (MediaType.ARTIST not in media_types or len(artists) >= limit)
278 and (MediaType.AUDIOBOOK not in media_types or len(audiobooks) >= limit)
279 and (MediaType.PODCAST not in media_types or len(podcasts) >= limit)
280 )
281
282 async def _process_search_result(
283 self,
284 doc: dict[str, Any],
285 tracks: list[Track],
286 albums: list[Album],
287 artists: list[Artist],
288 audiobooks: list[Audiobook],
289 podcasts: list[Podcast],
290 media_types: list[MediaType],
291 ) -> None:
292 """
293 Process a single search result document from Internet Archive.
294
295 Determines the appropriate media type and creates corresponding objects.
296 Uses improved heuristics to classify items as tracks, albums, or audiobooks.
297 """
298 identifier = doc.get("identifier")
299 if not identifier:
300 raise InvalidDataError("Missing identifier in search result")
301
302 title = clean_text(doc.get("title"))
303 creator = clean_text(doc.get("creator"))
304
305 # Be lenient - allow items without title if they have identifier
306 if not title and not identifier:
307 raise InvalidDataError("Missing both title and identifier in search result")
308
309 # Use identifier as fallback title if needed
310 if not title:
311 title = create_title_from_identifier(identifier)
312
313 # Determine what type of item this is
314 mediatype = doc.get("mediatype", "")
315 collection = doc.get("collection", [])
316 if isinstance(collection, str):
317 collection = [collection]
318
319 # Check if this is audiobook content using improved detection
320 if is_audiobook_content(doc) and MediaType.AUDIOBOOK in media_types:
321 audiobook = doc_to_audiobook(
322 doc, self.domain, self.instance_id, self.client.get_item_url
323 )
324 if audiobook:
325 audiobooks.append(audiobook)
326 return # Don't process as other media types
327
328 # Check if this is podcast content
329 if is_podcast_content(doc) and MediaType.PODCAST in media_types:
330 podcast = doc_to_podcast(doc, self.domain, self.instance_id, self.client.get_item_url)
331 if podcast:
332 podcasts.append(podcast)
333 return # Don't process as other media types
334
335 # For etree items, usually each item is an album (concert)
336 if mediatype == "etree" or "etree" in collection:
337 if MediaType.ALBUM in media_types:
338 album = doc_to_album(doc, self.domain, self.instance_id, self.client.get_item_url)
339 if album:
340 albums.append(album)
341
342 if MediaType.ARTIST in media_types and creator:
343 artist = create_artist(creator, self.domain, self.instance_id)
344 if artist and not artist_exists(artist, artists):
345 artists.append(artist)
346
347 elif mediatype == "audio":
348 # Use heuristics to determine album vs track without expensive API calls
349 if is_likely_album(doc):
350 if MediaType.ALBUM in media_types:
351 album = doc_to_album(
352 doc, self.domain, self.instance_id, self.client.get_item_url
353 )
354 if album:
355 albums.append(album)
356 elif MediaType.TRACK in media_types:
357 track = doc_to_track(doc, self.domain, self.instance_id, self.client.get_item_url)
358 if track:
359 tracks.append(track)
360
361 if MediaType.ARTIST in media_types and creator:
362 artist = create_artist(creator, self.domain, self.instance_id)
363 if artist and not artist_exists(artist, artists):
364 artists.append(artist)
365
366 @use_cache(expiration=86400 * 60) # Cache for 60 days - artist "tracks" change infrequently
367 async def get_track(self, prov_track_id: str) -> Track:
368 """Get full track details by id."""
369 metadata = await self._get_metadata(prov_track_id)
370 item_metadata = metadata.get("metadata", {})
371
372 title = clean_text(item_metadata.get("title"))
373 creator = clean_text(item_metadata.get("creator"))
374
375 if not title:
376 raise MediaNotFoundError(f"Track {prov_track_id} not found or invalid")
377
378 track = Track(
379 item_id=prov_track_id,
380 provider=self.instance_id,
381 name=title,
382 provider_mappings={
383 create_provider_mapping(
384 prov_track_id, self.domain, self.instance_id, self.client.get_item_url
385 )
386 },
387 )
388
389 # Add artist
390 if creator:
391 track.artists = UniqueList([create_artist(creator, self.domain, self.instance_id)])
392 else:
393 track.artists = UniqueList(
394 [create_artist(UNKNOWN_ARTIST, self.domain, self.instance_id)]
395 )
396
397 # Add duration from first audio file
398 try:
399 audio_files = await self._get_audio_files(prov_track_id)
400 if audio_files and audio_files[0].get("length"):
401 duration = parse_duration(audio_files[0]["length"])
402 if duration:
403 track.duration = duration
404 except (TimeoutError, aiohttp.ClientError) as err:
405 self.logger.debug("Network error getting duration for track %s: %s", prov_track_id, err)
406 except (KeyError, ValueError, TypeError) as err:
407 self.logger.debug("Could not parse duration for track %s: %s", prov_track_id, err)
408
409 # Add metadata
410 if description := clean_text(item_metadata.get("description")):
411 track.metadata.description = description
412
413 # Add thumbnail
414 add_item_image(track, prov_track_id, self.instance_id)
415
416 return track
417
418 @use_cache(expiration=86400 * 60) # Cache for 60 days - album catalogs change infrequently
419 async def get_album(self, prov_album_id: str) -> Album:
420 """Get full album details by id."""
421 metadata = await self._get_metadata(prov_album_id)
422 item_metadata = metadata.get("metadata", {})
423
424 title = clean_text(item_metadata.get("title"))
425 creator = clean_text(item_metadata.get("creator"))
426
427 if not title:
428 raise MediaNotFoundError(f"Album {prov_album_id} not found or invalid")
429
430 album = Album(
431 item_id=prov_album_id,
432 provider=self.instance_id,
433 name=title,
434 provider_mappings={
435 create_provider_mapping(
436 prov_album_id, self.domain, self.instance_id, self.client.get_item_url
437 )
438 },
439 )
440
441 # Add artist
442 if creator:
443 album.artists = UniqueList([create_artist(creator, self.domain, self.instance_id)])
444 else:
445 album.artists = UniqueList(
446 [create_artist(UNKNOWN_ARTIST, self.domain, self.instance_id)]
447 )
448
449 # Add metadata
450 if date := extract_year(item_metadata.get("date")):
451 album.year = date
452
453 if description := clean_text(item_metadata.get("description")):
454 album.metadata.description = description
455
456 # Add thumbnail
457 add_item_image(album, prov_album_id, self.instance_id)
458
459 return album
460
461 @use_cache(expiration=86400 * 60) # Cache for 60 days - artist catalogs change infrequently
462 async def get_artist(self, prov_artist_id: str) -> Artist:
463 """
464 Get full artist details by id.
465
466 Args:
467 prov_artist_id: Provider-specific artist identifier (artist name)
468
469 Returns:
470 Artist object
471 """
472 # Artist IDs are just the creator names
473 return Artist(
474 item_id=prov_artist_id,
475 provider=self.instance_id,
476 name=prov_artist_id,
477 provider_mappings={
478 ProviderMapping(
479 item_id=prov_artist_id,
480 provider_domain=self.domain,
481 provider_instance=self.instance_id,
482 )
483 },
484 )
485
486 @use_cache(expiration=86400 * 30) # Cache for 30 days - audiobook catalogs change infrequently
487 async def get_audiobook(self, prov_audiobook_id: str) -> Audiobook:
488 """Get full audiobook details by id."""
489 metadata = await self._get_metadata(prov_audiobook_id)
490 item_metadata = metadata.get("metadata", {})
491
492 title = clean_text(item_metadata.get("title"))
493 creator = clean_text(item_metadata.get("creator"))
494
495 if not title:
496 raise MediaNotFoundError(f"Audiobook {prov_audiobook_id} not found or invalid")
497
498 audiobook = Audiobook(
499 item_id=prov_audiobook_id,
500 provider=self.instance_id,
501 name=title,
502 provider_mappings={
503 create_provider_mapping(
504 prov_audiobook_id, self.domain, self.instance_id, self.client.get_item_url
505 )
506 },
507 )
508
509 # Add author/narrator
510 if creator:
511 author_list = [creator]
512 audiobook.authors = UniqueList(author_list)
513
514 # Add metadata
515 if description := clean_text(item_metadata.get("description")):
516 audiobook.metadata.description = description
517
518 # Add thumbnail
519 add_item_image(audiobook, prov_audiobook_id, self.instance_id)
520
521 # Calculate duration and chapters
522 try:
523 total_duration, chapters = await self._calculate_audiobook_duration_and_chapters(
524 prov_audiobook_id
525 )
526 audiobook.duration = total_duration
527 if len(chapters) > 1:
528 audiobook.metadata.chapters = chapters
529
530 except Exception as err:
531 self.logger.warning(
532 f"Could not process audio files for audiobook {prov_audiobook_id}: {err}"
533 )
534 audiobook.duration = 0
535 audiobook.metadata.chapters = []
536
537 return audiobook
538
539 async def get_album_tracks(self, prov_album_id: str) -> list[Track]:
540 """Get album tracks for given album id."""
541 metadata = await self._get_metadata(prov_album_id)
542 item_metadata = metadata.get("metadata", {})
543 audio_files = await self._get_audio_files(prov_album_id)
544 tracks = []
545
546 # Pre-create album artist to avoid duplicates
547 album_artist = clean_text(item_metadata.get("creator"))
548 album_artist_normalized = album_artist.lower() if album_artist else ""
549 album_artist_obj = None
550 if album_artist:
551 album_artist_obj = create_artist(album_artist, self.domain, self.instance_id)
552 else:
553 album_artist_obj = create_artist(UNKNOWN_ARTIST, self.domain, self.instance_id)
554
555 for i, file_info in enumerate(audio_files, 1):
556 filename = file_info.get("name", "")
557
558 # Use file's title if available, otherwise clean up filename
559 track_name = file_info.get("title", filename)
560 if not track_name or track_name == filename:
561 track_name = filename.rsplit(".", 1)[0] if "." in filename else filename
562
563 # Try to extract track number from file metadata first, then filename
564 track_number = self._extract_track_number(file_info, track_name, i)
565
566 track = Track(
567 item_id=f"{prov_album_id}#{filename}",
568 provider=self.instance_id,
569 name=track_name,
570 track_number=track_number,
571 provider_mappings={
572 ProviderMapping(
573 item_id=f"{prov_album_id}#{filename}",
574 provider_domain=self.domain,
575 provider_instance=self.instance_id,
576 url=self.client.get_download_url(prov_album_id, filename),
577 available=True,
578 )
579 },
580 )
581
582 # Add file-specific artist if available, otherwise use album artist
583 file_artist = file_info.get("artist") or file_info.get("creator")
584 if file_artist:
585 file_artist_cleaned = clean_text(file_artist)
586 file_artist_normalized = file_artist_cleaned.lower()
587 # Check if this is the same as album artist to avoid duplicates (case-insensitive)
588 if album_artist_normalized and file_artist_normalized == album_artist_normalized:
589 track.artists = UniqueList([album_artist_obj])
590 else:
591 track.artists = UniqueList(
592 [create_artist(file_artist_cleaned, self.domain, self.instance_id)]
593 )
594 else:
595 # Use pre-created album artist object
596 track.artists = UniqueList([album_artist_obj])
597
598 # Add duration if available
599 if duration_str := file_info.get("length"):
600 if duration := parse_duration(duration_str):
601 track.duration = duration
602
603 # Add genre if available
604 if genre := file_info.get("genre"):
605 track.metadata.genres = {clean_text(genre)}
606
607 tracks.append(track)
608
609 return tracks
610
611 def _extract_track_number(
612 self, file_info: dict[str, Any], track_name: str, fallback: int
613 ) -> int:
614 """Extract track number from file metadata or filename."""
615 track_number = None
616
617 if "track" in file_info:
618 with contextlib.suppress(ValueError, AttributeError):
619 track_number = int(str(file_info["track"]).split("/")[0])
620
621 if track_number is None:
622 # Fallback to filename parsing
623 track_num_match = re.search(r"^(\d+)[\s\-_.]*(.+)", track_name)
624 track_number = int(track_num_match.group(1)) if track_num_match else fallback
625
626 return track_number
627
628 @use_cache(
629 expiration=86400 * 30, allow_expired_cache=True
630 ) # Cache for 30 days - artist catalogs change infrequently
631 async def get_artist_albums(self, prov_artist_id: str) -> list[Album]:
632 """
633 Get albums for a specific artist.
634
635 Uses metadata heuristics to determine likely albums without expensive
636 API calls for better performance.
637
638 Args:
639 prov_artist_id: Provider-specific artist identifier (artist name)
640
641 Returns:
642 List of Album objects by the artist
643 """
644 albums: list[Album] = []
645 page = 0
646 page_size = 200 # IA's maximum
647
648 while len(albums) < 1000: # Reasonable upper limit
649 search_response = await self._search(
650 query=f'creator:"{prov_artist_id}" AND (format:"VBR MP3" OR format:"FLAC" \
651 OR format:"Ogg Vorbis")',
652 sort="downloads desc",
653 rows=page_size,
654 page=page,
655 )
656
657 docs = search_response.get("response", {}).get("docs", [])
658 if not docs:
659 break
660
661 for doc in docs:
662 try:
663 # Use metadata heuristics instead of expensive API calls
664 # to determine if item is an album
665 if is_likely_album(doc):
666 album = doc_to_album(
667 doc, self.domain, self.instance_id, self.client.get_item_url
668 )
669 if album:
670 albums.append(album)
671 except (KeyError, ValueError, TypeError) as err:
672 self.logger.debug(
673 "Skipping invalid album for artist %s: %s", prov_artist_id, err
674 )
675 continue
676 except (TimeoutError, aiohttp.ClientError) as err:
677 self.logger.debug(
678 "Network error processing album for artist %s: %s", prov_artist_id, err
679 )
680 continue
681 except Exception as err:
682 self.logger.exception(
683 "Unexpected error processing album for artist %s: %s", prov_artist_id, err
684 )
685 continue
686 page += 1
687 return albums
688
689 @use_cache(expiration=86400 * 7, allow_expired_cache=True) # Cache for 1 week
690 async def get_artist_toptracks(self, prov_artist_id: str) -> list[Track]:
691 """
692 Get top tracks for a specific artist.
693
694 Uses the same search as get_artist_albums but filters for single tracks.
695
696 Args:
697 prov_artist_id: Provider-specific artist identifier (artist name)
698
699 Returns:
700 List of Track objects representing the artist's top tracks
701 """
702 tracks = []
703 search_response = await self._search(
704 query=(
705 f'creator:"{prov_artist_id}" AND '
706 f'(format:"VBR MP3" OR format:"FLAC" OR format:"Ogg Vorbis")'
707 ),
708 rows=25, # Limit for "top" tracks
709 sort="downloads desc",
710 )
711
712 response_data = search_response.get("response", {})
713 docs = response_data.get("docs", [])
714
715 for doc in docs:
716 try:
717 # Only include items that are NOT classified as albums
718 if not is_likely_album(doc):
719 track = doc_to_track(
720 doc, self.domain, self.instance_id, self.client.get_item_url
721 )
722 if track:
723 tracks.append(track)
724 except (KeyError, ValueError, TypeError) as err:
725 self.logger.debug("Skipping invalid track for artist %s: %s", prov_artist_id, err)
726 continue
727 except (TimeoutError, aiohttp.ClientError) as err:
728 self.logger.debug(
729 "Network error processing track for artist %s: %s", prov_artist_id, err
730 )
731 continue
732 except Exception as err:
733 self.logger.exception(
734 "Unexpected error processing track for artist %s: %s", prov_artist_id, err
735 )
736 continue
737
738 if len(tracks) >= 25:
739 break
740
741 return tracks
742
743 async def get_stream_details(self, item_id: str, media_type: MediaType) -> StreamDetails:
744 """
745 Get streamdetails for a track or audiobook.
746
747 Delegates to the streaming handler for proper multi-file support.
748
749 Args:
750 item_id: Provider-specific item identifier
751 media_type: The type of media being requested
752
753 Returns:
754 StreamDetails object configured for the specific item type
755
756 Raises:
757 MediaNotFoundError: If no audio files are found for the item
758 """
759 return await self.streaming.get_stream_details(item_id, media_type)
760
761 async def _calculate_audiobook_duration_and_chapters(
762 self, item_id: str
763 ) -> tuple[int, list[MediaItemChapter]]:
764 """Calculate duration and chapters for audiobooks."""
765 audio_files = await self._get_audio_files(item_id)
766 total_duration = 0
767 chapters = []
768 current_position = 0.0
769
770 for i, file_info in enumerate(audio_files, 1):
771 chapter_duration = parse_duration(file_info.get("length", "0")) or 0
772 total_duration += chapter_duration
773
774 chapter_name = file_info.get("title") or file_info.get("name", f"Chapter {i}")
775 chapter = MediaItemChapter(
776 position=i,
777 name=clean_text(chapter_name),
778 start=current_position,
779 end=current_position + chapter_duration if chapter_duration > 0 else None,
780 )
781 chapters.append(chapter)
782 current_position += chapter_duration
783
784 return total_duration, chapters
785
786 async def get_audio_stream(
787 self, streamdetails: StreamDetails, seek_position: int = 0
788 ) -> AsyncGenerator[bytes]:
789 """Get audio stream from Internet Archive."""
790 # Use sock_read=None to allow long audiobook chapters to stream fully
791 timeout = aiohttp.ClientTimeout(sock_read=None, total=None)
792
793 if streamdetails.media_type == MediaType.AUDIOBOOK and isinstance(streamdetails.data, dict):
794 chapter_urls = streamdetails.data.get("chapters", [])
795 chapters_data = streamdetails.data.get("chapters_data", [])
796
797 # Calculate which chapter to start from based on seek_position
798 seek_position_ms = seek_position * 1000
799 start_chapter = 0
800
801 if seek_position > 0 and chapters_data:
802 accumulated_duration_ms = 0
803
804 for i, chapter_data in enumerate(chapters_data):
805 chapter_duration_ms = (
806 parse_duration(chapter_data.get("length", "0")) or 0
807 ) * 1000
808
809 if accumulated_duration_ms + chapter_duration_ms > seek_position_ms:
810 start_chapter = i
811 break
812 accumulated_duration_ms += chapter_duration_ms
813
814 # Stream chapters starting from calculated position
815 chapters_yielded = False
816 for i in range(start_chapter, len(chapter_urls)):
817 chapter_url = chapter_urls[i]
818
819 try:
820 async with self.mass.http_session.get(chapter_url, timeout=timeout) as response:
821 response.raise_for_status()
822 async for chunk in response.content.iter_chunked(8192):
823 chapters_yielded = True
824 yield chunk
825 except Exception as e:
826 self.logger.error(f"Chapter {i + 1} streaming failed: {e}")
827 continue
828
829 # If no chapters succeeded, raise an error instead of silent failure
830 if not chapters_yielded:
831 raise MediaNotFoundError(
832 f"Failed to stream any chapters for audiobook {streamdetails.item_id}"
833 )
834
835 else:
836 # Handle single files
837 audio_files = await self._get_audio_files(streamdetails.item_id)
838 if audio_files:
839 download_url = self.client.get_download_url(
840 streamdetails.item_id, audio_files[0]["name"]
841 )
842 async with self.mass.http_session.get(download_url, timeout=timeout) as response:
843 response.raise_for_status()
844 async for chunk in response.content.iter_chunked(8192):
845 yield chunk
846
847 @use_cache(expiration=86400 * 7) # Cache for 1 week
848 async def get_podcast(self, prov_podcast_id: str) -> Podcast:
849 """Get full podcast details by id."""
850 metadata = await self._get_metadata(prov_podcast_id)
851 item_metadata = metadata.get("metadata", {})
852
853 title = clean_text(item_metadata.get("title"))
854 creator = clean_text(item_metadata.get("creator"))
855
856 if not title:
857 raise MediaNotFoundError(f"Podcast {prov_podcast_id} not found or invalid")
858
859 podcast = Podcast(
860 item_id=prov_podcast_id,
861 provider=self.instance_id,
862 name=title,
863 provider_mappings={
864 create_provider_mapping(
865 prov_podcast_id, self.domain, self.instance_id, self.client.get_item_url
866 )
867 },
868 )
869
870 # Add publisher/creator
871 if creator:
872 podcast.publisher = creator
873
874 # Add metadata
875 if description := clean_text(item_metadata.get("description")):
876 podcast.metadata.description = description
877
878 # Add thumbnail
879 add_item_image(podcast, prov_podcast_id, self.instance_id)
880
881 # Calculate total episodes
882 try:
883 audio_files = await self._get_audio_files(prov_podcast_id)
884 podcast.total_episodes = len(audio_files)
885 except Exception as err:
886 self.logger.warning(f"Could not get episode count for podcast {prov_podcast_id}: {err}")
887 podcast.total_episodes = None
888
889 return podcast
890
891 async def get_podcast_episodes(self, prov_podcast_id: str) -> AsyncGenerator[PodcastEpisode]:
892 """Get podcast episodes for given podcast id."""
893 metadata = await self._get_metadata(prov_podcast_id)
894 item_metadata = metadata.get("metadata", {})
895 audio_files = await self._get_audio_files(prov_podcast_id)
896
897 # Create podcast reference for episodes
898 podcast = Podcast(
899 item_id=prov_podcast_id,
900 provider=self.instance_id,
901 name=clean_text(item_metadata.get("title", prov_podcast_id)),
902 provider_mappings={
903 create_provider_mapping(
904 prov_podcast_id, self.domain, self.instance_id, self.client.get_item_url
905 )
906 },
907 )
908
909 for i, file_info in enumerate(audio_files, 1):
910 filename = file_info.get("name", "")
911
912 # Use file's title if available, otherwise clean up filename
913 episode_name = file_info.get("title", filename)
914 if not episode_name or episode_name == filename:
915 episode_name = filename.rsplit(".", 1)[0] if "." in filename else filename
916
917 # Try to extract episode number from file metadata first, then filename
918 episode_number = self._extract_track_number(file_info, episode_name, i)
919
920 episode = PodcastEpisode(
921 item_id=f"{prov_podcast_id}#{filename}",
922 provider=self.instance_id,
923 name=episode_name,
924 position=episode_number,
925 podcast=podcast,
926 provider_mappings={
927 ProviderMapping(
928 item_id=f"{prov_podcast_id}#{filename}",
929 provider_domain=self.domain,
930 provider_instance=self.instance_id,
931 url=self.client.get_download_url(prov_podcast_id, filename),
932 available=True,
933 )
934 },
935 )
936
937 # Add duration if available
938 if duration_str := file_info.get("length"):
939 if duration := parse_duration(duration_str):
940 episode.duration = duration
941
942 # Add episode metadata
943 if description := file_info.get("description"):
944 episode.metadata.description = clean_text(description)
945
946 yield episode
947
948 async def get_podcast_episode(self, prov_episode_id: str) -> PodcastEpisode:
949 """Get single podcast episode by id."""
950 if "#" not in prov_episode_id:
951 raise MediaNotFoundError(f"Invalid episode ID format: {prov_episode_id}")
952
953 podcast_id, _ = prov_episode_id.split("#", 1)
954
955 async for episode in self.get_podcast_episodes(podcast_id):
956 if episode.item_id == prov_episode_id:
957 return episode
958
959 raise MediaNotFoundError(f"Episode {prov_episode_id} not found")
960