/
/
1"""Helper for parsing and using audible api."""
2
3from __future__ import annotations
4
5import asyncio
6import hashlib
7import html
8import json
9import logging
10import os
11import re
12from collections.abc import AsyncGenerator
13from contextlib import suppress
14from datetime import UTC, datetime, timedelta
15from os import PathLike
16from typing import TYPE_CHECKING, Any
17from urllib.parse import parse_qs, urlparse
18
19import audible
20import audible.exceptions
21import audible.register
22from audible import AsyncClient
23
24if TYPE_CHECKING:
25 from aiohttp import ClientSession
26
27 from music_assistant.models.music_provider import MusicProvider
28from music_assistant_models.enums import ContentType, ImageType, MediaType, StreamType
29from music_assistant_models.errors import (
30 LoginFailed,
31 MediaNotFoundError,
32 ProviderUnavailableError,
33)
34from music_assistant_models.media_items import (
35 Audiobook,
36 AudioFormat,
37 ItemMapping,
38 MediaItemChapter,
39 MediaItemImage,
40 Podcast,
41 PodcastEpisode,
42 ProviderMapping,
43 UniqueList,
44)
45from music_assistant_models.streamdetails import StreamDetails
46
47from music_assistant.helpers.datetime import utc
48from music_assistant.mass import MusicAssistant
49
50CACHE_DOMAIN = "audible"
51CACHE_CATEGORY_API = 0
52CACHE_CATEGORY_AUDIOBOOK = 1
53CACHE_CATEGORY_CHAPTERS = 2
54CACHE_CATEGORY_PODCAST = 3
55CACHE_CATEGORY_PODCAST_EPISODES = 4
56
57# Content delivery types
58AUDIOBOOK_CONTENT_TYPES = ("SinglePartBook", "MultiPartBook")
59PODCAST_CONTENT_TYPES = ("PodcastParent",)
60
61_AUTH_CACHE: dict[str, audible.Authenticator] = {}
62
63
64async def refresh_access_token_compat(
65 refresh_token: str, domain: str, http_session: ClientSession, with_username: bool = False
66) -> dict[str, Any]:
67 """
68 Refresh tokens with compatibility for new Audible API format.
69
70 The Audible API changed from returning 'access_token' to 'actor_access_token'.
71 This function handles both formats for backward compatibility.
72
73 :param refresh_token: The refresh token obtained after device registration.
74 :param domain: The top level domain (e.g., com, de).
75 :param http_session: The HTTP client session to use for requests.
76 :param with_username: If True, use audible domain instead of amazon.
77 :return: Dict with access_token and expires timestamp.
78 """
79 logger = logging.getLogger("audible_helper")
80
81 body = {
82 "app_name": "Audible",
83 "app_version": "3.56.2",
84 "source_token": refresh_token,
85 "requested_token_type": "access_token",
86 "source_token_type": "refresh_token",
87 }
88
89 target_domain = "audible" if with_username else "amazon"
90 url = f"https://api.{target_domain}.{domain}/auth/token"
91
92 async with http_session.post(url, data=body) as resp:
93 resp.raise_for_status()
94 resp_dict = await resp.json()
95
96 expires_in_sec = int(resp_dict.get("expires_in", 3600))
97 expires = (utc() + timedelta(seconds=expires_in_sec)).timestamp()
98
99 # Handle new format (actor_access_token) or fall back to legacy (access_token)
100 access_token = resp_dict.get("actor_access_token") or resp_dict.get("access_token")
101
102 if not access_token:
103 logger.error("Token refresh response missing both actor_access_token and access_token")
104 raise LoginFailed("Token refresh failed: no access token in response")
105
106 logger.debug(
107 "Token refreshed successfully using %s format",
108 "new (actor)" if "actor_access_token" in resp_dict else "legacy",
109 )
110
111 return {"access_token": access_token, "expires": expires}
112
113
114async def cached_authenticator_from_file(path: str) -> audible.Authenticator:
115 """
116 Get an authenticator from file with caching and signing auth validation.
117
118 :param path: Path to the authenticator JSON file.
119 :return: The cached or loaded Authenticator instance.
120 """
121 logger = logging.getLogger("audible_helper")
122 if path in _AUTH_CACHE:
123 return _AUTH_CACHE[path]
124
125 logger.debug("Loading authenticator from file %s and caching it", path)
126 auth = await asyncio.to_thread(audible.Authenticator.from_file, path)
127
128 # Verify signing auth is available (not affected by API changes)
129 if auth.adp_token and auth.device_private_key:
130 logger.debug("Signing auth available - using stable RSA-signed requests")
131 else:
132 logger.warning(
133 "Signing auth not available - only bearer auth will work. "
134 "Consider re-authenticating for more stable auth."
135 )
136
137 _AUTH_CACHE[path] = auth
138 return auth
139
140
141class AudibleHelper:
142 """Helper for parsing and using audible api."""
143
144 def __init__(
145 self,
146 mass: MusicAssistant,
147 client: AsyncClient,
148 provider_domain: str,
149 provider_instance: str,
150 provider: MusicProvider,
151 logger: logging.Logger | None = None,
152 ):
153 """
154 Initialize the Audible Helper.
155
156 :param mass: The MusicAssistant instance.
157 :param client: An authenticated Audible API client.
158 :param provider_domain: Domain of the owning provider.
159 :param provider_instance: Instance id of the owning provider.
160 :param provider: The owning provider, used to report library items it had to skip.
161 :param logger: Logger to use, defaults to a module level logger.
162 """
163 self.mass = mass
164 self.client = client
165 self.provider_domain = provider_domain
166 self.provider_instance = provider_instance
167 self.provider = provider
168 self.logger = logger or logging.getLogger("audible_helper")
169 self._acr_cache: dict[tuple[str, MediaType], str] = {}
170
171 async def _fetch_library_items(
172 self,
173 response_groups: str,
174 content_types: tuple[str, ...],
175 ) -> AsyncGenerator[dict[str, Any]]:
176 """Fetch items from the library with pagination."""
177 page = 1
178 page_size = 50
179 total_processed = 0
180 max_iterations = 100
181 iteration = 0
182
183 while iteration < max_iterations:
184 iteration += 1
185 self.logger.debug(
186 "Audible: Fetching library page %s (processed so far: %s)",
187 page,
188 total_processed,
189 )
190
191 library = await self._call_api(
192 "library",
193 use_cache=False,
194 response_groups=response_groups,
195 page=page,
196 num_results=page_size,
197 )
198
199 items = library.get("items", [])
200
201 if not items:
202 break
203
204 items_processed_this_page = 0
205 for item in items:
206 # Filter by content type if specified
207 if content_types and item.get("content_delivery_type") not in content_types:
208 continue
209
210 yield item
211 items_processed_this_page += 1
212 total_processed += 1
213
214 self.logger.debug(
215 "Audible: Processed %s items on page %s", items_processed_this_page, page
216 )
217
218 page += 1
219 if len(items) < page_size:
220 break
221
222 if iteration >= max_iterations:
223 self.logger.warning(
224 "Audible: Reached maximum iteration limit (%s) with %s items processed",
225 max_iterations,
226 total_processed,
227 )
228
229 async def _process_audiobook_item(self, audiobook_data: dict[str, Any]) -> Audiobook | None:
230 """Process a single audiobook item from the library."""
231 # Ensure asin is a valid string
232 asin = str(audiobook_data.get("asin", ""))
233 cached_book = None
234 if asin:
235 cached_book = await self.mass.cache.get(
236 key=asin,
237 provider=self.provider_instance,
238 category=CACHE_CATEGORY_AUDIOBOOK,
239 default=None,
240 )
241
242 try:
243 if cached_book is not None:
244 return self._parse_audiobook(cached_book)
245 return self._parse_audiobook(audiobook_data)
246 except Exception as exc:
247 self.provider.report_skipped_sync_item(MediaType.AUDIOBOOK, asin or None, exc)
248 return None
249
250 async def get_library(self) -> AsyncGenerator[Audiobook]:
251 """Fetch the user's library with pagination."""
252 response_groups = [
253 "contributors",
254 "media",
255 "product_attrs",
256 "product_desc",
257 "product_details",
258 "product_extended_attrs",
259 ]
260
261 async for item in self._fetch_library_items(
262 ",".join(response_groups), AUDIOBOOK_CONTENT_TYPES
263 ):
264 if album := await self._process_audiobook_item(item):
265 yield album
266
267 async def get_audiobook(self, asin: str, use_cache: bool = True) -> Audiobook:
268 """
269 Fetch the full audiobook by asin with all details including chapters.
270
271 This method fetches complete audiobook details including chapters and resume position.
272 Use this when the user requests full details for a specific audiobook.
273 """
274 if use_cache:
275 cached_book = await self.mass.cache.get(
276 key=asin,
277 provider=self.provider_instance,
278 category=CACHE_CATEGORY_AUDIOBOOK,
279 default=None,
280 )
281 if cached_book is not None:
282 book = self._parse_audiobook(cached_book)
283 # Enrich with chapters and resume position
284 await self._enrich_audiobook(book, asin)
285 return book
286 response = await self._call_api(
287 f"library/{asin}",
288 response_groups="""
289 contributors, media, price, product_attrs, product_desc, product_details,
290 product_extended_attrs,is_finished
291 """,
292 )
293
294 if response is None:
295 raise MediaNotFoundError(f"Audiobook with ASIN {asin} not found")
296
297 item_data = response.get("item")
298 if item_data is None:
299 raise MediaNotFoundError(f"Audiobook data for ASIN {asin} is empty")
300
301 await self.mass.cache.set(
302 key=asin,
303 provider=self.provider_instance,
304 category=CACHE_CATEGORY_AUDIOBOOK,
305 data=item_data,
306 )
307 book = self._parse_audiobook(item_data)
308 # Enrich with chapters and resume position
309 await self._enrich_audiobook(book, asin)
310 return book
311
312 async def _enrich_audiobook(self, book: Audiobook, asin: str) -> None:
313 """
314 Enrich audiobook with chapters and resume position.
315
316 This makes additional API calls and should only be used for full audiobook details,
317 not during library sync.
318 """
319 # Fetch chapters
320 chapters_data = await self._fetch_chapters(asin=asin)
321 if chapters_data:
322 chapters: list[MediaItemChapter] = [
323 self._parse_chapter_data(chapter, idx) for idx, chapter in enumerate(chapters_data)
324 ]
325 book.metadata.chapters = chapters
326 # Update duration from chapters if available (more accurate)
327 try:
328 duration = int(sum(chapter.get("length_ms", 0) for chapter in chapters_data) / 1000)
329 if duration > 0:
330 book.duration = duration
331 except Exception as exc:
332 self.logger.warning(f"Error calculating duration from chapters for {asin}: {exc}")
333
334 # Fetch resume position
335 book.resume_position_ms = await self.get_last_position(asin=asin)
336
337 async def get_stream(
338 self, asin: str, media_type: MediaType = MediaType.AUDIOBOOK
339 ) -> StreamDetails:
340 """
341 Get stream details for an audiobook or podcast episode.
342
343 :param asin: The ASIN of the content.
344 :param media_type: The type of media (audiobook or podcast episode).
345 """
346 if not asin:
347 self.logger.error("Invalid ASIN provided to get_stream")
348 raise ValueError("Invalid ASIN provided to get_stream")
349
350 duration = 0
351 # For audiobooks, try to get duration from chapters
352 if media_type == MediaType.AUDIOBOOK:
353 chapters = await self._fetch_chapters(asin=asin)
354 if chapters:
355 try:
356 duration = int(sum(chapter.get("length_ms", 0) for chapter in chapters) / 1000)
357 except Exception as exc:
358 self.logger.warning(f"Error calculating duration for ASIN {asin}: {exc}")
359
360 try:
361 # Podcasts use Mpeg (non-DRM MP3), audiobooks use HLS
362 if media_type == MediaType.PODCAST_EPISODE:
363 playback_info = await self.client.post(
364 f"content/{asin}/licenserequest",
365 body={
366 "consumption_type": "Streaming",
367 "drm_type": "Mpeg",
368 "quality": "High",
369 },
370 )
371 else:
372 playback_info = await self.client.post(
373 f"content/{asin}/licenserequest",
374 body={
375 "quality": "High",
376 "response_groups": "content_reference,certificate",
377 "consumption_type": "Streaming",
378 "supported_media_features": {
379 "codecs": ["mp4a.40.2", "mp4a.40.42"],
380 "drm_types": [
381 "Hls",
382 ],
383 },
384 "spatial": False,
385 },
386 )
387
388 content_license = playback_info.get("content_license", {})
389 if not content_license:
390 self.logger.error(f"No content_license in playback_info for ASIN {asin}")
391 raise ValueError(f"Missing content_license for ASIN {asin}")
392
393 content_metadata = content_license.get("content_metadata", {})
394 content_reference = content_metadata.get("content_reference", {})
395 size = content_reference.get("content_size_in_bytes", 0)
396
397 stream_url = content_license.get("license_response")
398 if not stream_url:
399 self.logger.error(f"No license_response (stream URL) for ASIN {asin}")
400 raise ValueError(f"Missing stream URL for ASIN {asin}")
401
402 acr = content_license.get("acr", "")
403 if acr:
404 self._acr_cache[(asin, media_type)] = acr
405
406 content_type = (
407 ContentType.MP3 if media_type == MediaType.PODCAST_EPISODE else ContentType.AAC
408 )
409 except Exception as exc:
410 self.logger.error(f"Error getting stream details for ASIN {asin}: {exc}")
411 raise ValueError(f"Failed to get stream details: {exc}") from exc
412
413 return StreamDetails(
414 provider=self.provider_instance,
415 size=size,
416 item_id=f"{asin}",
417 audio_format=AudioFormat(content_type=content_type),
418 media_type=media_type,
419 stream_type=StreamType.HTTP,
420 path=stream_url,
421 can_seek=True,
422 allow_seek=True,
423 duration=duration,
424 data={"acr": acr},
425 )
426
427 async def _fetch_chapters(self, asin: str) -> list[dict[str, Any]]:
428 """Fetch chapter data for an audiobook."""
429 if not asin or asin == "error":
430 self.logger.warning(
431 "Invalid ASIN provided to _fetch_chapters, returning empty chapter list"
432 )
433 return []
434
435 chapters_data: list[Any] = await self.mass.cache.get(
436 key=asin, provider=self.provider_instance, category=CACHE_CATEGORY_CHAPTERS, default=[]
437 )
438
439 if not chapters_data:
440 try:
441 response = await self._call_api(
442 f"content/{asin}/metadata",
443 response_groups="chapter_info, always-returned, content_reference, content_url",
444 chapter_titles_type="Flat",
445 )
446
447 if not response:
448 self.logger.warning(f"Failed to get metadata for ASIN {asin}")
449 return []
450
451 content_metadata = response.get("content_metadata")
452 if not content_metadata:
453 self.logger.warning(f"No content_metadata for ASIN {asin}")
454 return []
455
456 chapter_info = content_metadata.get("chapter_info")
457 if not chapter_info:
458 self.logger.warning(f"No chapter_info for ASIN {asin}")
459 return []
460
461 chapters_data = chapter_info.get("chapters") or []
462
463 await self.mass.cache.set(
464 key=asin,
465 data=chapters_data,
466 provider=self.provider_instance,
467 category=CACHE_CATEGORY_CHAPTERS,
468 )
469 except Exception as exc:
470 self.logger.error(f"Error fetching chapters for ASIN {asin}: {exc}")
471 chapters_data = []
472
473 return chapters_data
474
475 @staticmethod
476 def _parse_audible_timestamp(raw_ts: Any) -> datetime | None:
477 """
478 Parse an Audible timestamp value into a timezone-aware datetime.
479
480 :param raw_ts: The raw timestamp value from the Audible annotation payload.
481 """
482 if not raw_ts:
483 return None
484 try:
485 parsed = datetime.fromisoformat(str(raw_ts))
486 except ValueError, TypeError:
487 return None
488 if parsed.tzinfo is None:
489 parsed = parsed.replace(tzinfo=UTC)
490 return parsed
491
492 async def _fetch_last_position(self, asin: str) -> tuple[int, datetime | None] | None:
493 """
494 Fetch the last-heard position for a single ASIN from Audible.
495
496 :param asin: The audiobook ASIN to query.
497 """
498 response = await self._call_api("annotations/lastpositions", asins=asin)
499 if not response:
500 return None
501
502 annotations = response.get("asin_last_position_heard_annots")
503 if not annotations or not isinstance(annotations, list):
504 return None
505
506 annotation = annotations[0]
507 if not isinstance(annotation, dict):
508 return None
509
510 last_position = annotation.get("last_position_heard")
511 if not isinstance(last_position, dict):
512 return None
513
514 position_ms = int(last_position.get("position_ms", 0))
515
516 timestamp: datetime | None = None
517 for field in ("last_updated", "reported_time", "last_updated_time", "timestamp"):
518 timestamp = self._parse_audible_timestamp(
519 last_position.get(field) or annotation.get(field)
520 )
521 if timestamp is not None:
522 break
523
524 return position_ms, timestamp
525
526 async def get_last_position(self, asin: str) -> int:
527 """
528 Fetch the last-heard position in milliseconds for the given ASIN.
529
530 :param asin: The audiobook ASIN to query.
531 """
532 if not asin or asin == "error":
533 return 0
534 try:
535 result = await self._fetch_last_position(asin)
536 except (ProviderUnavailableError, KeyError, TypeError, ValueError) as exc:
537 self.logger.error("Error getting last position for ASIN %s: %s", asin, exc)
538 return 0
539 return result[0] if result else 0
540
541 async def set_last_position(
542 self, asin: str, pos: int, media_type: MediaType = MediaType.AUDIOBOOK
543 ) -> None:
544 """
545 Report last position to Audible.
546
547 :param asin: The content ID (audiobook or podcast episode).
548 :param pos: Position in seconds.
549 :param media_type: The type of media (audiobook or podcast episode).
550 """
551 if not asin or asin == "error" or pos <= 0:
552 return
553
554 try:
555 position_ms = pos * 1000
556
557 # Try to get ACR from cache first
558 acr = self._acr_cache.get((asin, media_type))
559 if not acr:
560 stream_details = await self.get_stream(asin=asin, media_type=media_type)
561 acr = stream_details.data.get("acr")
562
563 if not acr:
564 self.logger.warning(f"No ACR available for ASIN {asin}, cannot report position")
565 return
566
567 await self.client.put(
568 f"lastpositions/{asin}", body={"acr": acr, "asin": asin, "position_ms": position_ms}
569 )
570
571 self.logger.debug(f"Successfully reported position {position_ms}ms for ASIN {asin}")
572
573 except (KeyError, TypeError) as exc:
574 self.logger.error(
575 f"Error accessing data while reporting position for ASIN {asin}: {exc}"
576 )
577 except TimeoutError as exc:
578 self.logger.error(f"Timeout while reporting position for ASIN {asin}: {exc}")
579 except ConnectionError as exc:
580 self.logger.error(f"Connection error while reporting position for ASIN {asin}: {exc}")
581 except Exception as exc:
582 self.logger.error(f"Unexpected error reporting position for ASIN {asin}: {exc}")
583
584 async def get_audible_resume_position(self, asin: str) -> tuple[bool, int, datetime | None]:
585 """
586 Return resume state for the given ASIN from Audible.
587
588 :param asin: The audiobook ASIN to query.
589 """
590 if not asin or asin == "error":
591 raise NotImplementedError
592 try:
593 result = await self._fetch_last_position(asin)
594 except (ProviderUnavailableError, KeyError, TypeError, ValueError) as exc:
595 self.logger.debug("Audible lastpositions fetch failed for %s: %s", asin, exc)
596 raise NotImplementedError from exc
597 if not result or result[0] == 0:
598 raise NotImplementedError
599 position_ms, timestamp = result
600 return False, position_ms, timestamp
601
602 async def _call_api(self, path: str, **kwargs: Any) -> Any:
603 response = None
604 use_cache = kwargs.pop("use_cache", False)
605 params_str = json.dumps(kwargs, sort_keys=True)
606 params_hash = hashlib.md5(params_str.encode()).hexdigest()
607 cache_key_with_params = f"{path}:{params_hash}"
608 if use_cache:
609 response = await self.mass.cache.get(
610 key=cache_key_with_params,
611 provider=self.provider_instance,
612 category=CACHE_CATEGORY_API,
613 )
614 if not response:
615 try:
616 response = await self.client.get(path, **kwargs)
617 except audible.exceptions.RequestError as exc:
618 raise ProviderUnavailableError(
619 f"Audible API request failed for '{path}': {exc}"
620 ) from exc
621 await self.mass.cache.set(
622 key=cache_key_with_params, provider=self.provider_instance, data=response
623 )
624 return response
625
626 def _parse_contributors(
627 self, contributors_list: list[dict[str, Any]] | None, default_name: str
628 ) -> list[str]:
629 """Parse contributors (authors, narrators) from API response."""
630 result: list[str] = []
631 contributors: list[dict[str, Any]] = contributors_list or []
632 if isinstance(contributors, list):
633 for contributor in contributors:
634 if contributor and isinstance(contributor, dict):
635 result.append(contributor.get("name", default_name))
636 return result
637
638 def _create_images(self, image_path: str | None) -> list[MediaItemImage]:
639 """Create image objects if image path exists."""
640 images: list[MediaItemImage] = []
641 if image_path:
642 images.append(
643 MediaItemImage(
644 type=ImageType.THUMB,
645 path=image_path,
646 provider=self.provider_instance,
647 remotely_accessible=True,
648 )
649 )
650 images.append(
651 MediaItemImage(
652 type=ImageType.CLEARART,
653 path=image_path,
654 provider=self.provider_instance,
655 remotely_accessible=True,
656 )
657 )
658 return images
659
660 def _parse_chapter_data(self, chapter_data: dict[str, Any], index: int) -> MediaItemChapter:
661 """Parse chapter data into MediaItemChapter object."""
662 try:
663 start = int(chapter_data.get("start_offset_sec", 0))
664 except TypeError, ValueError:
665 start = 0
666
667 try:
668 length = int(chapter_data.get("length_ms", 0)) / 1000
669 except TypeError, ValueError:
670 length = 0
671
672 raw_title = chapter_data.get("title")
673 chapter_title: str
674 if raw_title is None:
675 chapter_title = f"Chapter {index + 1}"
676 elif isinstance(raw_title, str):
677 chapter_title = raw_title
678 else:
679 chapter_title = str(raw_title)
680
681 return MediaItemChapter(position=index, name=chapter_title, start=start, end=start + length)
682
683 def _parse_audiobook(self, audiobook_data: dict[str, Any] | None) -> Audiobook:
684 """
685 Parse audiobook data from API response.
686
687 NOTE: This is a pure parser - no API calls allowed here.
688 Chapters and resume position are fetched lazily when needed.
689 """
690 if audiobook_data is None:
691 self.logger.error("Received None audiobook_data in _parse_audiobook")
692 raise MediaNotFoundError("Audiobook data not found")
693
694 asin = audiobook_data.get("asin", "")
695 title = audiobook_data.get("title", "")
696
697 # Parse authors and narrators
698 narrators = self._parse_contributors(audiobook_data.get("narrators"), "Unknown Narrator")
699 authors = self._parse_contributors(audiobook_data.get("authors"), "Unknown Author")
700
701 # Get duration from runtime_length_min (provided by 'media' response group)
702 # Chapters are fetched lazily when streaming, not during library sync
703 runtime_minutes = audiobook_data.get("runtime_length_min", 0)
704 duration = runtime_minutes * 60 if runtime_minutes else 0
705
706 # Create audiobook object
707 book = Audiobook(
708 item_id=asin,
709 provider=self.provider_instance,
710 name=title,
711 duration=duration,
712 provider_mappings={
713 ProviderMapping(
714 item_id=asin,
715 provider_domain=self.provider_domain,
716 provider_instance=self.provider_instance,
717 )
718 },
719 publisher=audiobook_data.get("publisher_name"),
720 authors=UniqueList(authors),
721 narrators=UniqueList(narrators),
722 )
723
724 # Set metadata
725 book.metadata.copyright = audiobook_data.get("copyright")
726 book.metadata.description = _html_to_txt(
727 str(audiobook_data.get("extended_product_description", ""))
728 )
729 book.metadata.languages = UniqueList([audiobook_data.get("language") or ""])
730 if release_date := audiobook_data.get("release_date"):
731 with suppress(ValueError):
732 parsed_date = datetime.strptime(release_date, "%Y-%m-%d").astimezone(UTC)
733 book.metadata.release_date = parsed_date
734
735 # Set review if available
736 reviews = audiobook_data.get("editorial_reviews", [])
737 if reviews and reviews[0]:
738 book.metadata.review = _html_to_txt(str(reviews[0]))
739
740 # Set genres
741 book.metadata.genres = {
742 genre.replace("_", " ") for genre in (audiobook_data.get("platinum_keywords") or [])
743 }
744
745 # Add images
746 image_path = audiobook_data.get("product_images", {}).get("500")
747 book.metadata.images = UniqueList(self._create_images(image_path))
748
749 # Chapters are not fetched during parsing - they are fetched lazily when streaming
750 # This avoids N+1 API calls during library sync
751
752 return book
753
754 async def _process_podcast_item(self, podcast_data: dict[str, Any]) -> Podcast | None:
755 """Process a single podcast item from the library."""
756 asin = str(podcast_data.get("asin", ""))
757 cached_podcast = None
758 if asin:
759 cached_podcast = await self.mass.cache.get(
760 key=asin,
761 provider=self.provider_instance,
762 category=CACHE_CATEGORY_PODCAST,
763 default=None,
764 )
765
766 try:
767 if cached_podcast is not None:
768 return self._parse_podcast(cached_podcast)
769 return self._parse_podcast(podcast_data)
770 except Exception as exc:
771 self.provider.report_skipped_sync_item(MediaType.PODCAST, asin or None, exc)
772 return None
773
774 async def get_library_podcasts(self) -> AsyncGenerator[Podcast]:
775 """Fetch podcasts from the user's library with pagination."""
776 response_groups = [
777 "contributors",
778 "media",
779 "product_attrs",
780 "product_desc",
781 "product_details",
782 "product_extended_attrs",
783 ]
784
785 async for item in self._fetch_library_items(
786 ",".join(response_groups), PODCAST_CONTENT_TYPES
787 ):
788 if podcast := await self._process_podcast_item(item):
789 yield podcast
790
791 async def get_podcast(self, asin: str, use_cache: bool = True) -> Podcast:
792 """
793 Fetch full podcast details by ASIN.
794
795 :param asin: The ASIN of the podcast.
796 :param use_cache: Whether to use cached data if available.
797 """
798 if use_cache:
799 cached_podcast = await self.mass.cache.get(
800 key=asin,
801 provider=self.provider_instance,
802 category=CACHE_CATEGORY_PODCAST,
803 default=None,
804 )
805 if cached_podcast is not None:
806 return self._parse_podcast(cached_podcast)
807
808 response = await self._call_api(
809 f"library/{asin}",
810 response_groups="""
811 contributors, media, price, product_attrs, product_desc, product_details,
812 product_extended_attrs, relationships
813 """,
814 )
815
816 if response is None:
817 raise MediaNotFoundError(f"Podcast with ASIN {asin} not found")
818
819 item_data = response.get("item")
820 if item_data is None:
821 raise MediaNotFoundError(f"Podcast data for ASIN {asin} is empty")
822
823 await self.mass.cache.set(
824 key=asin,
825 provider=self.provider_instance,
826 category=CACHE_CATEGORY_PODCAST,
827 data=item_data,
828 )
829 return self._parse_podcast(item_data)
830
831 async def get_podcast_episodes(self, podcast_asin: str) -> AsyncGenerator[PodcastEpisode]:
832 """
833 Fetch all episodes for a podcast.
834
835 :param podcast_asin: The ASIN of the parent podcast.
836 """
837 podcast = await self.get_podcast(podcast_asin)
838
839 # Fetch episodes - they're typically in relationships or we need to query children
840 response_groups = [
841 "contributors",
842 "media",
843 "product_attrs",
844 "product_desc",
845 "product_details",
846 "relationships",
847 ]
848
849 page = 1
850 page_size = 50
851 position = 0
852
853 while True:
854 # Query for children of the podcast parent
855 response = await self._call_api(
856 "library",
857 use_cache=False,
858 response_groups=",".join(response_groups),
859 parent_asin=podcast_asin,
860 page=page,
861 num_results=page_size,
862 )
863
864 items = response.get("items", [])
865 if not items:
866 break
867
868 for episode_data in items:
869 try:
870 episode = self._parse_podcast_episode(episode_data, podcast, position)
871 position += 1
872 yield episode
873 except Exception as exc:
874 asin = episode_data.get("asin", "unknown")
875 self.logger.warning(f"Error parsing podcast episode {asin}: {exc}")
876
877 page += 1
878 if len(items) < page_size:
879 break
880
881 async def get_podcast_episode(self, episode_asin: str) -> PodcastEpisode:
882 """
883 Fetch full podcast episode details by ASIN.
884
885 :param episode_asin: The ASIN of the podcast episode.
886 """
887 response = await self._call_api(
888 f"library/{episode_asin}",
889 response_groups="""
890 contributors, media, price, product_attrs, product_desc, product_details,
891 product_extended_attrs, relationships
892 """,
893 )
894
895 if response is None:
896 raise MediaNotFoundError(f"Podcast episode with ASIN {episode_asin} not found")
897
898 item_data = response.get("item")
899 if item_data is None:
900 raise MediaNotFoundError(f"Podcast episode data for ASIN {episode_asin} is empty")
901
902 # Try to get parent podcast info from relationships
903 podcast: Podcast | None = None
904 relationships = item_data.get("relationships", [])
905 for rel in relationships:
906 if rel.get("relationship_type") == "parent":
907 parent_asin = rel.get("asin")
908 if parent_asin:
909 with suppress(MediaNotFoundError):
910 podcast = await self.get_podcast(parent_asin)
911 break
912
913 return self._parse_podcast_episode(item_data, podcast, 0)
914
915 def _parse_podcast(self, podcast_data: dict[str, Any] | None) -> Podcast:
916 """
917 Parse podcast data from API response.
918
919 :param podcast_data: Raw podcast data from the Audible API.
920 """
921 if podcast_data is None:
922 self.logger.error("Received None podcast_data in _parse_podcast")
923 raise MediaNotFoundError("Podcast data not found")
924
925 asin = podcast_data.get("asin", "")
926 title = podcast_data.get("title", "")
927 publisher = podcast_data.get("publisher_name", "")
928
929 # Create podcast object
930 podcast = Podcast(
931 item_id=asin,
932 provider=self.provider_instance,
933 name=title,
934 publisher=publisher,
935 provider_mappings={
936 ProviderMapping(
937 item_id=asin,
938 provider_domain=self.provider_domain,
939 provider_instance=self.provider_instance,
940 )
941 },
942 )
943
944 # Set metadata
945 podcast.metadata.description = _html_to_txt(
946 str(
947 podcast_data.get("publisher_summary", "")
948 or podcast_data.get("extended_product_description", "")
949 )
950 )
951 podcast.metadata.languages = UniqueList([podcast_data.get("language") or ""])
952
953 # Set genres
954 podcast.metadata.genres = {
955 genre.replace("_", " ") for genre in (podcast_data.get("platinum_keywords") or [])
956 }
957
958 # Add images
959 image_path = podcast_data.get("product_images", {}).get("500")
960 podcast.metadata.images = UniqueList(self._create_images(image_path))
961
962 return podcast
963
964 def _parse_podcast_episode(
965 self,
966 episode_data: dict[str, Any] | None,
967 podcast: Podcast | None,
968 position: int,
969 ) -> PodcastEpisode:
970 """
971 Parse podcast episode data from API response.
972
973 :param episode_data: Raw episode data from the Audible API.
974 :param podcast: Parent podcast object (optional).
975 :param position: Position/index of the episode in the podcast.
976 """
977 if episode_data is None:
978 self.logger.error("Received None episode_data in _parse_podcast_episode")
979 raise MediaNotFoundError("Podcast episode data not found")
980
981 asin = episode_data.get("asin", "")
982 title = episode_data.get("title", "")
983
984 # Get duration from runtime_length_min
985 runtime_minutes = episode_data.get("runtime_length_min", 0)
986 duration = runtime_minutes * 60 if runtime_minutes else 0
987
988 # Create podcast reference - use Podcast object or create ItemMapping
989 podcast_ref: Podcast | ItemMapping
990 if podcast is not None:
991 podcast_ref = podcast
992 else:
993 # Try to get parent_asin from relationships for ItemMapping
994 parent_asin = ""
995 relationships = episode_data.get("relationships", [])
996 for rel in relationships:
997 if rel.get("relationship_type") == "parent":
998 parent_asin = rel.get("asin", "")
999 break
1000
1001 if not parent_asin:
1002 self.logger.warning(
1003 "No parent_asin found for podcast episode %s; parent podcast is unknown",
1004 asin,
1005 )
1006
1007 podcast_ref = ItemMapping(
1008 item_id=parent_asin or "",
1009 provider=self.provider_instance,
1010 name="Unknown Podcast",
1011 media_type=MediaType.PODCAST,
1012 )
1013
1014 # Create episode object
1015 episode = PodcastEpisode(
1016 item_id=asin,
1017 provider=self.provider_instance,
1018 name=title,
1019 duration=duration,
1020 position=position,
1021 podcast=podcast_ref,
1022 provider_mappings={
1023 ProviderMapping(
1024 item_id=asin,
1025 provider_domain=self.provider_domain,
1026 provider_instance=self.provider_instance,
1027 )
1028 },
1029 )
1030
1031 # Set metadata
1032 episode.metadata.description = _html_to_txt(
1033 str(
1034 episode_data.get("publisher_summary", "")
1035 or episode_data.get("extended_product_description", "")
1036 )
1037 )
1038
1039 # Add images
1040 image_path = episode_data.get("product_images", {}).get("500")
1041 episode.metadata.images = UniqueList(self._create_images(image_path))
1042
1043 return episode
1044
1045 async def get_authors(self) -> dict[str, str]:
1046 """
1047 Get all unique authors from the library.
1048
1049 Returns dict mapping author ASIN to author name.
1050 """
1051 authors: dict[str, str] = {}
1052 async for item in self._fetch_library_items(
1053 "contributors,product_attrs", AUDIOBOOK_CONTENT_TYPES
1054 ):
1055 for author in item.get("authors") or []:
1056 asin = author.get("asin")
1057 name = author.get("name")
1058 if asin and name:
1059 authors[asin] = name
1060 return authors
1061
1062 async def get_series(self) -> dict[str, str]:
1063 """
1064 Get all unique series from the library.
1065
1066 Returns dict mapping series ASIN to series title.
1067 """
1068 series: dict[str, str] = {}
1069 async for item in self._fetch_library_items(
1070 "series,product_attrs", AUDIOBOOK_CONTENT_TYPES
1071 ):
1072 for s in item.get("series") or []:
1073 asin = s.get("asin")
1074 title = s.get("title")
1075 if asin and title:
1076 series[asin] = title
1077 return series
1078
1079 async def get_narrators(self) -> dict[str, str]:
1080 """
1081 Get all unique narrators from the library.
1082
1083 Returns dict mapping narrator ASIN to narrator name.
1084 """
1085 narrators: dict[str, str] = {}
1086 async for item in self._fetch_library_items(
1087 "contributors,product_attrs", AUDIOBOOK_CONTENT_TYPES
1088 ):
1089 for narrator in item.get("narrators") or []:
1090 asin = narrator.get("asin")
1091 name = narrator.get("name")
1092 if asin and name:
1093 narrators[asin] = name
1094 return narrators
1095
1096 async def get_genres(self) -> set[str]:
1097 """Get all unique genres from the library."""
1098 genres: set[str] = set()
1099 async for item in self._fetch_library_items("product_attrs", AUDIOBOOK_CONTENT_TYPES):
1100 for keyword in item.get("thesaurus_subject_keywords") or []:
1101 genres.add(keyword.replace("_", " ").replace("-", " ").title())
1102 return genres
1103
1104 async def get_publishers(self) -> set[str]:
1105 """Get all unique publishers from the library."""
1106 publishers: set[str] = set()
1107 async for item in self._fetch_library_items("product_attrs", AUDIOBOOK_CONTENT_TYPES):
1108 publisher = item.get("publisher_name")
1109 if publisher:
1110 publishers.add(publisher)
1111 return publishers
1112
1113 async def get_audiobooks_by_author(self, author_asin: str) -> list[Audiobook]:
1114 """Get all audiobooks by a specific author, sorted by release date."""
1115 audiobooks: list[tuple[str, Audiobook]] = []
1116 async for item in self._fetch_library_items(
1117 "contributors,media,product_attrs,product_desc,series", AUDIOBOOK_CONTENT_TYPES
1118 ):
1119 for author in item.get("authors") or []:
1120 if author.get("asin") == author_asin:
1121 release_date = item.get("release_date") or "0000-00-00"
1122 audiobooks.append((release_date, self._parse_audiobook(item)))
1123 break
1124 audiobooks.sort(key=lambda x: x[0], reverse=True)
1125 return [book for _, book in audiobooks]
1126
1127 async def get_audiobooks_by_narrator(self, narrator_asin: str) -> list[Audiobook]:
1128 """Get all audiobooks by a specific narrator, sorted by release date."""
1129 audiobooks: list[tuple[str, Audiobook]] = []
1130 async for item in self._fetch_library_items(
1131 "contributors,media,product_attrs,product_desc,series", AUDIOBOOK_CONTENT_TYPES
1132 ):
1133 for narrator in item.get("narrators") or []:
1134 if narrator.get("asin") == narrator_asin:
1135 release_date = item.get("release_date") or "0000-00-00"
1136 audiobooks.append((release_date, self._parse_audiobook(item)))
1137 break
1138 audiobooks.sort(key=lambda x: x[0], reverse=True)
1139 return [book for _, book in audiobooks]
1140
1141 async def get_audiobooks_by_genre(self, genre: str) -> list[Audiobook]:
1142 """Get all audiobooks matching a genre, sorted by release date."""
1143 audiobooks: list[tuple[str, Audiobook]] = []
1144 genre_key = genre.lower().replace(" ", "_")
1145 genre_key_alt = genre.lower().replace(" ", "-")
1146 async for item in self._fetch_library_items(
1147 "contributors,media,product_attrs,product_desc,series", AUDIOBOOK_CONTENT_TYPES
1148 ):
1149 keywords = item.get("thesaurus_subject_keywords") or []
1150 if genre_key in keywords or genre_key_alt in keywords:
1151 release_date = item.get("release_date") or "0000-00-00"
1152 audiobooks.append((release_date, self._parse_audiobook(item)))
1153 audiobooks.sort(key=lambda x: x[0], reverse=True)
1154 return [book for _, book in audiobooks]
1155
1156 async def get_audiobooks_by_publisher(self, publisher: str) -> list[Audiobook]:
1157 """Get all audiobooks from a specific publisher, sorted by release date."""
1158 audiobooks: list[tuple[str, Audiobook]] = []
1159 async for item in self._fetch_library_items(
1160 "contributors,media,product_attrs,product_desc,series", AUDIOBOOK_CONTENT_TYPES
1161 ):
1162 if item.get("publisher_name") == publisher:
1163 release_date = item.get("release_date") or "0000-00-00"
1164 audiobooks.append((release_date, self._parse_audiobook(item)))
1165 audiobooks.sort(key=lambda x: x[0], reverse=True)
1166 return [book for _, book in audiobooks]
1167
1168 async def get_audiobooks_by_series(self, series_asin: str) -> list[Audiobook]:
1169 """Get all audiobooks in a specific series, ordered by sequence."""
1170 audiobooks: list[tuple[float, Audiobook]] = []
1171 async for item in self._fetch_library_items(
1172 "contributors,media,product_attrs,product_desc,series", AUDIOBOOK_CONTENT_TYPES
1173 ):
1174 for s in item.get("series") or []:
1175 if s.get("asin") == series_asin:
1176 sequence = s.get("sequence")
1177 try:
1178 seq_num = float(sequence) if sequence else 999
1179 except ValueError, TypeError:
1180 seq_num = 999
1181 audiobooks.append((seq_num, self._parse_audiobook(item)))
1182 break
1183 audiobooks.sort(key=lambda x: x[0])
1184 return [book for _, book in audiobooks]
1185
1186 async def deregister(self) -> None:
1187 """Deregister this provider from Audible."""
1188 await asyncio.to_thread(self.client.auth.deregister_device)
1189
1190
1191def _html_to_txt(html_text: str) -> str:
1192 txt = html.unescape(html_text)
1193 tags = re.findall("<[^>]+>", txt)
1194 for tag in tags:
1195 txt = txt.replace(tag, "")
1196 return txt
1197
1198
1199async def audible_get_auth_info(locale: str) -> tuple[str, str, str]:
1200 """
1201 Generate the login URL and auth info for Audible OAuth flow.
1202
1203 :param locale: The locale string (e.g., 'us', 'uk', 'de').
1204 :return: Tuple of (code_verifier, oauth_url, serial).
1205 """
1206 locale_obj = audible.localization.Locale(locale)
1207 code_verifier = await asyncio.to_thread(audible.login.create_code_verifier)
1208 oauth_url, serial = await asyncio.to_thread(
1209 audible.login.build_oauth_url,
1210 country_code=locale_obj.country_code,
1211 domain=locale_obj.domain,
1212 market_place_id=locale_obj.market_place_id,
1213 code_verifier=code_verifier,
1214 with_username=False,
1215 )
1216
1217 return code_verifier.decode(), oauth_url, serial
1218
1219
1220async def audible_custom_login(
1221 code_verifier: str, response_url: str, serial: str, locale: str
1222) -> audible.Authenticator:
1223 """
1224 Complete the authentication using the code_verifier, response_url, and serial.
1225
1226 :param code_verifier: The code verifier string used in OAuth flow.
1227 :param response_url: The response URL containing the authorization code.
1228 :param serial: The device serial number.
1229 :param locale: The locale string.
1230 :return: Audible Authenticator object.
1231 :raises LoginFailed: If authorization code is not found in the URL.
1232 """
1233 logger = logging.getLogger("audible_helper")
1234 auth = audible.Authenticator()
1235 auth.locale = audible.localization.Locale(locale)
1236
1237 response_url_parsed = urlparse(response_url)
1238 parsed_qs = parse_qs(response_url_parsed.query)
1239
1240 # Try multiple parameter names for authorization code
1241 # Audible may use different parameter names depending on the flow
1242 authorization_code = None
1243 for param_name in ["openid.oa2.authorization_code", "authorization_code", "code"]:
1244 if codes := parsed_qs.get(param_name):
1245 authorization_code = codes[0]
1246 logger.debug("Found authorization code in parameter: %s", param_name)
1247 break
1248
1249 if not authorization_code:
1250 available_params = list(parsed_qs.keys())
1251 raise LoginFailed(
1252 f"Authorization code not found in URL. "
1253 f"Expected 'openid.oa2.authorization_code' but found parameters: {available_params}"
1254 )
1255
1256 registration_data = await asyncio.to_thread(
1257 audible.register.register,
1258 authorization_code=authorization_code,
1259 code_verifier=code_verifier.encode(),
1260 domain=auth.locale.domain,
1261 serial=serial,
1262 )
1263 auth._update_attrs(**registration_data)
1264
1265 # Log what auth methods are available after registration
1266 if auth.adp_token and auth.device_private_key:
1267 logger.info("Registration successful with signing auth (stable)")
1268 else:
1269 logger.warning("Registration successful but signing auth not available")
1270
1271 return auth
1272
1273
1274async def check_file_exists(path: str | PathLike[str]) -> bool:
1275 """Async file exists check."""
1276 return await asyncio.to_thread(os.path.exists, path)
1277
1278
1279async def remove_file(path: str | PathLike[str]) -> None:
1280 """Async file delete."""
1281 await asyncio.to_thread(os.remove, path)
1282