/
/
1"""Helper for parsing and using audible api."""
2
3from __future__ import annotations
4
5import asyncio
6import hashlib
7import html
8import json
9import logging
10import os
11import re
12from collections.abc import AsyncGenerator
13from contextlib import suppress
14from datetime import UTC, datetime, timedelta
15from os import PathLike
16from typing import TYPE_CHECKING, Any
17from urllib.parse import parse_qs, urlparse
18
19import audible
20import audible.exceptions
21import audible.register
22from audible import AsyncClient
23
24if TYPE_CHECKING:
25 from aiohttp import ClientSession
26
27 from music_assistant.models.music_provider import MusicProvider
28from music_assistant_models.enums import ContentType, ImageType, MediaType, StreamType
29from music_assistant_models.errors import (
30 LoginFailed,
31 MediaNotFoundError,
32 ProviderUnavailableError,
33)
34from music_assistant_models.media_items import (
35 Audiobook,
36 AudioFormat,
37 ItemMapping,
38 MediaItemChapter,
39 MediaItemImage,
40 Podcast,
41 PodcastEpisode,
42 ProviderMapping,
43 UniqueList,
44)
45from music_assistant_models.streamdetails import StreamDetails
46
47from music_assistant.helpers.datetime import utc
48from music_assistant.mass import MusicAssistant
49
50CACHE_DOMAIN = "audible"
51CACHE_CATEGORY_API = 0
52CACHE_CATEGORY_AUDIOBOOK = 1
53CACHE_CATEGORY_CHAPTERS = 2
54CACHE_CATEGORY_PODCAST = 3
55CACHE_CATEGORY_PODCAST_EPISODES = 4
56
57# Content delivery types
58AUDIOBOOK_CONTENT_TYPES = ("SinglePartBook", "MultiPartBook")
59# Podcasts are normally reported as "PodcastParent", but (older) Audible Original
60# series are still reported with the legacy "Periodical" delivery type.
61PODCAST_CONTENT_TYPES = ("PodcastParent", "Periodical")
62
63_AUTH_CACHE: dict[str, audible.Authenticator] = {}
64
65
66async def refresh_access_token_compat(
67 refresh_token: str, domain: str, http_session: ClientSession, with_username: bool = False
68) -> dict[str, Any]:
69 """
70 Refresh tokens with compatibility for new Audible API format.
71
72 The Audible API changed from returning 'access_token' to 'actor_access_token'.
73 This function handles both formats for backward compatibility.
74
75 :param refresh_token: The refresh token obtained after device registration.
76 :param domain: The top level domain (e.g., com, de).
77 :param http_session: The HTTP client session to use for requests.
78 :param with_username: If True, use audible domain instead of amazon.
79 :return: Dict with access_token and expires timestamp.
80 """
81 logger = logging.getLogger("audible_helper")
82
83 body = {
84 "app_name": "Audible",
85 "app_version": "3.56.2",
86 "source_token": refresh_token,
87 "requested_token_type": "access_token",
88 "source_token_type": "refresh_token",
89 }
90
91 target_domain = "audible" if with_username else "amazon"
92 url = f"https://api.{target_domain}.{domain}/auth/token"
93
94 async with http_session.post(url, data=body) as resp:
95 resp.raise_for_status()
96 resp_dict = await resp.json()
97
98 expires_in_sec = int(resp_dict.get("expires_in", 3600))
99 expires = (utc() + timedelta(seconds=expires_in_sec)).timestamp()
100
101 # Handle new format (actor_access_token) or fall back to legacy (access_token)
102 access_token = resp_dict.get("actor_access_token") or resp_dict.get("access_token")
103
104 if not access_token:
105 logger.error("Token refresh response missing both actor_access_token and access_token")
106 raise LoginFailed("Token refresh failed: no access token in response")
107
108 logger.debug(
109 "Token refreshed successfully using %s format",
110 "new (actor)" if "actor_access_token" in resp_dict else "legacy",
111 )
112
113 return {"access_token": access_token, "expires": expires}
114
115
116async def cached_authenticator_from_file(
117 path: str, locale: str | None = None
118) -> audible.Authenticator:
119 """
120 Get an authenticator from file with caching and signing auth validation.
121
122 :param path: Path to the authenticator JSON file.
123 :param locale: The configured marketplace locale; when the stored file disagrees,
124 the configured locale wins and the file is corrected.
125 :return: The cached or loaded Authenticator instance.
126 """
127 logger = logging.getLogger("audible_helper")
128 auth = _AUTH_CACHE.get(path)
129 if auth is None:
130 logger.debug("Loading authenticator from file %s and caching it", path)
131 auth = await asyncio.to_thread(audible.Authenticator.from_file, path)
132
133 # Verify signing auth is available (not affected by API changes)
134 if auth.adp_token and auth.device_private_key:
135 logger.debug("Signing auth available - using stable RSA-signed requests")
136 else:
137 logger.warning(
138 "Signing auth not available - only bearer auth will work. "
139 "Consider re-authenticating for more stable auth."
140 )
141
142 _AUTH_CACHE[path] = auth
143
144 # auth files written by older versions can hold the marketplace from before a
145 # locale change; the configured locale is authoritative, so correct the file
146 if locale and (auth.locale is None or auth.locale.country_code != locale):
147 logger.warning(
148 "Marketplace in auth file (%s) does not match the configured locale (%s), correcting",
149 auth.locale.country_code if auth.locale else None,
150 locale,
151 )
152 auth.locale = audible.localization.Locale(locale)
153 await asyncio.to_thread(auth.to_file, path)
154
155 return auth
156
157
158def evict_cached_authenticator(path: str) -> None:
159 """
160 Drop the cached authenticator for the given file path, if any.
161
162 :param path: Path to the authenticator JSON file.
163 """
164 _AUTH_CACHE.pop(path, None)
165
166
167async def deregister_auth_file(path: str) -> None:
168 """
169 Deregister the virtual device registration stored in the given auth file.
170
171 :param path: Path to the authenticator JSON file.
172 """
173 auth = _AUTH_CACHE.pop(path, None)
174 if auth is None:
175 auth = await asyncio.to_thread(audible.Authenticator.from_file, path)
176 await asyncio.to_thread(auth.deregister_device)
177
178
179class AudibleHelper:
180 """Helper for parsing and using audible api."""
181
182 def __init__(
183 self,
184 mass: MusicAssistant,
185 client: AsyncClient,
186 provider_domain: str,
187 provider_instance: str,
188 provider: MusicProvider,
189 logger: logging.Logger | None = None,
190 ):
191 """
192 Initialize the Audible Helper.
193
194 :param mass: The MusicAssistant instance.
195 :param client: An authenticated Audible API client.
196 :param provider_domain: Domain of the owning provider.
197 :param provider_instance: Instance id of the owning provider.
198 :param provider: The owning provider, used to report library items it had to skip.
199 :param logger: Logger to use, defaults to a module level logger.
200 """
201 self.mass = mass
202 self.client = client
203 self.provider_domain = provider_domain
204 self.provider_instance = provider_instance
205 self.provider = provider
206 self.logger = logger or logging.getLogger("audible_helper")
207 self._acr_cache: dict[tuple[str, MediaType], str] = {}
208
209 async def _fetch_library_items(
210 self,
211 response_groups: str,
212 content_types: tuple[str, ...],
213 ) -> AsyncGenerator[dict[str, Any]]:
214 """Fetch items from the library with pagination."""
215 page = 1
216 page_size = 50
217 total_processed = 0
218 max_iterations = 100
219 iteration = 0
220
221 while iteration < max_iterations:
222 iteration += 1
223 self.logger.debug(
224 "Audible: Fetching library page %s (processed so far: %s)",
225 page,
226 total_processed,
227 )
228
229 library = await self._call_api(
230 "library",
231 use_cache=False,
232 response_groups=response_groups,
233 page=page,
234 num_results=page_size,
235 )
236
237 items = library.get("items", [])
238
239 if not items:
240 break
241
242 items_processed_this_page = 0
243 for item in items:
244 # Filter by content type if specified
245 if content_types and item.get("content_delivery_type") not in content_types:
246 continue
247
248 yield item
249 items_processed_this_page += 1
250 total_processed += 1
251
252 self.logger.debug(
253 "Audible: Processed %s items on page %s", items_processed_this_page, page
254 )
255
256 page += 1
257 if len(items) < page_size:
258 break
259
260 if iteration >= max_iterations:
261 self.logger.warning(
262 "Audible: Reached maximum iteration limit (%s) with %s items processed",
263 max_iterations,
264 total_processed,
265 )
266
267 async def _process_audiobook_item(self, audiobook_data: dict[str, Any]) -> Audiobook | None:
268 """Process a single audiobook item from the library."""
269 # Ensure asin is a valid string
270 asin = str(audiobook_data.get("asin", ""))
271 cached_book = None
272 if asin:
273 cached_book = await self.mass.cache.get(
274 key=asin,
275 provider=self.provider_instance,
276 category=CACHE_CATEGORY_AUDIOBOOK,
277 default=None,
278 )
279
280 try:
281 if cached_book is not None:
282 return self._parse_audiobook(cached_book)
283 return self._parse_audiobook(audiobook_data)
284 except Exception as exc:
285 self.provider.report_skipped_sync_item(MediaType.AUDIOBOOK, asin or None, exc)
286 return None
287
288 async def get_library(self) -> AsyncGenerator[Audiobook]:
289 """Fetch the user's library with pagination."""
290 response_groups = [
291 "contributors",
292 "media",
293 "product_attrs",
294 "product_desc",
295 "product_details",
296 "product_extended_attrs",
297 ]
298
299 async for item in self._fetch_library_items(
300 ",".join(response_groups), AUDIOBOOK_CONTENT_TYPES
301 ):
302 if album := await self._process_audiobook_item(item):
303 yield album
304
305 async def get_audiobook(self, asin: str, use_cache: bool = True) -> Audiobook:
306 """
307 Fetch the full audiobook by asin with all details including chapters.
308
309 This method fetches complete audiobook details including chapters and resume position.
310 Use this when the user requests full details for a specific audiobook.
311 """
312 if use_cache:
313 cached_book = await self.mass.cache.get(
314 key=asin,
315 provider=self.provider_instance,
316 category=CACHE_CATEGORY_AUDIOBOOK,
317 default=None,
318 )
319 if cached_book is not None:
320 book = self._parse_audiobook(cached_book)
321 # Enrich with chapters and resume position
322 await self._enrich_audiobook(book, asin)
323 return book
324 response = await self._call_api(
325 f"library/{asin}",
326 response_groups="""
327 contributors, media, price, product_attrs, product_desc, product_details,
328 product_extended_attrs,is_finished
329 """,
330 )
331
332 if response is None:
333 raise MediaNotFoundError(f"Audiobook with ASIN {asin} not found")
334
335 item_data = response.get("item")
336 if item_data is None:
337 raise MediaNotFoundError(f"Audiobook data for ASIN {asin} is empty")
338
339 await self.mass.cache.set(
340 key=asin,
341 provider=self.provider_instance,
342 category=CACHE_CATEGORY_AUDIOBOOK,
343 data=item_data,
344 )
345 book = self._parse_audiobook(item_data)
346 # Enrich with chapters and resume position
347 await self._enrich_audiobook(book, asin)
348 return book
349
350 async def _enrich_audiobook(self, book: Audiobook, asin: str) -> None:
351 """
352 Enrich audiobook with chapters and resume position.
353
354 This makes additional API calls and should only be used for full audiobook details,
355 not during library sync.
356 """
357 # Fetch chapters
358 chapters_data = await self._fetch_chapters(asin=asin)
359 if chapters_data:
360 chapters: list[MediaItemChapter] = [
361 self._parse_chapter_data(chapter, idx) for idx, chapter in enumerate(chapters_data)
362 ]
363 book.metadata.chapters = chapters
364 # Update duration from chapters if available (more accurate)
365 try:
366 duration = int(sum(chapter.get("length_ms", 0) for chapter in chapters_data) / 1000)
367 if duration > 0:
368 book.duration = duration
369 except Exception as exc:
370 self.logger.warning(f"Error calculating duration from chapters for {asin}: {exc}")
371
372 # Fetch resume position
373 book.resume_position_ms = await self.get_last_position(asin=asin)
374
375 async def get_stream(
376 self, asin: str, media_type: MediaType = MediaType.AUDIOBOOK
377 ) -> StreamDetails:
378 """
379 Get stream details for an audiobook or podcast episode.
380
381 :param asin: The ASIN of the content.
382 :param media_type: The type of media (audiobook or podcast episode).
383 """
384 if not asin:
385 self.logger.error("Invalid ASIN provided to get_stream")
386 raise ValueError("Invalid ASIN provided to get_stream")
387
388 duration = 0
389 # For audiobooks, try to get duration from chapters
390 if media_type == MediaType.AUDIOBOOK:
391 chapters = await self._fetch_chapters(asin=asin)
392 if chapters:
393 try:
394 duration = int(sum(chapter.get("length_ms", 0) for chapter in chapters) / 1000)
395 except Exception as exc:
396 self.logger.warning(f"Error calculating duration for ASIN {asin}: {exc}")
397
398 try:
399 # Podcasts use Mpeg (non-DRM MP3), audiobooks use HLS
400 if media_type == MediaType.PODCAST_EPISODE:
401 playback_info = await self.client.post(
402 f"content/{asin}/licenserequest",
403 body={
404 "consumption_type": "Streaming",
405 "drm_type": "Mpeg",
406 "quality": "High",
407 },
408 )
409 else:
410 playback_info = await self.client.post(
411 f"content/{asin}/licenserequest",
412 body={
413 "quality": "High",
414 "response_groups": "content_reference,certificate",
415 "consumption_type": "Streaming",
416 "supported_media_features": {
417 "codecs": ["mp4a.40.2", "mp4a.40.42"],
418 "drm_types": [
419 "Hls",
420 ],
421 },
422 "spatial": False,
423 },
424 )
425
426 content_license = playback_info.get("content_license", {})
427 if not content_license:
428 self.logger.error(f"No content_license in playback_info for ASIN {asin}")
429 raise ValueError(f"Missing content_license for ASIN {asin}")
430
431 content_metadata = content_license.get("content_metadata", {})
432 content_reference = content_metadata.get("content_reference", {})
433 size = content_reference.get("content_size_in_bytes", 0)
434
435 stream_url = content_license.get("license_response")
436 if not stream_url:
437 self.logger.error(f"No license_response (stream URL) for ASIN {asin}")
438 raise ValueError(f"Missing stream URL for ASIN {asin}")
439
440 acr = content_license.get("acr", "")
441 if acr:
442 self._acr_cache[(asin, media_type)] = acr
443
444 content_type = (
445 ContentType.MP3 if media_type == MediaType.PODCAST_EPISODE else ContentType.AAC
446 )
447 except Exception as exc:
448 self.logger.error(f"Error getting stream details for ASIN {asin}: {exc}")
449 raise ValueError(f"Failed to get stream details: {exc}") from exc
450
451 return StreamDetails(
452 provider=self.provider_instance,
453 size=size,
454 item_id=f"{asin}",
455 audio_format=AudioFormat(content_type=content_type),
456 media_type=media_type,
457 stream_type=StreamType.HTTP,
458 path=stream_url,
459 can_seek=True,
460 allow_seek=True,
461 duration=duration,
462 data={"acr": acr},
463 )
464
465 async def _fetch_chapters(self, asin: str) -> list[dict[str, Any]]:
466 """Fetch chapter data for an audiobook."""
467 if not asin or asin == "error":
468 self.logger.warning(
469 "Invalid ASIN provided to _fetch_chapters, returning empty chapter list"
470 )
471 return []
472
473 chapters_data: list[Any] = await self.mass.cache.get(
474 key=asin, provider=self.provider_instance, category=CACHE_CATEGORY_CHAPTERS, default=[]
475 )
476
477 if not chapters_data:
478 try:
479 response = await self._call_api(
480 f"content/{asin}/metadata",
481 response_groups="chapter_info, always-returned, content_reference, content_url",
482 chapter_titles_type="Flat",
483 )
484
485 if not response:
486 self.logger.warning(f"Failed to get metadata for ASIN {asin}")
487 return []
488
489 content_metadata = response.get("content_metadata")
490 if not content_metadata:
491 self.logger.warning(f"No content_metadata for ASIN {asin}")
492 return []
493
494 chapter_info = content_metadata.get("chapter_info")
495 if not chapter_info:
496 self.logger.warning(f"No chapter_info for ASIN {asin}")
497 return []
498
499 chapters_data = chapter_info.get("chapters") or []
500
501 await self.mass.cache.set(
502 key=asin,
503 data=chapters_data,
504 provider=self.provider_instance,
505 category=CACHE_CATEGORY_CHAPTERS,
506 )
507 except Exception as exc:
508 self.logger.error(f"Error fetching chapters for ASIN {asin}: {exc}")
509 chapters_data = []
510
511 return chapters_data
512
513 @staticmethod
514 def _parse_audible_timestamp(raw_ts: Any) -> datetime | None:
515 """
516 Parse an Audible timestamp value into a timezone-aware datetime.
517
518 :param raw_ts: The raw timestamp value from the Audible annotation payload.
519 """
520 if not raw_ts:
521 return None
522 try:
523 parsed = datetime.fromisoformat(str(raw_ts))
524 except ValueError, TypeError:
525 return None
526 if parsed.tzinfo is None:
527 parsed = parsed.replace(tzinfo=UTC)
528 return parsed
529
530 async def _fetch_last_position(self, asin: str) -> tuple[int, datetime | None] | None:
531 """
532 Fetch the last-heard position for a single ASIN from Audible.
533
534 :param asin: The audiobook ASIN to query.
535 """
536 response = await self._call_api("annotations/lastpositions", asins=asin)
537 if not response:
538 return None
539
540 annotations = response.get("asin_last_position_heard_annots")
541 if not annotations or not isinstance(annotations, list):
542 return None
543
544 annotation = annotations[0]
545 if not isinstance(annotation, dict):
546 return None
547
548 last_position = annotation.get("last_position_heard")
549 if not isinstance(last_position, dict):
550 return None
551
552 position_ms = int(last_position.get("position_ms", 0))
553
554 timestamp: datetime | None = None
555 for field in ("last_updated", "reported_time", "last_updated_time", "timestamp"):
556 timestamp = self._parse_audible_timestamp(
557 last_position.get(field) or annotation.get(field)
558 )
559 if timestamp is not None:
560 break
561
562 return position_ms, timestamp
563
564 async def get_last_position(self, asin: str) -> int:
565 """
566 Fetch the last-heard position in milliseconds for the given ASIN.
567
568 :param asin: The audiobook ASIN to query.
569 """
570 if not asin or asin == "error":
571 return 0
572 try:
573 result = await self._fetch_last_position(asin)
574 except (ProviderUnavailableError, KeyError, TypeError, ValueError) as exc:
575 self.logger.error("Error getting last position for ASIN %s: %s", asin, exc)
576 return 0
577 return result[0] if result else 0
578
579 async def set_last_position(
580 self, asin: str, pos: int, media_type: MediaType = MediaType.AUDIOBOOK
581 ) -> None:
582 """
583 Report last position to Audible.
584
585 :param asin: The content ID (audiobook or podcast episode).
586 :param pos: Position in seconds.
587 :param media_type: The type of media (audiobook or podcast episode).
588 """
589 if not asin or asin == "error" or pos <= 0:
590 return
591
592 try:
593 position_ms = pos * 1000
594
595 # Try to get ACR from cache first
596 acr = self._acr_cache.get((asin, media_type))
597 if not acr:
598 stream_details = await self.get_stream(asin=asin, media_type=media_type)
599 acr = stream_details.data.get("acr")
600
601 if not acr:
602 self.logger.warning(f"No ACR available for ASIN {asin}, cannot report position")
603 return
604
605 await self.client.put(
606 f"lastpositions/{asin}", body={"acr": acr, "asin": asin, "position_ms": position_ms}
607 )
608
609 self.logger.debug(f"Successfully reported position {position_ms}ms for ASIN {asin}")
610
611 except (KeyError, TypeError) as exc:
612 self.logger.error(
613 f"Error accessing data while reporting position for ASIN {asin}: {exc}"
614 )
615 except TimeoutError as exc:
616 self.logger.error(f"Timeout while reporting position for ASIN {asin}: {exc}")
617 except ConnectionError as exc:
618 self.logger.error(f"Connection error while reporting position for ASIN {asin}: {exc}")
619 except Exception as exc:
620 self.logger.error(f"Unexpected error reporting position for ASIN {asin}: {exc}")
621
622 async def get_audible_resume_position(self, asin: str) -> tuple[bool, int, datetime | None]:
623 """
624 Return resume state for the given ASIN from Audible.
625
626 :param asin: The audiobook ASIN to query.
627 """
628 if not asin or asin == "error":
629 raise NotImplementedError
630 try:
631 result = await self._fetch_last_position(asin)
632 except (ProviderUnavailableError, KeyError, TypeError, ValueError) as exc:
633 self.logger.debug("Audible lastpositions fetch failed for %s: %s", asin, exc)
634 raise NotImplementedError from exc
635 if not result or result[0] == 0:
636 raise NotImplementedError
637 position_ms, timestamp = result
638 return False, position_ms, timestamp
639
640 async def _call_api(self, path: str, **kwargs: Any) -> Any:
641 response = None
642 use_cache = kwargs.pop("use_cache", False)
643 params_str = json.dumps(kwargs, sort_keys=True)
644 params_hash = hashlib.md5(params_str.encode()).hexdigest()
645 cache_key_with_params = f"{path}:{params_hash}"
646 if use_cache:
647 response = await self.mass.cache.get(
648 key=cache_key_with_params,
649 provider=self.provider_instance,
650 category=CACHE_CATEGORY_API,
651 )
652 if not response:
653 try:
654 response = await self.client.get(path, **kwargs)
655 except audible.exceptions.RequestError as exc:
656 raise ProviderUnavailableError(
657 f"Audible API request failed for '{path}': {exc}"
658 ) from exc
659 await self.mass.cache.set(
660 key=cache_key_with_params, provider=self.provider_instance, data=response
661 )
662 return response
663
664 def _parse_contributors(
665 self, contributors_list: list[dict[str, Any]] | None, default_name: str
666 ) -> list[str]:
667 """Parse contributors (authors, narrators) from API response."""
668 result: list[str] = []
669 contributors: list[dict[str, Any]] = contributors_list or []
670 if isinstance(contributors, list):
671 for contributor in contributors:
672 if contributor and isinstance(contributor, dict):
673 result.append(contributor.get("name", default_name))
674 return result
675
676 def _create_images(self, image_path: str | None) -> list[MediaItemImage]:
677 """Create image objects if image path exists."""
678 images: list[MediaItemImage] = []
679 if image_path:
680 images.append(
681 MediaItemImage(
682 type=ImageType.THUMB,
683 path=image_path,
684 provider=self.provider_instance,
685 remotely_accessible=True,
686 )
687 )
688 images.append(
689 MediaItemImage(
690 type=ImageType.CLEARART,
691 path=image_path,
692 provider=self.provider_instance,
693 remotely_accessible=True,
694 )
695 )
696 return images
697
698 def _parse_chapter_data(self, chapter_data: dict[str, Any], index: int) -> MediaItemChapter:
699 """Parse chapter data into MediaItemChapter object."""
700 try:
701 start = int(chapter_data.get("start_offset_sec", 0))
702 except TypeError, ValueError:
703 start = 0
704
705 try:
706 length = int(chapter_data.get("length_ms", 0)) / 1000
707 except TypeError, ValueError:
708 length = 0
709
710 raw_title = chapter_data.get("title")
711 chapter_title: str
712 if raw_title is None:
713 chapter_title = f"Chapter {index + 1}"
714 elif isinstance(raw_title, str):
715 chapter_title = raw_title
716 else:
717 chapter_title = str(raw_title)
718
719 return MediaItemChapter(position=index, name=chapter_title, start=start, end=start + length)
720
721 def _parse_audiobook(self, audiobook_data: dict[str, Any] | None) -> Audiobook:
722 """
723 Parse audiobook data from API response.
724
725 NOTE: This is a pure parser - no API calls allowed here.
726 Chapters and resume position are fetched lazily when needed.
727 """
728 if audiobook_data is None:
729 self.logger.error("Received None audiobook_data in _parse_audiobook")
730 raise MediaNotFoundError("Audiobook data not found")
731
732 asin = audiobook_data.get("asin", "")
733 title = audiobook_data.get("title", "")
734
735 # Parse authors and narrators
736 narrators = self._parse_contributors(audiobook_data.get("narrators"), "Unknown Narrator")
737 authors = self._parse_contributors(audiobook_data.get("authors"), "Unknown Author")
738
739 # Get duration from runtime_length_min (provided by 'media' response group)
740 # Chapters are fetched lazily when streaming, not during library sync
741 runtime_minutes = audiobook_data.get("runtime_length_min", 0)
742 duration = runtime_minutes * 60 if runtime_minutes else 0
743
744 # Create audiobook object
745 book = Audiobook(
746 item_id=asin,
747 provider=self.provider_instance,
748 name=title,
749 duration=duration,
750 provider_mappings={
751 ProviderMapping(
752 item_id=asin,
753 provider_domain=self.provider_domain,
754 provider_instance=self.provider_instance,
755 )
756 },
757 publisher=audiobook_data.get("publisher_name"),
758 authors=UniqueList(authors),
759 narrators=UniqueList(narrators),
760 )
761
762 # Set metadata
763 book.metadata.copyright = audiobook_data.get("copyright")
764 book.metadata.description = _html_to_txt(
765 str(audiobook_data.get("extended_product_description", ""))
766 )
767 book.metadata.languages = UniqueList([audiobook_data.get("language") or ""])
768 if release_date := audiobook_data.get("release_date"):
769 with suppress(ValueError):
770 parsed_date = datetime.strptime(release_date, "%Y-%m-%d").astimezone(UTC)
771 book.metadata.release_date = parsed_date
772
773 # Set review if available
774 reviews = audiobook_data.get("editorial_reviews", [])
775 if reviews and reviews[0]:
776 book.metadata.review = _html_to_txt(str(reviews[0]))
777
778 # Set genres
779 book.metadata.genres = {
780 genre.replace("_", " ") for genre in (audiobook_data.get("platinum_keywords") or [])
781 }
782
783 # Add images
784 image_path = audiobook_data.get("product_images", {}).get("500")
785 book.metadata.images = UniqueList(self._create_images(image_path))
786
787 # Chapters are not fetched during parsing - they are fetched lazily when streaming
788 # This avoids N+1 API calls during library sync
789
790 return book
791
792 async def _process_podcast_item(self, podcast_data: dict[str, Any]) -> Podcast | None:
793 """Process a single podcast item from the library."""
794 asin = str(podcast_data.get("asin", ""))
795 cached_podcast = None
796 if asin:
797 cached_podcast = await self.mass.cache.get(
798 key=asin,
799 provider=self.provider_instance,
800 category=CACHE_CATEGORY_PODCAST,
801 default=None,
802 )
803
804 try:
805 if cached_podcast is not None:
806 return self._parse_podcast(cached_podcast)
807 return self._parse_podcast(podcast_data)
808 except Exception as exc:
809 self.provider.report_skipped_sync_item(MediaType.PODCAST, asin or None, exc)
810 return None
811
812 async def get_library_podcasts(self) -> AsyncGenerator[Podcast]:
813 """Fetch podcasts from the user's library with pagination."""
814 response_groups = [
815 "contributors",
816 "media",
817 "product_attrs",
818 "product_desc",
819 "product_details",
820 "product_extended_attrs",
821 ]
822
823 async for item in self._fetch_library_items(
824 ",".join(response_groups), PODCAST_CONTENT_TYPES
825 ):
826 if podcast := await self._process_podcast_item(item):
827 yield podcast
828
829 async def get_podcast(self, asin: str, use_cache: bool = True) -> Podcast:
830 """
831 Fetch full podcast details by ASIN.
832
833 :param asin: The ASIN of the podcast.
834 :param use_cache: Whether to use cached data if available.
835 """
836 if use_cache:
837 cached_podcast = await self.mass.cache.get(
838 key=asin,
839 provider=self.provider_instance,
840 category=CACHE_CATEGORY_PODCAST,
841 default=None,
842 )
843 if cached_podcast is not None:
844 return self._parse_podcast(cached_podcast)
845
846 response = await self._call_api(
847 f"library/{asin}",
848 response_groups="""
849 contributors, media, price, product_attrs, product_desc, product_details,
850 product_extended_attrs, relationships
851 """,
852 )
853
854 if response is None:
855 raise MediaNotFoundError(f"Podcast with ASIN {asin} not found")
856
857 item_data = response.get("item")
858 if item_data is None:
859 raise MediaNotFoundError(f"Podcast data for ASIN {asin} is empty")
860
861 await self.mass.cache.set(
862 key=asin,
863 provider=self.provider_instance,
864 category=CACHE_CATEGORY_PODCAST,
865 data=item_data,
866 )
867 return self._parse_podcast(item_data)
868
869 async def get_podcast_episodes(self, podcast_asin: str) -> AsyncGenerator[PodcastEpisode]:
870 """
871 Fetch all episodes for a podcast.
872
873 :param podcast_asin: The ASIN of the parent podcast.
874 """
875 podcast = await self.get_podcast(podcast_asin)
876
877 # Fetch episodes - they're typically in relationships or we need to query children
878 response_groups = [
879 "contributors",
880 "media",
881 "product_attrs",
882 "product_desc",
883 "product_details",
884 "relationships",
885 ]
886
887 page = 1
888 page_size = 50
889 position = 0
890
891 while True:
892 # Query for children of the podcast parent
893 response = await self._call_api(
894 "library",
895 use_cache=False,
896 response_groups=",".join(response_groups),
897 parent_asin=podcast_asin,
898 page=page,
899 num_results=page_size,
900 )
901
902 items = response.get("items", [])
903 if not items:
904 break
905
906 for episode_data in items:
907 try:
908 episode = self._parse_podcast_episode(episode_data, podcast, position)
909 position += 1
910 yield episode
911 except Exception as exc:
912 asin = episode_data.get("asin", "unknown")
913 self.logger.warning(f"Error parsing podcast episode {asin}: {exc}")
914
915 page += 1
916 if len(items) < page_size:
917 break
918
919 async def get_podcast_episode(self, episode_asin: str) -> PodcastEpisode:
920 """
921 Fetch full podcast episode details by ASIN.
922
923 :param episode_asin: The ASIN of the podcast episode.
924 """
925 response = await self._call_api(
926 f"library/{episode_asin}",
927 response_groups="""
928 contributors, media, price, product_attrs, product_desc, product_details,
929 product_extended_attrs, relationships
930 """,
931 )
932
933 if response is None:
934 raise MediaNotFoundError(f"Podcast episode with ASIN {episode_asin} not found")
935
936 item_data = response.get("item")
937 if item_data is None:
938 raise MediaNotFoundError(f"Podcast episode data for ASIN {episode_asin} is empty")
939
940 # Try to get parent podcast info from relationships
941 podcast: Podcast | None = None
942 relationships = item_data.get("relationships", [])
943 for rel in relationships:
944 if rel.get("relationship_type") == "parent":
945 parent_asin = rel.get("asin")
946 if parent_asin:
947 with suppress(MediaNotFoundError):
948 podcast = await self.get_podcast(parent_asin)
949 break
950
951 return self._parse_podcast_episode(item_data, podcast, 0)
952
953 def _parse_podcast(self, podcast_data: dict[str, Any] | None) -> Podcast:
954 """
955 Parse podcast data from API response.
956
957 :param podcast_data: Raw podcast data from the Audible API.
958 """
959 if podcast_data is None:
960 self.logger.error("Received None podcast_data in _parse_podcast")
961 raise MediaNotFoundError("Podcast data not found")
962
963 asin = podcast_data.get("asin", "")
964 title = podcast_data.get("title", "")
965 publisher = podcast_data.get("publisher_name", "")
966
967 # Create podcast object
968 podcast = Podcast(
969 item_id=asin,
970 provider=self.provider_instance,
971 name=title,
972 publisher=publisher,
973 provider_mappings={
974 ProviderMapping(
975 item_id=asin,
976 provider_domain=self.provider_domain,
977 provider_instance=self.provider_instance,
978 )
979 },
980 )
981
982 # Set metadata
983 podcast.metadata.description = _html_to_txt(
984 str(
985 podcast_data.get("publisher_summary", "")
986 or podcast_data.get("extended_product_description", "")
987 )
988 )
989 podcast.metadata.languages = UniqueList([podcast_data.get("language") or ""])
990
991 # Set genres
992 podcast.metadata.genres = {
993 genre.replace("_", " ") for genre in (podcast_data.get("platinum_keywords") or [])
994 }
995
996 # Add images
997 image_path = podcast_data.get("product_images", {}).get("500")
998 podcast.metadata.images = UniqueList(self._create_images(image_path))
999
1000 return podcast
1001
1002 def _parse_podcast_episode(
1003 self,
1004 episode_data: dict[str, Any] | None,
1005 podcast: Podcast | None,
1006 position: int,
1007 ) -> PodcastEpisode:
1008 """
1009 Parse podcast episode data from API response.
1010
1011 :param episode_data: Raw episode data from the Audible API.
1012 :param podcast: Parent podcast object (optional).
1013 :param position: Position/index of the episode in the podcast.
1014 """
1015 if episode_data is None:
1016 self.logger.error("Received None episode_data in _parse_podcast_episode")
1017 raise MediaNotFoundError("Podcast episode data not found")
1018
1019 asin = episode_data.get("asin", "")
1020 title = episode_data.get("title", "")
1021
1022 # Get duration from runtime_length_min
1023 runtime_minutes = episode_data.get("runtime_length_min", 0)
1024 duration = runtime_minutes * 60 if runtime_minutes else 0
1025
1026 # Create podcast reference - use Podcast object or create ItemMapping
1027 podcast_ref: Podcast | ItemMapping
1028 if podcast is not None:
1029 podcast_ref = podcast
1030 else:
1031 # Try to get parent_asin from relationships for ItemMapping
1032 parent_asin = ""
1033 relationships = episode_data.get("relationships", [])
1034 for rel in relationships:
1035 if rel.get("relationship_type") == "parent":
1036 parent_asin = rel.get("asin", "")
1037 break
1038
1039 if not parent_asin:
1040 self.logger.warning(
1041 "No parent_asin found for podcast episode %s; parent podcast is unknown",
1042 asin,
1043 )
1044
1045 podcast_ref = ItemMapping(
1046 item_id=parent_asin or "",
1047 provider=self.provider_instance,
1048 name="Unknown Podcast",
1049 media_type=MediaType.PODCAST,
1050 )
1051
1052 # Create episode object
1053 episode = PodcastEpisode(
1054 item_id=asin,
1055 provider=self.provider_instance,
1056 name=title,
1057 duration=duration,
1058 position=position,
1059 podcast=podcast_ref,
1060 provider_mappings={
1061 ProviderMapping(
1062 item_id=asin,
1063 provider_domain=self.provider_domain,
1064 provider_instance=self.provider_instance,
1065 )
1066 },
1067 )
1068
1069 # Set metadata
1070 episode.metadata.description = _html_to_txt(
1071 str(
1072 episode_data.get("publisher_summary", "")
1073 or episode_data.get("extended_product_description", "")
1074 )
1075 )
1076
1077 # Add images
1078 image_path = episode_data.get("product_images", {}).get("500")
1079 episode.metadata.images = UniqueList(self._create_images(image_path))
1080
1081 return episode
1082
1083 async def get_authors(self) -> dict[str, str]:
1084 """
1085 Get all unique authors from the library.
1086
1087 Returns dict mapping author ASIN to author name.
1088 """
1089 authors: dict[str, str] = {}
1090 async for item in self._fetch_library_items(
1091 "contributors,product_attrs", AUDIOBOOK_CONTENT_TYPES
1092 ):
1093 for author in item.get("authors") or []:
1094 asin = author.get("asin")
1095 name = author.get("name")
1096 if asin and name:
1097 authors[asin] = name
1098 return authors
1099
1100 async def get_series(self) -> dict[str, str]:
1101 """
1102 Get all unique series from the library.
1103
1104 Returns dict mapping series ASIN to series title.
1105 """
1106 series: dict[str, str] = {}
1107 async for item in self._fetch_library_items(
1108 "series,product_attrs", AUDIOBOOK_CONTENT_TYPES
1109 ):
1110 for s in item.get("series") or []:
1111 asin = s.get("asin")
1112 title = s.get("title")
1113 if asin and title:
1114 series[asin] = title
1115 return series
1116
1117 async def get_narrators(self) -> dict[str, str]:
1118 """
1119 Get all unique narrators from the library.
1120
1121 Returns dict mapping narrator ASIN to narrator name.
1122 """
1123 narrators: dict[str, str] = {}
1124 async for item in self._fetch_library_items(
1125 "contributors,product_attrs", AUDIOBOOK_CONTENT_TYPES
1126 ):
1127 for narrator in item.get("narrators") or []:
1128 asin = narrator.get("asin")
1129 name = narrator.get("name")
1130 if asin and name:
1131 narrators[asin] = name
1132 return narrators
1133
1134 async def get_genres(self) -> set[str]:
1135 """Get all unique genres from the library."""
1136 genres: set[str] = set()
1137 async for item in self._fetch_library_items("product_attrs", AUDIOBOOK_CONTENT_TYPES):
1138 for keyword in item.get("thesaurus_subject_keywords") or []:
1139 genres.add(keyword.replace("_", " ").replace("-", " ").title())
1140 return genres
1141
1142 async def get_publishers(self) -> set[str]:
1143 """Get all unique publishers from the library."""
1144 publishers: set[str] = set()
1145 async for item in self._fetch_library_items("product_attrs", AUDIOBOOK_CONTENT_TYPES):
1146 publisher = item.get("publisher_name")
1147 if publisher:
1148 publishers.add(publisher)
1149 return publishers
1150
1151 async def get_audiobooks_by_author(self, author_asin: str) -> list[Audiobook]:
1152 """Get all audiobooks by a specific author, sorted by release date."""
1153 audiobooks: list[tuple[str, Audiobook]] = []
1154 async for item in self._fetch_library_items(
1155 "contributors,media,product_attrs,product_desc,series", AUDIOBOOK_CONTENT_TYPES
1156 ):
1157 for author in item.get("authors") or []:
1158 if author.get("asin") == author_asin:
1159 release_date = item.get("release_date") or "0000-00-00"
1160 audiobooks.append((release_date, self._parse_audiobook(item)))
1161 break
1162 audiobooks.sort(key=lambda x: x[0], reverse=True)
1163 return [book for _, book in audiobooks]
1164
1165 async def get_audiobooks_by_narrator(self, narrator_asin: str) -> list[Audiobook]:
1166 """Get all audiobooks by a specific narrator, sorted by release date."""
1167 audiobooks: list[tuple[str, Audiobook]] = []
1168 async for item in self._fetch_library_items(
1169 "contributors,media,product_attrs,product_desc,series", AUDIOBOOK_CONTENT_TYPES
1170 ):
1171 for narrator in item.get("narrators") or []:
1172 if narrator.get("asin") == narrator_asin:
1173 release_date = item.get("release_date") or "0000-00-00"
1174 audiobooks.append((release_date, self._parse_audiobook(item)))
1175 break
1176 audiobooks.sort(key=lambda x: x[0], reverse=True)
1177 return [book for _, book in audiobooks]
1178
1179 async def get_audiobooks_by_genre(self, genre: str) -> list[Audiobook]:
1180 """Get all audiobooks matching a genre, sorted by release date."""
1181 audiobooks: list[tuple[str, Audiobook]] = []
1182 genre_key = genre.lower().replace(" ", "_")
1183 genre_key_alt = genre.lower().replace(" ", "-")
1184 async for item in self._fetch_library_items(
1185 "contributors,media,product_attrs,product_desc,series", AUDIOBOOK_CONTENT_TYPES
1186 ):
1187 keywords = item.get("thesaurus_subject_keywords") or []
1188 if genre_key in keywords or genre_key_alt in keywords:
1189 release_date = item.get("release_date") or "0000-00-00"
1190 audiobooks.append((release_date, self._parse_audiobook(item)))
1191 audiobooks.sort(key=lambda x: x[0], reverse=True)
1192 return [book for _, book in audiobooks]
1193
1194 async def get_audiobooks_by_publisher(self, publisher: str) -> list[Audiobook]:
1195 """Get all audiobooks from a specific publisher, sorted by release date."""
1196 audiobooks: list[tuple[str, Audiobook]] = []
1197 async for item in self._fetch_library_items(
1198 "contributors,media,product_attrs,product_desc,series", AUDIOBOOK_CONTENT_TYPES
1199 ):
1200 if item.get("publisher_name") == publisher:
1201 release_date = item.get("release_date") or "0000-00-00"
1202 audiobooks.append((release_date, self._parse_audiobook(item)))
1203 audiobooks.sort(key=lambda x: x[0], reverse=True)
1204 return [book for _, book in audiobooks]
1205
1206 async def get_audiobooks_by_series(self, series_asin: str) -> list[Audiobook]:
1207 """Get all audiobooks in a specific series, ordered by sequence."""
1208 audiobooks: list[tuple[float, Audiobook]] = []
1209 async for item in self._fetch_library_items(
1210 "contributors,media,product_attrs,product_desc,series", AUDIOBOOK_CONTENT_TYPES
1211 ):
1212 for s in item.get("series") or []:
1213 if s.get("asin") == series_asin:
1214 sequence = s.get("sequence")
1215 try:
1216 seq_num = float(sequence) if sequence else 999
1217 except ValueError, TypeError:
1218 seq_num = 999
1219 audiobooks.append((seq_num, self._parse_audiobook(item)))
1220 break
1221 audiobooks.sort(key=lambda x: x[0])
1222 return [book for _, book in audiobooks]
1223
1224 async def deregister(self) -> None:
1225 """Deregister this provider from Audible."""
1226 await asyncio.to_thread(self.client.auth.deregister_device)
1227
1228
1229def _html_to_txt(html_text: str) -> str:
1230 txt = html.unescape(html_text)
1231 tags = re.findall("<[^>]+>", txt)
1232 for tag in tags:
1233 txt = txt.replace(tag, "")
1234 return txt
1235
1236
1237async def audible_get_auth_info(locale: str) -> tuple[str, str, str]:
1238 """
1239 Generate the login URL and auth info for Audible OAuth flow.
1240
1241 :param locale: The locale string (e.g., 'us', 'uk', 'de').
1242 :return: Tuple of (code_verifier, oauth_url, serial).
1243 """
1244 locale_obj = audible.localization.Locale(locale)
1245 code_verifier = await asyncio.to_thread(audible.login.create_code_verifier)
1246 oauth_url, serial = await asyncio.to_thread(
1247 audible.login.build_oauth_url,
1248 country_code=locale_obj.country_code,
1249 domain=locale_obj.domain,
1250 market_place_id=locale_obj.market_place_id,
1251 code_verifier=code_verifier,
1252 with_username=False,
1253 )
1254
1255 return code_verifier.decode(), oauth_url, serial
1256
1257
1258async def audible_custom_login(
1259 code_verifier: str, response_url: str, serial: str, locale: str
1260) -> audible.Authenticator:
1261 """
1262 Complete the authentication using the code_verifier, response_url, and serial.
1263
1264 :param code_verifier: The code verifier string used in OAuth flow.
1265 :param response_url: The response URL containing the authorization code.
1266 :param serial: The device serial number.
1267 :param locale: The locale string.
1268 :return: Audible Authenticator object.
1269 :raises LoginFailed: If authorization code is not found in the URL.
1270 """
1271 logger = logging.getLogger("audible_helper")
1272 auth = audible.Authenticator()
1273 auth.locale = audible.localization.Locale(locale)
1274
1275 response_url_parsed = urlparse(response_url)
1276 parsed_qs = parse_qs(response_url_parsed.query)
1277
1278 # Try multiple parameter names for authorization code
1279 # Audible may use different parameter names depending on the flow
1280 authorization_code = None
1281 for param_name in ["openid.oa2.authorization_code", "authorization_code", "code"]:
1282 if codes := parsed_qs.get(param_name):
1283 authorization_code = codes[0]
1284 logger.debug("Found authorization code in parameter: %s", param_name)
1285 break
1286
1287 if not authorization_code:
1288 available_params = list(parsed_qs.keys())
1289 raise LoginFailed(
1290 f"Authorization code not found in URL. "
1291 f"Expected 'openid.oa2.authorization_code' but found parameters: {available_params}"
1292 )
1293
1294 registration_data = await asyncio.to_thread(
1295 audible.register.register,
1296 authorization_code=authorization_code,
1297 code_verifier=code_verifier.encode(),
1298 domain=auth.locale.domain,
1299 serial=serial,
1300 )
1301 auth._update_attrs(**registration_data)
1302
1303 # Log what auth methods are available after registration
1304 if auth.adp_token and auth.device_private_key:
1305 logger.info("Registration successful with signing auth (stable)")
1306 else:
1307 logger.warning("Registration successful but signing auth not available")
1308
1309 return auth
1310
1311
1312async def check_file_exists(path: str | PathLike[str]) -> bool:
1313 """Async file exists check."""
1314 return await asyncio.to_thread(os.path.exists, path)
1315
1316
1317async def remove_file(path: str | PathLike[str]) -> None:
1318 """Async file delete."""
1319 await asyncio.to_thread(os.remove, path)
1320