/
/
1"""Helper for parsing and using audible api."""
2
3from __future__ import annotations
4
5import asyncio
6import hashlib
7import html
8import json
9import logging
10import os
11import re
12from collections.abc import AsyncGenerator
13from contextlib import suppress
14from datetime import UTC, datetime, timedelta
15from os import PathLike
16from typing import TYPE_CHECKING, Any
17from urllib.parse import parse_qs, urlparse
18
19import audible
20import audible.exceptions
21import audible.register
22from audible import AsyncClient
23
24if TYPE_CHECKING:
25 from aiohttp import ClientSession
26
27 from music_assistant.models.music_provider import MusicProvider
28from music_assistant_models.enums import ContentType, ImageType, MediaType, StreamType
29from music_assistant_models.errors import (
30 LoginFailed,
31 MediaNotFoundError,
32 ProviderUnavailableError,
33)
34from music_assistant_models.media_items import (
35 Audiobook,
36 AudioFormat,
37 ItemMapping,
38 MediaItemChapter,
39 MediaItemImage,
40 Podcast,
41 PodcastEpisode,
42 ProviderMapping,
43 UniqueList,
44)
45from music_assistant_models.streamdetails import StreamDetails
46
47from music_assistant.helpers.datetime import utc
48from music_assistant.mass import MusicAssistant
49
50CACHE_DOMAIN = "audible"
51CACHE_CATEGORY_API = 0
52CACHE_CATEGORY_AUDIOBOOK = 1
53CACHE_CATEGORY_CHAPTERS = 2
54CACHE_CATEGORY_PODCAST = 3
55CACHE_CATEGORY_PODCAST_EPISODES = 4
56
57# Content delivery types
58AUDIOBOOK_CONTENT_TYPES = ("SinglePartBook", "MultiPartBook")
59# Podcasts are normally reported as "PodcastParent", but (older) Audible Original
60# series are still reported with the legacy "Periodical" delivery type.
61PODCAST_CONTENT_TYPES = ("PodcastParent", "Periodical")
62
63_AUTH_CACHE: dict[str, audible.Authenticator] = {}
64
65
66async def refresh_access_token_compat(
67 refresh_token: str, domain: str, http_session: ClientSession, with_username: bool = False
68) -> dict[str, Any]:
69 """
70 Refresh tokens with compatibility for new Audible API format.
71
72 The Audible API changed from returning 'access_token' to 'actor_access_token'.
73 This function handles both formats for backward compatibility.
74
75 :param refresh_token: The refresh token obtained after device registration.
76 :param domain: The top level domain (e.g., com, de).
77 :param http_session: The HTTP client session to use for requests.
78 :param with_username: If True, use audible domain instead of amazon.
79 :return: Dict with access_token and expires timestamp.
80 """
81 logger = logging.getLogger("audible_helper")
82
83 body = {
84 "app_name": "Audible",
85 "app_version": "3.56.2",
86 "source_token": refresh_token,
87 "requested_token_type": "access_token",
88 "source_token_type": "refresh_token",
89 }
90
91 target_domain = "audible" if with_username else "amazon"
92 url = f"https://api.{target_domain}.{domain}/auth/token"
93
94 async with http_session.post(url, data=body) as resp:
95 resp.raise_for_status()
96 resp_dict = await resp.json()
97
98 expires_in_sec = int(resp_dict.get("expires_in", 3600))
99 expires = (utc() + timedelta(seconds=expires_in_sec)).timestamp()
100
101 # Handle new format (actor_access_token) or fall back to legacy (access_token)
102 access_token = resp_dict.get("actor_access_token") or resp_dict.get("access_token")
103
104 if not access_token:
105 logger.error("Token refresh response missing both actor_access_token and access_token")
106 raise LoginFailed("Token refresh failed: no access token in response")
107
108 logger.debug(
109 "Token refreshed successfully using %s format",
110 "new (actor)" if "actor_access_token" in resp_dict else "legacy",
111 )
112
113 return {"access_token": access_token, "expires": expires}
114
115
116async def cached_authenticator_from_file(path: str) -> audible.Authenticator:
117 """
118 Get an authenticator from file with caching and signing auth validation.
119
120 :param path: Path to the authenticator JSON file.
121 :return: The cached or loaded Authenticator instance.
122 """
123 logger = logging.getLogger("audible_helper")
124 if path in _AUTH_CACHE:
125 return _AUTH_CACHE[path]
126
127 logger.debug("Loading authenticator from file %s and caching it", path)
128 auth = await asyncio.to_thread(audible.Authenticator.from_file, path)
129
130 # Verify signing auth is available (not affected by API changes)
131 if auth.adp_token and auth.device_private_key:
132 logger.debug("Signing auth available - using stable RSA-signed requests")
133 else:
134 logger.warning(
135 "Signing auth not available - only bearer auth will work. "
136 "Consider re-authenticating for more stable auth."
137 )
138
139 _AUTH_CACHE[path] = auth
140 return auth
141
142
143class AudibleHelper:
144 """Helper for parsing and using audible api."""
145
146 def __init__(
147 self,
148 mass: MusicAssistant,
149 client: AsyncClient,
150 provider_domain: str,
151 provider_instance: str,
152 provider: MusicProvider,
153 logger: logging.Logger | None = None,
154 ):
155 """
156 Initialize the Audible Helper.
157
158 :param mass: The MusicAssistant instance.
159 :param client: An authenticated Audible API client.
160 :param provider_domain: Domain of the owning provider.
161 :param provider_instance: Instance id of the owning provider.
162 :param provider: The owning provider, used to report library items it had to skip.
163 :param logger: Logger to use, defaults to a module level logger.
164 """
165 self.mass = mass
166 self.client = client
167 self.provider_domain = provider_domain
168 self.provider_instance = provider_instance
169 self.provider = provider
170 self.logger = logger or logging.getLogger("audible_helper")
171 self._acr_cache: dict[tuple[str, MediaType], str] = {}
172
173 async def _fetch_library_items(
174 self,
175 response_groups: str,
176 content_types: tuple[str, ...],
177 ) -> AsyncGenerator[dict[str, Any]]:
178 """Fetch items from the library with pagination."""
179 page = 1
180 page_size = 50
181 total_processed = 0
182 max_iterations = 100
183 iteration = 0
184
185 while iteration < max_iterations:
186 iteration += 1
187 self.logger.debug(
188 "Audible: Fetching library page %s (processed so far: %s)",
189 page,
190 total_processed,
191 )
192
193 library = await self._call_api(
194 "library",
195 use_cache=False,
196 response_groups=response_groups,
197 page=page,
198 num_results=page_size,
199 )
200
201 items = library.get("items", [])
202
203 if not items:
204 break
205
206 items_processed_this_page = 0
207 for item in items:
208 # Filter by content type if specified
209 if content_types and item.get("content_delivery_type") not in content_types:
210 continue
211
212 yield item
213 items_processed_this_page += 1
214 total_processed += 1
215
216 self.logger.debug(
217 "Audible: Processed %s items on page %s", items_processed_this_page, page
218 )
219
220 page += 1
221 if len(items) < page_size:
222 break
223
224 if iteration >= max_iterations:
225 self.logger.warning(
226 "Audible: Reached maximum iteration limit (%s) with %s items processed",
227 max_iterations,
228 total_processed,
229 )
230
231 async def _process_audiobook_item(self, audiobook_data: dict[str, Any]) -> Audiobook | None:
232 """Process a single audiobook item from the library."""
233 # Ensure asin is a valid string
234 asin = str(audiobook_data.get("asin", ""))
235 cached_book = None
236 if asin:
237 cached_book = await self.mass.cache.get(
238 key=asin,
239 provider=self.provider_instance,
240 category=CACHE_CATEGORY_AUDIOBOOK,
241 default=None,
242 )
243
244 try:
245 if cached_book is not None:
246 return self._parse_audiobook(cached_book)
247 return self._parse_audiobook(audiobook_data)
248 except Exception as exc:
249 self.provider.report_skipped_sync_item(MediaType.AUDIOBOOK, asin or None, exc)
250 return None
251
252 async def get_library(self) -> AsyncGenerator[Audiobook]:
253 """Fetch the user's library with pagination."""
254 response_groups = [
255 "contributors",
256 "media",
257 "product_attrs",
258 "product_desc",
259 "product_details",
260 "product_extended_attrs",
261 ]
262
263 async for item in self._fetch_library_items(
264 ",".join(response_groups), AUDIOBOOK_CONTENT_TYPES
265 ):
266 if album := await self._process_audiobook_item(item):
267 yield album
268
269 async def get_audiobook(self, asin: str, use_cache: bool = True) -> Audiobook:
270 """
271 Fetch the full audiobook by asin with all details including chapters.
272
273 This method fetches complete audiobook details including chapters and resume position.
274 Use this when the user requests full details for a specific audiobook.
275 """
276 if use_cache:
277 cached_book = await self.mass.cache.get(
278 key=asin,
279 provider=self.provider_instance,
280 category=CACHE_CATEGORY_AUDIOBOOK,
281 default=None,
282 )
283 if cached_book is not None:
284 book = self._parse_audiobook(cached_book)
285 # Enrich with chapters and resume position
286 await self._enrich_audiobook(book, asin)
287 return book
288 response = await self._call_api(
289 f"library/{asin}",
290 response_groups="""
291 contributors, media, price, product_attrs, product_desc, product_details,
292 product_extended_attrs,is_finished
293 """,
294 )
295
296 if response is None:
297 raise MediaNotFoundError(f"Audiobook with ASIN {asin} not found")
298
299 item_data = response.get("item")
300 if item_data is None:
301 raise MediaNotFoundError(f"Audiobook data for ASIN {asin} is empty")
302
303 await self.mass.cache.set(
304 key=asin,
305 provider=self.provider_instance,
306 category=CACHE_CATEGORY_AUDIOBOOK,
307 data=item_data,
308 )
309 book = self._parse_audiobook(item_data)
310 # Enrich with chapters and resume position
311 await self._enrich_audiobook(book, asin)
312 return book
313
314 async def _enrich_audiobook(self, book: Audiobook, asin: str) -> None:
315 """
316 Enrich audiobook with chapters and resume position.
317
318 This makes additional API calls and should only be used for full audiobook details,
319 not during library sync.
320 """
321 # Fetch chapters
322 chapters_data = await self._fetch_chapters(asin=asin)
323 if chapters_data:
324 chapters: list[MediaItemChapter] = [
325 self._parse_chapter_data(chapter, idx) for idx, chapter in enumerate(chapters_data)
326 ]
327 book.metadata.chapters = chapters
328 # Update duration from chapters if available (more accurate)
329 try:
330 duration = int(sum(chapter.get("length_ms", 0) for chapter in chapters_data) / 1000)
331 if duration > 0:
332 book.duration = duration
333 except Exception as exc:
334 self.logger.warning(f"Error calculating duration from chapters for {asin}: {exc}")
335
336 # Fetch resume position
337 book.resume_position_ms = await self.get_last_position(asin=asin)
338
339 async def get_stream(
340 self, asin: str, media_type: MediaType = MediaType.AUDIOBOOK
341 ) -> StreamDetails:
342 """
343 Get stream details for an audiobook or podcast episode.
344
345 :param asin: The ASIN of the content.
346 :param media_type: The type of media (audiobook or podcast episode).
347 """
348 if not asin:
349 self.logger.error("Invalid ASIN provided to get_stream")
350 raise ValueError("Invalid ASIN provided to get_stream")
351
352 duration = 0
353 # For audiobooks, try to get duration from chapters
354 if media_type == MediaType.AUDIOBOOK:
355 chapters = await self._fetch_chapters(asin=asin)
356 if chapters:
357 try:
358 duration = int(sum(chapter.get("length_ms", 0) for chapter in chapters) / 1000)
359 except Exception as exc:
360 self.logger.warning(f"Error calculating duration for ASIN {asin}: {exc}")
361
362 try:
363 # Podcasts use Mpeg (non-DRM MP3), audiobooks use HLS
364 if media_type == MediaType.PODCAST_EPISODE:
365 playback_info = await self.client.post(
366 f"content/{asin}/licenserequest",
367 body={
368 "consumption_type": "Streaming",
369 "drm_type": "Mpeg",
370 "quality": "High",
371 },
372 )
373 else:
374 playback_info = await self.client.post(
375 f"content/{asin}/licenserequest",
376 body={
377 "quality": "High",
378 "response_groups": "content_reference,certificate",
379 "consumption_type": "Streaming",
380 "supported_media_features": {
381 "codecs": ["mp4a.40.2", "mp4a.40.42"],
382 "drm_types": [
383 "Hls",
384 ],
385 },
386 "spatial": False,
387 },
388 )
389
390 content_license = playback_info.get("content_license", {})
391 if not content_license:
392 self.logger.error(f"No content_license in playback_info for ASIN {asin}")
393 raise ValueError(f"Missing content_license for ASIN {asin}")
394
395 content_metadata = content_license.get("content_metadata", {})
396 content_reference = content_metadata.get("content_reference", {})
397 size = content_reference.get("content_size_in_bytes", 0)
398
399 stream_url = content_license.get("license_response")
400 if not stream_url:
401 self.logger.error(f"No license_response (stream URL) for ASIN {asin}")
402 raise ValueError(f"Missing stream URL for ASIN {asin}")
403
404 acr = content_license.get("acr", "")
405 if acr:
406 self._acr_cache[(asin, media_type)] = acr
407
408 content_type = (
409 ContentType.MP3 if media_type == MediaType.PODCAST_EPISODE else ContentType.AAC
410 )
411 except Exception as exc:
412 self.logger.error(f"Error getting stream details for ASIN {asin}: {exc}")
413 raise ValueError(f"Failed to get stream details: {exc}") from exc
414
415 return StreamDetails(
416 provider=self.provider_instance,
417 size=size,
418 item_id=f"{asin}",
419 audio_format=AudioFormat(content_type=content_type),
420 media_type=media_type,
421 stream_type=StreamType.HTTP,
422 path=stream_url,
423 can_seek=True,
424 allow_seek=True,
425 duration=duration,
426 data={"acr": acr},
427 )
428
429 async def _fetch_chapters(self, asin: str) -> list[dict[str, Any]]:
430 """Fetch chapter data for an audiobook."""
431 if not asin or asin == "error":
432 self.logger.warning(
433 "Invalid ASIN provided to _fetch_chapters, returning empty chapter list"
434 )
435 return []
436
437 chapters_data: list[Any] = await self.mass.cache.get(
438 key=asin, provider=self.provider_instance, category=CACHE_CATEGORY_CHAPTERS, default=[]
439 )
440
441 if not chapters_data:
442 try:
443 response = await self._call_api(
444 f"content/{asin}/metadata",
445 response_groups="chapter_info, always-returned, content_reference, content_url",
446 chapter_titles_type="Flat",
447 )
448
449 if not response:
450 self.logger.warning(f"Failed to get metadata for ASIN {asin}")
451 return []
452
453 content_metadata = response.get("content_metadata")
454 if not content_metadata:
455 self.logger.warning(f"No content_metadata for ASIN {asin}")
456 return []
457
458 chapter_info = content_metadata.get("chapter_info")
459 if not chapter_info:
460 self.logger.warning(f"No chapter_info for ASIN {asin}")
461 return []
462
463 chapters_data = chapter_info.get("chapters") or []
464
465 await self.mass.cache.set(
466 key=asin,
467 data=chapters_data,
468 provider=self.provider_instance,
469 category=CACHE_CATEGORY_CHAPTERS,
470 )
471 except Exception as exc:
472 self.logger.error(f"Error fetching chapters for ASIN {asin}: {exc}")
473 chapters_data = []
474
475 return chapters_data
476
477 @staticmethod
478 def _parse_audible_timestamp(raw_ts: Any) -> datetime | None:
479 """
480 Parse an Audible timestamp value into a timezone-aware datetime.
481
482 :param raw_ts: The raw timestamp value from the Audible annotation payload.
483 """
484 if not raw_ts:
485 return None
486 try:
487 parsed = datetime.fromisoformat(str(raw_ts))
488 except ValueError, TypeError:
489 return None
490 if parsed.tzinfo is None:
491 parsed = parsed.replace(tzinfo=UTC)
492 return parsed
493
494 async def _fetch_last_position(self, asin: str) -> tuple[int, datetime | None] | None:
495 """
496 Fetch the last-heard position for a single ASIN from Audible.
497
498 :param asin: The audiobook ASIN to query.
499 """
500 response = await self._call_api("annotations/lastpositions", asins=asin)
501 if not response:
502 return None
503
504 annotations = response.get("asin_last_position_heard_annots")
505 if not annotations or not isinstance(annotations, list):
506 return None
507
508 annotation = annotations[0]
509 if not isinstance(annotation, dict):
510 return None
511
512 last_position = annotation.get("last_position_heard")
513 if not isinstance(last_position, dict):
514 return None
515
516 position_ms = int(last_position.get("position_ms", 0))
517
518 timestamp: datetime | None = None
519 for field in ("last_updated", "reported_time", "last_updated_time", "timestamp"):
520 timestamp = self._parse_audible_timestamp(
521 last_position.get(field) or annotation.get(field)
522 )
523 if timestamp is not None:
524 break
525
526 return position_ms, timestamp
527
528 async def get_last_position(self, asin: str) -> int:
529 """
530 Fetch the last-heard position in milliseconds for the given ASIN.
531
532 :param asin: The audiobook ASIN to query.
533 """
534 if not asin or asin == "error":
535 return 0
536 try:
537 result = await self._fetch_last_position(asin)
538 except (ProviderUnavailableError, KeyError, TypeError, ValueError) as exc:
539 self.logger.error("Error getting last position for ASIN %s: %s", asin, exc)
540 return 0
541 return result[0] if result else 0
542
543 async def set_last_position(
544 self, asin: str, pos: int, media_type: MediaType = MediaType.AUDIOBOOK
545 ) -> None:
546 """
547 Report last position to Audible.
548
549 :param asin: The content ID (audiobook or podcast episode).
550 :param pos: Position in seconds.
551 :param media_type: The type of media (audiobook or podcast episode).
552 """
553 if not asin or asin == "error" or pos <= 0:
554 return
555
556 try:
557 position_ms = pos * 1000
558
559 # Try to get ACR from cache first
560 acr = self._acr_cache.get((asin, media_type))
561 if not acr:
562 stream_details = await self.get_stream(asin=asin, media_type=media_type)
563 acr = stream_details.data.get("acr")
564
565 if not acr:
566 self.logger.warning(f"No ACR available for ASIN {asin}, cannot report position")
567 return
568
569 await self.client.put(
570 f"lastpositions/{asin}", body={"acr": acr, "asin": asin, "position_ms": position_ms}
571 )
572
573 self.logger.debug(f"Successfully reported position {position_ms}ms for ASIN {asin}")
574
575 except (KeyError, TypeError) as exc:
576 self.logger.error(
577 f"Error accessing data while reporting position for ASIN {asin}: {exc}"
578 )
579 except TimeoutError as exc:
580 self.logger.error(f"Timeout while reporting position for ASIN {asin}: {exc}")
581 except ConnectionError as exc:
582 self.logger.error(f"Connection error while reporting position for ASIN {asin}: {exc}")
583 except Exception as exc:
584 self.logger.error(f"Unexpected error reporting position for ASIN {asin}: {exc}")
585
586 async def get_audible_resume_position(self, asin: str) -> tuple[bool, int, datetime | None]:
587 """
588 Return resume state for the given ASIN from Audible.
589
590 :param asin: The audiobook ASIN to query.
591 """
592 if not asin or asin == "error":
593 raise NotImplementedError
594 try:
595 result = await self._fetch_last_position(asin)
596 except (ProviderUnavailableError, KeyError, TypeError, ValueError) as exc:
597 self.logger.debug("Audible lastpositions fetch failed for %s: %s", asin, exc)
598 raise NotImplementedError from exc
599 if not result or result[0] == 0:
600 raise NotImplementedError
601 position_ms, timestamp = result
602 return False, position_ms, timestamp
603
604 async def _call_api(self, path: str, **kwargs: Any) -> Any:
605 response = None
606 use_cache = kwargs.pop("use_cache", False)
607 params_str = json.dumps(kwargs, sort_keys=True)
608 params_hash = hashlib.md5(params_str.encode()).hexdigest()
609 cache_key_with_params = f"{path}:{params_hash}"
610 if use_cache:
611 response = await self.mass.cache.get(
612 key=cache_key_with_params,
613 provider=self.provider_instance,
614 category=CACHE_CATEGORY_API,
615 )
616 if not response:
617 try:
618 response = await self.client.get(path, **kwargs)
619 except audible.exceptions.RequestError as exc:
620 raise ProviderUnavailableError(
621 f"Audible API request failed for '{path}': {exc}"
622 ) from exc
623 await self.mass.cache.set(
624 key=cache_key_with_params, provider=self.provider_instance, data=response
625 )
626 return response
627
628 def _parse_contributors(
629 self, contributors_list: list[dict[str, Any]] | None, default_name: str
630 ) -> list[str]:
631 """Parse contributors (authors, narrators) from API response."""
632 result: list[str] = []
633 contributors: list[dict[str, Any]] = contributors_list or []
634 if isinstance(contributors, list):
635 for contributor in contributors:
636 if contributor and isinstance(contributor, dict):
637 result.append(contributor.get("name", default_name))
638 return result
639
640 def _create_images(self, image_path: str | None) -> list[MediaItemImage]:
641 """Create image objects if image path exists."""
642 images: list[MediaItemImage] = []
643 if image_path:
644 images.append(
645 MediaItemImage(
646 type=ImageType.THUMB,
647 path=image_path,
648 provider=self.provider_instance,
649 remotely_accessible=True,
650 )
651 )
652 images.append(
653 MediaItemImage(
654 type=ImageType.CLEARART,
655 path=image_path,
656 provider=self.provider_instance,
657 remotely_accessible=True,
658 )
659 )
660 return images
661
662 def _parse_chapter_data(self, chapter_data: dict[str, Any], index: int) -> MediaItemChapter:
663 """Parse chapter data into MediaItemChapter object."""
664 try:
665 start = int(chapter_data.get("start_offset_sec", 0))
666 except TypeError, ValueError:
667 start = 0
668
669 try:
670 length = int(chapter_data.get("length_ms", 0)) / 1000
671 except TypeError, ValueError:
672 length = 0
673
674 raw_title = chapter_data.get("title")
675 chapter_title: str
676 if raw_title is None:
677 chapter_title = f"Chapter {index + 1}"
678 elif isinstance(raw_title, str):
679 chapter_title = raw_title
680 else:
681 chapter_title = str(raw_title)
682
683 return MediaItemChapter(position=index, name=chapter_title, start=start, end=start + length)
684
685 def _parse_audiobook(self, audiobook_data: dict[str, Any] | None) -> Audiobook:
686 """
687 Parse audiobook data from API response.
688
689 NOTE: This is a pure parser - no API calls allowed here.
690 Chapters and resume position are fetched lazily when needed.
691 """
692 if audiobook_data is None:
693 self.logger.error("Received None audiobook_data in _parse_audiobook")
694 raise MediaNotFoundError("Audiobook data not found")
695
696 asin = audiobook_data.get("asin", "")
697 title = audiobook_data.get("title", "")
698
699 # Parse authors and narrators
700 narrators = self._parse_contributors(audiobook_data.get("narrators"), "Unknown Narrator")
701 authors = self._parse_contributors(audiobook_data.get("authors"), "Unknown Author")
702
703 # Get duration from runtime_length_min (provided by 'media' response group)
704 # Chapters are fetched lazily when streaming, not during library sync
705 runtime_minutes = audiobook_data.get("runtime_length_min", 0)
706 duration = runtime_minutes * 60 if runtime_minutes else 0
707
708 # Create audiobook object
709 book = Audiobook(
710 item_id=asin,
711 provider=self.provider_instance,
712 name=title,
713 duration=duration,
714 provider_mappings={
715 ProviderMapping(
716 item_id=asin,
717 provider_domain=self.provider_domain,
718 provider_instance=self.provider_instance,
719 )
720 },
721 publisher=audiobook_data.get("publisher_name"),
722 authors=UniqueList(authors),
723 narrators=UniqueList(narrators),
724 )
725
726 # Set metadata
727 book.metadata.copyright = audiobook_data.get("copyright")
728 book.metadata.description = _html_to_txt(
729 str(audiobook_data.get("extended_product_description", ""))
730 )
731 book.metadata.languages = UniqueList([audiobook_data.get("language") or ""])
732 if release_date := audiobook_data.get("release_date"):
733 with suppress(ValueError):
734 parsed_date = datetime.strptime(release_date, "%Y-%m-%d").astimezone(UTC)
735 book.metadata.release_date = parsed_date
736
737 # Set review if available
738 reviews = audiobook_data.get("editorial_reviews", [])
739 if reviews and reviews[0]:
740 book.metadata.review = _html_to_txt(str(reviews[0]))
741
742 # Set genres
743 book.metadata.genres = {
744 genre.replace("_", " ") for genre in (audiobook_data.get("platinum_keywords") or [])
745 }
746
747 # Add images
748 image_path = audiobook_data.get("product_images", {}).get("500")
749 book.metadata.images = UniqueList(self._create_images(image_path))
750
751 # Chapters are not fetched during parsing - they are fetched lazily when streaming
752 # This avoids N+1 API calls during library sync
753
754 return book
755
756 async def _process_podcast_item(self, podcast_data: dict[str, Any]) -> Podcast | None:
757 """Process a single podcast item from the library."""
758 asin = str(podcast_data.get("asin", ""))
759 cached_podcast = None
760 if asin:
761 cached_podcast = await self.mass.cache.get(
762 key=asin,
763 provider=self.provider_instance,
764 category=CACHE_CATEGORY_PODCAST,
765 default=None,
766 )
767
768 try:
769 if cached_podcast is not None:
770 return self._parse_podcast(cached_podcast)
771 return self._parse_podcast(podcast_data)
772 except Exception as exc:
773 self.provider.report_skipped_sync_item(MediaType.PODCAST, asin or None, exc)
774 return None
775
776 async def get_library_podcasts(self) -> AsyncGenerator[Podcast]:
777 """Fetch podcasts from the user's library with pagination."""
778 response_groups = [
779 "contributors",
780 "media",
781 "product_attrs",
782 "product_desc",
783 "product_details",
784 "product_extended_attrs",
785 ]
786
787 async for item in self._fetch_library_items(
788 ",".join(response_groups), PODCAST_CONTENT_TYPES
789 ):
790 if podcast := await self._process_podcast_item(item):
791 yield podcast
792
793 async def get_podcast(self, asin: str, use_cache: bool = True) -> Podcast:
794 """
795 Fetch full podcast details by ASIN.
796
797 :param asin: The ASIN of the podcast.
798 :param use_cache: Whether to use cached data if available.
799 """
800 if use_cache:
801 cached_podcast = await self.mass.cache.get(
802 key=asin,
803 provider=self.provider_instance,
804 category=CACHE_CATEGORY_PODCAST,
805 default=None,
806 )
807 if cached_podcast is not None:
808 return self._parse_podcast(cached_podcast)
809
810 response = await self._call_api(
811 f"library/{asin}",
812 response_groups="""
813 contributors, media, price, product_attrs, product_desc, product_details,
814 product_extended_attrs, relationships
815 """,
816 )
817
818 if response is None:
819 raise MediaNotFoundError(f"Podcast with ASIN {asin} not found")
820
821 item_data = response.get("item")
822 if item_data is None:
823 raise MediaNotFoundError(f"Podcast data for ASIN {asin} is empty")
824
825 await self.mass.cache.set(
826 key=asin,
827 provider=self.provider_instance,
828 category=CACHE_CATEGORY_PODCAST,
829 data=item_data,
830 )
831 return self._parse_podcast(item_data)
832
833 async def get_podcast_episodes(self, podcast_asin: str) -> AsyncGenerator[PodcastEpisode]:
834 """
835 Fetch all episodes for a podcast.
836
837 :param podcast_asin: The ASIN of the parent podcast.
838 """
839 podcast = await self.get_podcast(podcast_asin)
840
841 # Fetch episodes - they're typically in relationships or we need to query children
842 response_groups = [
843 "contributors",
844 "media",
845 "product_attrs",
846 "product_desc",
847 "product_details",
848 "relationships",
849 ]
850
851 page = 1
852 page_size = 50
853 position = 0
854
855 while True:
856 # Query for children of the podcast parent
857 response = await self._call_api(
858 "library",
859 use_cache=False,
860 response_groups=",".join(response_groups),
861 parent_asin=podcast_asin,
862 page=page,
863 num_results=page_size,
864 )
865
866 items = response.get("items", [])
867 if not items:
868 break
869
870 for episode_data in items:
871 try:
872 episode = self._parse_podcast_episode(episode_data, podcast, position)
873 position += 1
874 yield episode
875 except Exception as exc:
876 asin = episode_data.get("asin", "unknown")
877 self.logger.warning(f"Error parsing podcast episode {asin}: {exc}")
878
879 page += 1
880 if len(items) < page_size:
881 break
882
883 async def get_podcast_episode(self, episode_asin: str) -> PodcastEpisode:
884 """
885 Fetch full podcast episode details by ASIN.
886
887 :param episode_asin: The ASIN of the podcast episode.
888 """
889 response = await self._call_api(
890 f"library/{episode_asin}",
891 response_groups="""
892 contributors, media, price, product_attrs, product_desc, product_details,
893 product_extended_attrs, relationships
894 """,
895 )
896
897 if response is None:
898 raise MediaNotFoundError(f"Podcast episode with ASIN {episode_asin} not found")
899
900 item_data = response.get("item")
901 if item_data is None:
902 raise MediaNotFoundError(f"Podcast episode data for ASIN {episode_asin} is empty")
903
904 # Try to get parent podcast info from relationships
905 podcast: Podcast | None = None
906 relationships = item_data.get("relationships", [])
907 for rel in relationships:
908 if rel.get("relationship_type") == "parent":
909 parent_asin = rel.get("asin")
910 if parent_asin:
911 with suppress(MediaNotFoundError):
912 podcast = await self.get_podcast(parent_asin)
913 break
914
915 return self._parse_podcast_episode(item_data, podcast, 0)
916
917 def _parse_podcast(self, podcast_data: dict[str, Any] | None) -> Podcast:
918 """
919 Parse podcast data from API response.
920
921 :param podcast_data: Raw podcast data from the Audible API.
922 """
923 if podcast_data is None:
924 self.logger.error("Received None podcast_data in _parse_podcast")
925 raise MediaNotFoundError("Podcast data not found")
926
927 asin = podcast_data.get("asin", "")
928 title = podcast_data.get("title", "")
929 publisher = podcast_data.get("publisher_name", "")
930
931 # Create podcast object
932 podcast = Podcast(
933 item_id=asin,
934 provider=self.provider_instance,
935 name=title,
936 publisher=publisher,
937 provider_mappings={
938 ProviderMapping(
939 item_id=asin,
940 provider_domain=self.provider_domain,
941 provider_instance=self.provider_instance,
942 )
943 },
944 )
945
946 # Set metadata
947 podcast.metadata.description = _html_to_txt(
948 str(
949 podcast_data.get("publisher_summary", "")
950 or podcast_data.get("extended_product_description", "")
951 )
952 )
953 podcast.metadata.languages = UniqueList([podcast_data.get("language") or ""])
954
955 # Set genres
956 podcast.metadata.genres = {
957 genre.replace("_", " ") for genre in (podcast_data.get("platinum_keywords") or [])
958 }
959
960 # Add images
961 image_path = podcast_data.get("product_images", {}).get("500")
962 podcast.metadata.images = UniqueList(self._create_images(image_path))
963
964 return podcast
965
966 def _parse_podcast_episode(
967 self,
968 episode_data: dict[str, Any] | None,
969 podcast: Podcast | None,
970 position: int,
971 ) -> PodcastEpisode:
972 """
973 Parse podcast episode data from API response.
974
975 :param episode_data: Raw episode data from the Audible API.
976 :param podcast: Parent podcast object (optional).
977 :param position: Position/index of the episode in the podcast.
978 """
979 if episode_data is None:
980 self.logger.error("Received None episode_data in _parse_podcast_episode")
981 raise MediaNotFoundError("Podcast episode data not found")
982
983 asin = episode_data.get("asin", "")
984 title = episode_data.get("title", "")
985
986 # Get duration from runtime_length_min
987 runtime_minutes = episode_data.get("runtime_length_min", 0)
988 duration = runtime_minutes * 60 if runtime_minutes else 0
989
990 # Create podcast reference - use Podcast object or create ItemMapping
991 podcast_ref: Podcast | ItemMapping
992 if podcast is not None:
993 podcast_ref = podcast
994 else:
995 # Try to get parent_asin from relationships for ItemMapping
996 parent_asin = ""
997 relationships = episode_data.get("relationships", [])
998 for rel in relationships:
999 if rel.get("relationship_type") == "parent":
1000 parent_asin = rel.get("asin", "")
1001 break
1002
1003 if not parent_asin:
1004 self.logger.warning(
1005 "No parent_asin found for podcast episode %s; parent podcast is unknown",
1006 asin,
1007 )
1008
1009 podcast_ref = ItemMapping(
1010 item_id=parent_asin or "",
1011 provider=self.provider_instance,
1012 name="Unknown Podcast",
1013 media_type=MediaType.PODCAST,
1014 )
1015
1016 # Create episode object
1017 episode = PodcastEpisode(
1018 item_id=asin,
1019 provider=self.provider_instance,
1020 name=title,
1021 duration=duration,
1022 position=position,
1023 podcast=podcast_ref,
1024 provider_mappings={
1025 ProviderMapping(
1026 item_id=asin,
1027 provider_domain=self.provider_domain,
1028 provider_instance=self.provider_instance,
1029 )
1030 },
1031 )
1032
1033 # Set metadata
1034 episode.metadata.description = _html_to_txt(
1035 str(
1036 episode_data.get("publisher_summary", "")
1037 or episode_data.get("extended_product_description", "")
1038 )
1039 )
1040
1041 # Add images
1042 image_path = episode_data.get("product_images", {}).get("500")
1043 episode.metadata.images = UniqueList(self._create_images(image_path))
1044
1045 return episode
1046
1047 async def get_authors(self) -> dict[str, str]:
1048 """
1049 Get all unique authors from the library.
1050
1051 Returns dict mapping author ASIN to author name.
1052 """
1053 authors: dict[str, str] = {}
1054 async for item in self._fetch_library_items(
1055 "contributors,product_attrs", AUDIOBOOK_CONTENT_TYPES
1056 ):
1057 for author in item.get("authors") or []:
1058 asin = author.get("asin")
1059 name = author.get("name")
1060 if asin and name:
1061 authors[asin] = name
1062 return authors
1063
1064 async def get_series(self) -> dict[str, str]:
1065 """
1066 Get all unique series from the library.
1067
1068 Returns dict mapping series ASIN to series title.
1069 """
1070 series: dict[str, str] = {}
1071 async for item in self._fetch_library_items(
1072 "series,product_attrs", AUDIOBOOK_CONTENT_TYPES
1073 ):
1074 for s in item.get("series") or []:
1075 asin = s.get("asin")
1076 title = s.get("title")
1077 if asin and title:
1078 series[asin] = title
1079 return series
1080
1081 async def get_narrators(self) -> dict[str, str]:
1082 """
1083 Get all unique narrators from the library.
1084
1085 Returns dict mapping narrator ASIN to narrator name.
1086 """
1087 narrators: dict[str, str] = {}
1088 async for item in self._fetch_library_items(
1089 "contributors,product_attrs", AUDIOBOOK_CONTENT_TYPES
1090 ):
1091 for narrator in item.get("narrators") or []:
1092 asin = narrator.get("asin")
1093 name = narrator.get("name")
1094 if asin and name:
1095 narrators[asin] = name
1096 return narrators
1097
1098 async def get_genres(self) -> set[str]:
1099 """Get all unique genres from the library."""
1100 genres: set[str] = set()
1101 async for item in self._fetch_library_items("product_attrs", AUDIOBOOK_CONTENT_TYPES):
1102 for keyword in item.get("thesaurus_subject_keywords") or []:
1103 genres.add(keyword.replace("_", " ").replace("-", " ").title())
1104 return genres
1105
1106 async def get_publishers(self) -> set[str]:
1107 """Get all unique publishers from the library."""
1108 publishers: set[str] = set()
1109 async for item in self._fetch_library_items("product_attrs", AUDIOBOOK_CONTENT_TYPES):
1110 publisher = item.get("publisher_name")
1111 if publisher:
1112 publishers.add(publisher)
1113 return publishers
1114
1115 async def get_audiobooks_by_author(self, author_asin: str) -> list[Audiobook]:
1116 """Get all audiobooks by a specific author, sorted by release date."""
1117 audiobooks: list[tuple[str, Audiobook]] = []
1118 async for item in self._fetch_library_items(
1119 "contributors,media,product_attrs,product_desc,series", AUDIOBOOK_CONTENT_TYPES
1120 ):
1121 for author in item.get("authors") or []:
1122 if author.get("asin") == author_asin:
1123 release_date = item.get("release_date") or "0000-00-00"
1124 audiobooks.append((release_date, self._parse_audiobook(item)))
1125 break
1126 audiobooks.sort(key=lambda x: x[0], reverse=True)
1127 return [book for _, book in audiobooks]
1128
1129 async def get_audiobooks_by_narrator(self, narrator_asin: str) -> list[Audiobook]:
1130 """Get all audiobooks by a specific narrator, sorted by release date."""
1131 audiobooks: list[tuple[str, Audiobook]] = []
1132 async for item in self._fetch_library_items(
1133 "contributors,media,product_attrs,product_desc,series", AUDIOBOOK_CONTENT_TYPES
1134 ):
1135 for narrator in item.get("narrators") or []:
1136 if narrator.get("asin") == narrator_asin:
1137 release_date = item.get("release_date") or "0000-00-00"
1138 audiobooks.append((release_date, self._parse_audiobook(item)))
1139 break
1140 audiobooks.sort(key=lambda x: x[0], reverse=True)
1141 return [book for _, book in audiobooks]
1142
1143 async def get_audiobooks_by_genre(self, genre: str) -> list[Audiobook]:
1144 """Get all audiobooks matching a genre, sorted by release date."""
1145 audiobooks: list[tuple[str, Audiobook]] = []
1146 genre_key = genre.lower().replace(" ", "_")
1147 genre_key_alt = genre.lower().replace(" ", "-")
1148 async for item in self._fetch_library_items(
1149 "contributors,media,product_attrs,product_desc,series", AUDIOBOOK_CONTENT_TYPES
1150 ):
1151 keywords = item.get("thesaurus_subject_keywords") or []
1152 if genre_key in keywords or genre_key_alt in keywords:
1153 release_date = item.get("release_date") or "0000-00-00"
1154 audiobooks.append((release_date, self._parse_audiobook(item)))
1155 audiobooks.sort(key=lambda x: x[0], reverse=True)
1156 return [book for _, book in audiobooks]
1157
1158 async def get_audiobooks_by_publisher(self, publisher: str) -> list[Audiobook]:
1159 """Get all audiobooks from a specific publisher, sorted by release date."""
1160 audiobooks: list[tuple[str, Audiobook]] = []
1161 async for item in self._fetch_library_items(
1162 "contributors,media,product_attrs,product_desc,series", AUDIOBOOK_CONTENT_TYPES
1163 ):
1164 if item.get("publisher_name") == publisher:
1165 release_date = item.get("release_date") or "0000-00-00"
1166 audiobooks.append((release_date, self._parse_audiobook(item)))
1167 audiobooks.sort(key=lambda x: x[0], reverse=True)
1168 return [book for _, book in audiobooks]
1169
1170 async def get_audiobooks_by_series(self, series_asin: str) -> list[Audiobook]:
1171 """Get all audiobooks in a specific series, ordered by sequence."""
1172 audiobooks: list[tuple[float, Audiobook]] = []
1173 async for item in self._fetch_library_items(
1174 "contributors,media,product_attrs,product_desc,series", AUDIOBOOK_CONTENT_TYPES
1175 ):
1176 for s in item.get("series") or []:
1177 if s.get("asin") == series_asin:
1178 sequence = s.get("sequence")
1179 try:
1180 seq_num = float(sequence) if sequence else 999
1181 except ValueError, TypeError:
1182 seq_num = 999
1183 audiobooks.append((seq_num, self._parse_audiobook(item)))
1184 break
1185 audiobooks.sort(key=lambda x: x[0])
1186 return [book for _, book in audiobooks]
1187
1188 async def deregister(self) -> None:
1189 """Deregister this provider from Audible."""
1190 await asyncio.to_thread(self.client.auth.deregister_device)
1191
1192
1193def _html_to_txt(html_text: str) -> str:
1194 txt = html.unescape(html_text)
1195 tags = re.findall("<[^>]+>", txt)
1196 for tag in tags:
1197 txt = txt.replace(tag, "")
1198 return txt
1199
1200
1201async def audible_get_auth_info(locale: str) -> tuple[str, str, str]:
1202 """
1203 Generate the login URL and auth info for Audible OAuth flow.
1204
1205 :param locale: The locale string (e.g., 'us', 'uk', 'de').
1206 :return: Tuple of (code_verifier, oauth_url, serial).
1207 """
1208 locale_obj = audible.localization.Locale(locale)
1209 code_verifier = await asyncio.to_thread(audible.login.create_code_verifier)
1210 oauth_url, serial = await asyncio.to_thread(
1211 audible.login.build_oauth_url,
1212 country_code=locale_obj.country_code,
1213 domain=locale_obj.domain,
1214 market_place_id=locale_obj.market_place_id,
1215 code_verifier=code_verifier,
1216 with_username=False,
1217 )
1218
1219 return code_verifier.decode(), oauth_url, serial
1220
1221
1222async def audible_custom_login(
1223 code_verifier: str, response_url: str, serial: str, locale: str
1224) -> audible.Authenticator:
1225 """
1226 Complete the authentication using the code_verifier, response_url, and serial.
1227
1228 :param code_verifier: The code verifier string used in OAuth flow.
1229 :param response_url: The response URL containing the authorization code.
1230 :param serial: The device serial number.
1231 :param locale: The locale string.
1232 :return: Audible Authenticator object.
1233 :raises LoginFailed: If authorization code is not found in the URL.
1234 """
1235 logger = logging.getLogger("audible_helper")
1236 auth = audible.Authenticator()
1237 auth.locale = audible.localization.Locale(locale)
1238
1239 response_url_parsed = urlparse(response_url)
1240 parsed_qs = parse_qs(response_url_parsed.query)
1241
1242 # Try multiple parameter names for authorization code
1243 # Audible may use different parameter names depending on the flow
1244 authorization_code = None
1245 for param_name in ["openid.oa2.authorization_code", "authorization_code", "code"]:
1246 if codes := parsed_qs.get(param_name):
1247 authorization_code = codes[0]
1248 logger.debug("Found authorization code in parameter: %s", param_name)
1249 break
1250
1251 if not authorization_code:
1252 available_params = list(parsed_qs.keys())
1253 raise LoginFailed(
1254 f"Authorization code not found in URL. "
1255 f"Expected 'openid.oa2.authorization_code' but found parameters: {available_params}"
1256 )
1257
1258 registration_data = await asyncio.to_thread(
1259 audible.register.register,
1260 authorization_code=authorization_code,
1261 code_verifier=code_verifier.encode(),
1262 domain=auth.locale.domain,
1263 serial=serial,
1264 )
1265 auth._update_attrs(**registration_data)
1266
1267 # Log what auth methods are available after registration
1268 if auth.adp_token and auth.device_private_key:
1269 logger.info("Registration successful with signing auth (stable)")
1270 else:
1271 logger.warning("Registration successful but signing auth not available")
1272
1273 return auth
1274
1275
1276async def check_file_exists(path: str | PathLike[str]) -> bool:
1277 """Async file exists check."""
1278 return await asyncio.to_thread(os.path.exists, path)
1279
1280
1281async def remove_file(path: str | PathLike[str]) -> None:
1282 """Async file delete."""
1283 await asyncio.to_thread(os.remove, path)
1284