/
/
1"""Filesystem musicprovider support for MusicAssistant."""
2
3from __future__ import annotations
4
5import asyncio
6import contextlib
7import logging
8import os
9import os.path
10import posixpath
11import urllib.parse
12from collections.abc import AsyncGenerator, Sequence
13from datetime import UTC, datetime
14from pathlib import Path
15from typing import TYPE_CHECKING, Any, ClassVar, cast
16from xml.parsers.expat import ExpatError
17
18import aiofiles
19import shortuuid
20import xmltodict
21from aiofiles.os import wrap
22from music_assistant_models.enums import (
23 ContentType,
24 EventType,
25 ExternalID,
26 ImageType,
27 MediaType,
28 ProviderFeature,
29 StreamType,
30)
31from music_assistant_models.errors import (
32 InvalidDataError,
33 MediaNotFoundError,
34 MusicAssistantError,
35 SetupFailedError,
36)
37from music_assistant_models.helpers import create_safe_string
38from music_assistant_models.media_items import (
39 Album,
40 Artist,
41 Audiobook,
42 AudioFormat,
43 BrowseFolder,
44 ItemMapping,
45 MediaItemChapter,
46 MediaItemImage,
47 MediaItemType,
48 Playlist,
49 Podcast,
50 PodcastEpisode,
51 ProviderMapping,
52 SearchResults,
53 SoundEffect,
54 Track,
55 UniqueList,
56 is_track,
57)
58from music_assistant_models.streamdetails import MultiPartPath, StreamDetails
59
60from music_assistant.constants import (
61 CONF_PATH,
62 DB_TABLE_ALBUM_ARTISTS,
63 DB_TABLE_ALBUM_TRACKS,
64 DB_TABLE_ALBUMS,
65 DB_TABLE_ARTISTS,
66 DB_TABLE_PROVIDER_MAPPINGS,
67 DB_TABLE_TRACK_ARTISTS,
68 VARIOUS_ARTISTS_MBID,
69 VARIOUS_ARTISTS_NAME,
70 VERBOSE_LOG_LEVEL,
71)
72from music_assistant.controllers.tasks.context import (
73 report_current_task_failure,
74 update_current_task_progress_from_index,
75 update_current_task_progress_text,
76)
77from music_assistant.helpers import lyrics
78from music_assistant.helpers.compare import compare_strings
79from music_assistant.helpers.json import SerializableType, json_loads
80from music_assistant.helpers.playlists import parse_m3u, parse_pls
81from music_assistant.helpers.tags import AudioTags, async_parse_tags, clean_mbid, split_items
82from music_assistant.helpers.util import (
83 TaskManager,
84 detect_charset,
85 parse_title_and_version,
86 try_parse_int,
87)
88from music_assistant.models.music_provider import MusicProvider
89
90from .constants import (
91 AUDIOBOOK_EXTENSIONS,
92 AVAILABILITY_PROBE_INTERVAL,
93 CACHE_CATEGORY_ALBUM_INFO,
94 CACHE_CATEGORY_ARTIST_INFO,
95 CACHE_CATEGORY_AUDIOBOOK_CHAPTERS,
96 CACHE_CATEGORY_FOLDER_IMAGES,
97 CACHE_CATEGORY_PODCAST_EPISODES,
98 CACHE_CATEGORY_PODCAST_METADATA,
99 CACHE_CATEGORY_SOUND_EFFECTS,
100 CONF_CONTENT_TYPE,
101 CONF_ENTRY_CONTENT_TYPE,
102 CONF_ENTRY_IGNORE_ALBUM_PLAYLISTS,
103 CONF_ENTRY_LIBRARY_SYNC_AUDIOBOOKS,
104 CONF_ENTRY_LIBRARY_SYNC_PLAYLISTS,
105 CONF_ENTRY_LIBRARY_SYNC_PODCASTS,
106 CONF_ENTRY_LIBRARY_SYNC_TRACKS,
107 CONF_ENTRY_MISSING_ALBUM_ARTIST,
108 CONF_ENTRY_PROPAGATE_GENRES,
109 CUE_EXTENSIONS,
110 DEFAULT_AUDIOBOOK_PODCAST_GENRE,
111 IMAGE_EXTENSIONS,
112 PARTIAL_LISTING_CACHE_EXPIRATION,
113 PLAYLIST_EXTENSIONS,
114 PODCAST_EPISODE_EXTENSIONS,
115 SOUND_EFFECT_EXTENSIONS,
116 SUPPORTED_EXTENSIONS,
117 TRACK_EXTENSIONS,
118 IsChapterFile,
119 content_type_config_entry,
120)
121from .cue import (
122 CueSheetHandler,
123 cue_metadata_checksum,
124 cue_referenced_audio_stem,
125 make_cue_track_id,
126 parse_cue_track_id,
127)
128from .helpers import (
129 FileSystemItem,
130 ScanErrors,
131 get_absolute_path,
132 get_album_dir,
133 get_artist_dir,
134 get_folder_signature,
135 get_relative_path,
136 recursive_iter,
137 sorted_scandir,
138)
139from .parsers import parse_album_nfo
140
141if TYPE_CHECKING:
142 from music_assistant_models.config_entries import ConfigEntry, ProviderConfig
143 from music_assistant_models.provider import ProviderManifest
144
145 from music_assistant.mass import MusicAssistant
146 from music_assistant.models import ProviderInstanceType
147 from music_assistant.providers.musicbrainz import MusicbrainzProvider
148
149
150isdir = wrap(os.path.isdir)
151isfile = wrap(os.path.isfile)
152ismount = wrap(os.path.ismount)
153exists = wrap(os.path.exists)
154makedirs = wrap(os.makedirs)
155
156SUPPORTED_FEATURES = {
157 ProviderFeature.BROWSE,
158 ProviderFeature.SEARCH,
159}
160
161
162async def setup(
163 mass: MusicAssistant, manifest: ProviderManifest, config: ProviderConfig
164) -> ProviderInstanceType:
165 """Initialize provider(instance) with given configuration."""
166 return LocalFileSystemProvider(mass, manifest, config)
167
168
169class LocalFileSystemProvider(MusicProvider):
170 """
171 Implementation of a musicprovider for (local) files.
172
173 Reads ID3 tags from file and falls back to parsing filename.
174 Optionally reads metadata from nfo files and images in folder structure <artist>/<album>.
175 Supports m3u files for playlists.
176 """
177
178 # parallel workers per sync; subclasses lower this for slower transports
179 _SYNC_CONCURRENCY: ClassVar[int] = 16
180 _sync_tracks: bool = True
181 _sync_playlists: bool = True
182
183 def __init__(
184 self,
185 mass: MusicAssistant,
186 manifest: ProviderManifest,
187 config: ProviderConfig,
188 base_path: str | None = None,
189 ) -> None:
190 """Initialize MusicProvider."""
191 super().__init__(mass, manifest, config, SUPPORTED_FEATURES)
192 # subclasses (NFS/SMB/...) mount elsewhere and pass their own base_path;
193 # the plain local provider reads its scan directory from the setup data
194 self.base_path: str = (
195 base_path if base_path is not None else cast("str", self.get_setup_value(CONF_PATH))
196 )
197 self.write_access: bool = False
198 self.sync_running: bool = False
199 self.media_content_type = cast(
200 "str", self.get_setup_value(CONF_CONTENT_TYPE, CONF_ENTRY_CONTENT_TYPE.default_value)
201 )
202 self._cue = CueSheetHandler(self)
203
204 async def get_config_entries(self) -> tuple[ConfigEntry, ...]:
205 """Return Config entries to configure this provider."""
206 # content type and path are collected by the setup flow; surface the (immutable)
207 # content type read-only so the sync options' depends_on chains still resolve
208 content_type = str(
209 self.get_setup_value(CONF_CONTENT_TYPE, CONF_ENTRY_CONTENT_TYPE.default_value)
210 )
211 return (
212 content_type_config_entry(content_type),
213 CONF_ENTRY_MISSING_ALBUM_ARTIST,
214 CONF_ENTRY_IGNORE_ALBUM_PLAYLISTS,
215 CONF_ENTRY_LIBRARY_SYNC_TRACKS,
216 CONF_ENTRY_LIBRARY_SYNC_PLAYLISTS,
217 CONF_ENTRY_LIBRARY_SYNC_PODCASTS,
218 CONF_ENTRY_LIBRARY_SYNC_AUDIOBOOKS,
219 CONF_ENTRY_PROPAGATE_GENRES,
220 )
221
222 @property
223 def supported_features(self) -> set[ProviderFeature]:
224 """Return the features supported by this Provider."""
225 base_features = {*SUPPORTED_FEATURES}
226 if self.media_content_type == "audiobooks":
227 return {ProviderFeature.LIBRARY_AUDIOBOOKS, *base_features}
228 if self.media_content_type == "podcasts":
229 return {ProviderFeature.LIBRARY_PODCASTS, *base_features}
230 if self.media_content_type == "sound_effects":
231 # sound effects are live-fetched content, never synced into the library
232 return {ProviderFeature.SOUND_EFFECTS, *base_features}
233 music_features = {
234 ProviderFeature.LIBRARY_ALBUMS,
235 ProviderFeature.LIBRARY_ARTISTS,
236 ProviderFeature.LIBRARY_TRACKS,
237 ProviderFeature.LIBRARY_PLAYLISTS,
238 *base_features,
239 }
240 if self.write_access:
241 music_features.add(ProviderFeature.PLAYLIST_TRACKS_EDIT)
242 music_features.add(ProviderFeature.PLAYLIST_CREATE)
243 return music_features
244
245 @property
246 def is_streaming_provider(self) -> bool:
247 """Return True if the provider is a streaming provider."""
248 return False
249
250 @property
251 def instance_name_postfix(self) -> str | None:
252 """Return a (default) instance name postfix for this provider instance."""
253 return Path(self.base_path).name
254
255 async def handle_async_init(self) -> None:
256 """Handle async initialization of the provider."""
257 if not await isdir(self.base_path):
258 msg = f"Music Directory {self.base_path} does not exist"
259 raise SetupFailedError(
260 msg,
261 translation_key="music_directory_not_found",
262 translation_owner=self.translation_owner,
263 translation_args=[self.base_path],
264 )
265 await self.check_write_access()
266
267 async def unload(self, is_removed: bool = False) -> None:
268 """Handle unload/close of the provider."""
269 self._cancel_availability_probe()
270 # a check that already started runs as a task under the same id, and it would
271 # otherwise keep talking to storage this unload is in the middle of tearing down
272 self.mass.cancel_task(self._availability_probe_id)
273
274 async def get_diagnostics(self) -> dict[str, SerializableType]:
275 """Return diagnostics info for this provider to include in diagnostics reports."""
276 return {
277 "sync_running": self.sync_running,
278 "write_access": self.write_access,
279 "content_type": self.media_content_type,
280 }
281
282 async def search(
283 self,
284 search_query: str,
285 media_types: list[MediaType] | None,
286 limit: int = 5,
287 ) -> SearchResults:
288 """Perform search on this file based musicprovider."""
289 result = SearchResults()
290 # searching the filesystem is slow and unreliable,
291 # so instead we just query the db...
292 if media_types is None or MediaType.TRACK in media_types:
293 result.tracks = await self.mass.music.tracks.get_library_items_by_query(
294 search=search_query, provider_filter=[self.instance_id], limit=limit
295 )
296
297 if media_types is None or MediaType.ALBUM in media_types:
298 result.albums = await self.mass.music.albums.get_library_items_by_query(
299 search=search_query,
300 provider_filter=[self.instance_id],
301 limit=limit,
302 )
303
304 if media_types is None or MediaType.ARTIST in media_types:
305 result.artists = await self.mass.music.artists.get_library_items_by_query(
306 search=search_query,
307 provider_filter=[self.instance_id],
308 limit=limit,
309 )
310 if media_types is None or MediaType.PLAYLIST in media_types:
311 result.playlists = await self.mass.music.playlists.get_library_items_by_query(
312 search=search_query,
313 provider_filter=[self.instance_id],
314 limit=limit,
315 )
316 if media_types is None or MediaType.AUDIOBOOK in media_types:
317 result.audiobooks = await self.mass.music.audiobooks.get_library_items_by_query(
318 search=search_query,
319 provider_filter=[self.instance_id],
320 limit=limit,
321 )
322 if media_types is None or MediaType.PODCAST in media_types:
323 result.podcasts = await self.mass.music.podcasts.get_library_items_by_query(
324 search=search_query,
325 provider_filter=[self.instance_id],
326 limit=limit,
327 )
328 return result
329
330 async def browse(self, path: str) -> Sequence[MediaItemType | ItemMapping | BrowseFolder]:
331 """
332 Browse this provider's items.
333
334 :param path: The path to browse, (e.g. provid://artists).
335 """
336 # for audiobooks and podcasts we just return all library items
337 if self.media_content_type == "podcasts":
338 return await self.mass.music.podcasts.library_items(
339 provider=self.instance_id, summary=False
340 )
341 if self.media_content_type == "audiobooks":
342 return await self.mass.music.audiobooks.library_items(
343 provider=self.instance_id, summary=False
344 )
345 items: list[MediaItemType | ItemMapping | BrowseFolder] = []
346 item_path = path.split("://", 1)[1]
347 if not item_path:
348 item_path = ""
349 scanned = await self._scandir(item_path)
350 # expand CUE sheets into per-track entries and hide the companion audio;
351 # synthetic ids match those minted during sync so get_track resolves them
352 cue_stems: set[str] = set()
353 if self.media_content_type == "music":
354 for item in scanned:
355 if item.ext not in CUE_EXTENSIONS:
356 continue
357 cue_stems.add(item.absolute_path.rsplit(".", 1)[0])
358 try:
359 cue_sheet = await self._cue.load_cue_sheet(item)
360 except InvalidDataError as err:
361 self.logger.warning("Unable to parse CUE sheet %s: %s", item.relative_path, err)
362 continue
363 # also hide the audio file named in the CUE (may differ from its stem)
364 if companion_stem := cue_referenced_audio_stem(item, cue_sheet):
365 cue_stems.add(companion_stem)
366 for cue_track in cue_sheet.tracks:
367 items.append(
368 ItemMapping(
369 media_type=MediaType.TRACK,
370 item_id=make_cue_track_id(item.relative_path, cue_track.number),
371 provider=self.instance_id,
372 name=cue_track.title or f"Track {cue_track.number}",
373 )
374 )
375 for item in scanned:
376 if not item.is_dir and ("." not in item.filename or not item.ext):
377 # skip system files and files without extension
378 continue
379
380 if item.is_dir:
381 items.append(
382 BrowseFolder(
383 item_id=item.relative_path,
384 provider=self.instance_id,
385 path=f"{self.instance_id}://{item.relative_path}",
386 name=item.filename,
387 # mark folder as playable, assuming it contains tracks underneath
388 is_playable=True,
389 )
390 )
391 elif item.ext in TRACK_EXTENSIONS:
392 if item.absolute_path.rsplit(".", 1)[0] in cue_stems:
393 continue
394 items.append(
395 ItemMapping(
396 media_type=(
397 MediaType.SOUND_EFFECT
398 if self.media_content_type == "sound_effects"
399 else MediaType.TRACK
400 ),
401 item_id=item.relative_path,
402 provider=self.instance_id,
403 name=item.filename,
404 )
405 )
406 elif item.ext in PLAYLIST_EXTENSIONS and self.media_content_type == "music":
407 items.append(
408 ItemMapping(
409 media_type=MediaType.PLAYLIST,
410 item_id=item.relative_path,
411 provider=self.instance_id,
412 name=item.filename,
413 )
414 )
415 if self.media_content_type == "music":
416 track_indexes = [
417 index
418 for index, item in enumerate(items)
419 if isinstance(item, ItemMapping) and item.media_type == MediaType.TRACK
420 ]
421 library_tracks = await asyncio.gather(
422 *(
423 self.mass.music.tracks.get_library_item_by_prov_id(
424 items[index].item_id, self.instance_id
425 )
426 for index in track_indexes
427 )
428 )
429 for index, library_track in zip(track_indexes, library_tracks, strict=True):
430 if library_track:
431 items[index] = library_track
432 return items
433
434 async def sync_library(self, media_type: MediaType) -> None:
435 """Run library sync for this provider."""
436 if media_type in (MediaType.ARTIST, MediaType.ALBUM):
437 # artists and albums are synced as part of track sync
438 return
439 if self.media_content_type == "sound_effects":
440 # sound effects are live-fetched content, never synced into the library
441 return
442 # check if any sync options are enabled for this content type
443 # the filesystem provider processes all file types in one scan,
444 # so we can return early if nothing needs syncing
445 if self.media_content_type == "music":
446 self._sync_tracks = bool(self.config.get_value(CONF_ENTRY_LIBRARY_SYNC_TRACKS.key))
447 self._sync_playlists = bool(
448 self.config.get_value(CONF_ENTRY_LIBRARY_SYNC_PLAYLISTS.key)
449 )
450 if not self._sync_tracks and not self._sync_playlists:
451 return
452 elif self.media_content_type == "audiobooks":
453 if not self.config.get_value(CONF_ENTRY_LIBRARY_SYNC_AUDIOBOOKS.key):
454 return
455 elif self.media_content_type == "podcasts":
456 if not self.config.get_value(CONF_ENTRY_LIBRARY_SYNC_PODCASTS.key):
457 return
458 assert self.mass.music.database
459 if self.sync_running:
460 self.logger.warning("Library sync already running for %s", self.name)
461 return
462 file_checksums: dict[str, str] = {}
463 # NOTE: we always run a scan of the entire library, as we need to detect changes
464 # we ignore any given mediatype(s) and just scan all supported files
465 query = (
466 f"SELECT provider_item_id, details FROM {DB_TABLE_PROVIDER_MAPPINGS} "
467 f"WHERE provider_instance = '{self.instance_id}' "
468 f"AND media_type in ('track', 'playlist', 'audiobook', 'podcast_episode')"
469 )
470 for db_row in await self.mass.music.database.get_rows_from_query(query, limit=0):
471 file_checksums[db_row["provider_item_id"]] = str(db_row["details"])
472 # provider_mappings stores synthetic per-track ids for CUE sheets, not the
473 # CUE path, so collect every track checksum per path for the scan classifier
474 cue_file_checksums: dict[str, set[str]] = {}
475 for prov_item_id, checksum in file_checksums.items():
476 parsed = parse_cue_track_id(prov_item_id)
477 if parsed is not None:
478 cue_file_checksums.setdefault(parsed[0], set()).add(checksum)
479 # find all supported files in the base directory and all subfolders
480 # we work bottom up, as-in we derive all info from the tracks
481 cur_filenames: set[str] = set()
482 prev_filenames = set(file_checksums.keys())
483
484 items_to_process: list[tuple[FileSystemItem, str | None]] = []
485 unchanged_cue_items: list[FileSystemItem] = []
486 # absolute paths of every CUE sheet in this scan with the ".cue" stripped,
487 # used for O(1) companion-CUE lookups per audio file
488 cue_stems: set[str] = set()
489 # collects the errors raised while walking the tree; any error means the
490 # scan is incomplete, a fatal one means the provider is unreachable
491 scan_errors = ScanErrors()
492
493 self.sync_running = True
494 try:
495 await self._enumerate_files_for_sync(
496 file_checksums=file_checksums,
497 cue_file_checksums=cue_file_checksums,
498 cur_filenames=cur_filenames,
499 items_to_process=items_to_process,
500 unchanged_cue_items=unchanged_cue_items,
501 cue_stems=cue_stems,
502 scan_errors=scan_errors,
503 )
504 if scan_errors.fatal:
505 # the storage is gone, so reading the files collected before it went
506 # away would only add a timeout each
507 self.logger.error("Aborting sync for %s: %s", self.name, scan_errors.fatal)
508 report_current_task_failure("Sync aborted: filesystem unavailable during scan")
509 self._set_available(False)
510 return
511 # a CUE may name an audio file other than its own; hide that companion too
512 if self.media_content_type == "music":
513 for cue_item in (
514 *unchanged_cue_items,
515 *(item for item, _ in items_to_process if item.ext in CUE_EXTENSIONS),
516 ):
517 try:
518 cue_sheet = await self._cue.load_cue_sheet(cue_item)
519 except InvalidDataError:
520 continue
521 if companion_stem := cue_referenced_audio_stem(cue_item, cue_sheet):
522 cue_stems.add(companion_stem)
523 # drop CUE companion audio: absorbed into CUE tracks and not tracked in
524 # provider_mappings, so they would otherwise flag as changed every sync
525 items_to_process = [
526 (item, prev)
527 for item, prev in items_to_process
528 if not (
529 item.ext in TRACK_EXTENSIONS
530 and item.absolute_path.rsplit(".", 1)[0] in cue_stems
531 )
532 ]
533 # register synthetic track IDs for unchanged CUE files so the
534 # deletion pass does not treat them as removed
535 for cue_item in unchanged_cue_items:
536 try:
537 cue_sheet = await self._cue.load_cue_sheet(cue_item)
538 except InvalidDataError as err:
539 self.logger.warning(
540 "Unable to parse CUE sheet %s: %s", cue_item.relative_path, err
541 )
542 continue
543 for cue_track in cue_sheet.tracks:
544 cur_filenames.add(make_cue_track_id(cue_item.relative_path, cue_track.number))
545 total_items = len(items_to_process)
546 self.logger.info(
547 "Found %d changed/new items to process for %s",
548 total_items,
549 self.name,
550 )
551
552 # _SYNC_CONCURRENCY caps parallelism per provider (NFS/SMB/WebDAV friendly)
553 processed_count = 0
554
555 async def _process(item: FileSystemItem, prev_checksum: str | None) -> None:
556 nonlocal processed_count
557 if await self._process_item_async(
558 item, prev_checksum, cur_filenames, cue_stems, prev_filenames
559 ):
560 cur_filenames.add(item.relative_path)
561 processed_count += 1
562 if processed_count % 50 == 0 or processed_count == total_items:
563 update_current_task_progress_from_index(
564 processed_count,
565 total_items,
566 f"Processed {processed_count}/{total_items} files",
567 )
568
569 async with TaskManager(self.mass, self._SYNC_CONCURRENCY) as tm:
570 for item, prev_checksum in items_to_process:
571 await tm.create_task_with_limit(_process(item, prev_checksum))
572 finally:
573 self.sync_running = False
574
575 # do not run deletions on a clean but empty scan of a previously non-empty library
576 # (wrong share mounted, empty backup mount, ...)
577 if prev_filenames and not cur_filenames:
578 self.logger.error(
579 "Aborting sync for %s: scan found no files but %d were previously indexed",
580 self.name,
581 len(prev_filenames),
582 )
583 report_current_task_failure(
584 f"Sync aborted: scan found no files but {len(prev_filenames)} "
585 "were previously indexed"
586 )
587 return
588
589 # a scan that skipped folders or files is incomplete: what it missed is still
590 # there, so deleting it from the library would throw away valid content
591 if scan_errors.incomplete:
592 summary = scan_errors.describe()
593 self.logger.warning("Skipping deletions for %s: %s", self.name, summary)
594 report_current_task_failure(f"Deletions skipped: {summary}")
595 else:
596 deleted_files = prev_filenames - cur_filenames
597 await self._process_deletions(deleted_files)
598 await self._process_orphaned_albums_and_artists()
599
600 # flag provider as available again if an earlier sync had marked it down
601 self._set_available(True)
602
603 async def get_artist(self, prov_artist_id: str) -> Artist:
604 """Get full artist details by id."""
605 db_artist = await self.mass.music.artists.get_library_item_by_prov_id(
606 prov_artist_id, self.instance_id
607 )
608 if not db_artist:
609 # this may happen if the artist is not in the db yet
610 # e.g. when browsing the filesystem
611 if await self.exists(prov_artist_id):
612 return await self._parse_artist(prov_artist_id, artist_path=prov_artist_id)
613 return await self._parse_artist(prov_artist_id)
614
615 # prov_artist_id is either an actual (relative) path or a name (as fallback)
616 safe_artist_name = create_safe_string(prov_artist_id, lowercase=False, replace_space=False)
617 if await self.exists(prov_artist_id):
618 artist_path = prov_artist_id
619 elif await self.exists(safe_artist_name):
620 artist_path = safe_artist_name
621 else:
622 for prov_mapping in db_artist.provider_mappings:
623 if prov_mapping.provider_instance != self.instance_id:
624 continue
625 if prov_mapping.url:
626 artist_path = prov_mapping.url
627 break
628 else:
629 # this is an artist without an actual path on disk
630 # return the info we already have in the db
631 return db_artist
632 return await self._parse_artist(
633 db_artist.name,
634 sort_name=db_artist.sort_name,
635 mbid=db_artist.mbid,
636 artist_path=artist_path,
637 )
638
639 async def get_album(self, prov_album_id: str) -> Album:
640 """Get full album details by id."""
641 parsed_cue_paths: set[str] = set()
642 for track in await self.get_album_tracks(prov_album_id):
643 for prov_mapping in track.provider_mappings:
644 if prov_mapping.provider_instance != self.instance_id:
645 continue
646 if parsed := parse_cue_track_id(prov_mapping.item_id):
647 # every track from the same CUE shares the same album; only parse once
648 if parsed[0] in parsed_cue_paths:
649 continue
650 parsed_cue_paths.add(parsed[0])
651 cue_item = await self.resolve(parsed[0])
652 for cue_track in await self._cue.parse_tracks(cue_item):
653 if isinstance(cue_track.album, Album):
654 return cue_track.album
655 continue
656 file_item = await self.resolve(prov_mapping.item_id)
657 tags = await async_parse_tags(file_item.absolute_path, file_item.file_size)
658 full_track = await self._parse_track(file_item, tags)
659 assert isinstance(full_track.album, Album)
660 return full_track.album
661 msg = f"Album not found: {prov_album_id}"
662 raise MediaNotFoundError(msg)
663
664 async def get_track(self, prov_track_id: str) -> Track:
665 """Get full track details by id."""
666 # ruff: noqa: PLR0915
667 if parsed := parse_cue_track_id(prov_track_id):
668 cue_item = await self.resolve(parsed[0])
669 for cue_track in await self._cue.parse_tracks(cue_item):
670 if cue_track.item_id == prov_track_id:
671 return cue_track
672 msg = f"CUE track not found: {prov_track_id}"
673 raise MediaNotFoundError(msg)
674
675 if not await self.exists(prov_track_id):
676 msg = f"Track path does not exist: {prov_track_id}"
677 raise MediaNotFoundError(msg)
678
679 file_item = await self.resolve(prov_track_id)
680 tags = await async_parse_tags(file_item.absolute_path, file_item.file_size)
681 return await self._parse_track(file_item, tags=tags, full_album_metadata=True)
682
683 async def get_podcast_episode(self, prov_episode_id: str) -> PodcastEpisode:
684 """Get (full) podcast episode details by id."""
685 if not await self.exists(prov_episode_id):
686 msg = f"Episode path does not exist: {prov_episode_id}"
687 raise MediaNotFoundError(msg)
688 file_item = await self.resolve(prov_episode_id)
689 tags = await async_parse_tags(file_item.absolute_path, file_item.file_size)
690 return await self._parse_podcast_episode(file_item, tags=tags)
691
692 async def get_playlist(self, prov_playlist_id: str) -> Playlist:
693 """Get full playlist details by id."""
694 if not await self.exists(prov_playlist_id):
695 msg = f"Playlist path does not exist: {prov_playlist_id}"
696 raise MediaNotFoundError(msg)
697
698 file_item = await self.resolve(prov_playlist_id)
699 playlist = Playlist(
700 item_id=file_item.relative_path,
701 provider=self.instance_id,
702 name=file_item.name,
703 provider_mappings={
704 ProviderMapping(
705 item_id=file_item.relative_path,
706 provider_domain=self.domain,
707 provider_instance=self.instance_id,
708 details=file_item.checksum,
709 in_library=True,
710 )
711 },
712 )
713 playlist.is_editable = ProviderFeature.PLAYLIST_TRACKS_EDIT in self.supported_features
714 # only playlists in the root are editable - all other are read only
715 if "/" in prov_playlist_id or "\\" in prov_playlist_id:
716 playlist.is_editable = False
717 # we do not (yet) have support to edit/create pls playlists, only m3u files can be edited
718 if file_item.ext == "pls":
719 playlist.is_editable = False
720 playlist.owner = self.name
721 # Check for local image with the same basename
722 if local_image := await self._get_playlist_local_image(file_item):
723 playlist.metadata.images = UniqueList([local_image])
724 return playlist
725
726 async def get_audiobook(self, prov_audiobook_id: str) -> Audiobook:
727 """Get full audiobook details by id."""
728 # ruff: noqa: PLR0915
729 if not await self.exists(prov_audiobook_id):
730 msg = f"Audiobook path does not exist: {prov_audiobook_id}"
731 raise MediaNotFoundError(msg)
732
733 file_item = await self.resolve(prov_audiobook_id)
734 tags = await async_parse_tags(file_item.absolute_path, file_item.file_size)
735 return await self._parse_audiobook(file_item, tags=tags)
736
737 async def get_podcast(self, prov_podcast_id: str) -> Podcast:
738 """Get full podcast details by id."""
739 async for episode in self.get_podcast_episodes(prov_podcast_id):
740 assert isinstance(episode.podcast, Podcast)
741 return episode.podcast
742 msg = f"Podcast not found: {prov_podcast_id}"
743 raise MediaNotFoundError(msg)
744
745 async def get_sound_effect(self, prov_sound_effect_id: str) -> SoundEffect:
746 """Get full sound effect details by id."""
747 if not await self.exists(prov_sound_effect_id):
748 msg = f"Sound effect path does not exist: {prov_sound_effect_id}"
749 raise MediaNotFoundError(msg)
750 file_item = await self.resolve(prov_sound_effect_id)
751 return await self._get_or_parse_sound_effect(file_item)
752
753 async def get_sound_effects(self) -> AsyncGenerator[SoundEffect]:
754 """Get all sound effect items this provider offers."""
755
756 def _walk() -> list[FileSystemItem]:
757 return sorted(
758 recursive_iter(
759 self.base_path, self.base_path, SOUND_EFFECT_EXTENSIONS, self.logger
760 ),
761 key=lambda x: x.relative_path,
762 )
763
764 for file_item in await asyncio.to_thread(_walk):
765 yield await self._get_or_parse_sound_effect(file_item)
766
767 async def get_album_tracks(self, prov_album_id: str) -> list[Track]:
768 """Get album tracks for given album id."""
769 # filesystem items are always stored in db so we can query the database
770 db_album = await self.mass.music.albums.get_library_item_by_prov_id(
771 prov_album_id, self.instance_id
772 )
773 if db_album is None:
774 msg = f"Album not found: {prov_album_id}"
775 raise MediaNotFoundError(msg)
776 album_tracks = await self.mass.music.albums.get_library_album_tracks(db_album.item_id)
777 return [
778 track
779 for track in album_tracks
780 if any(x.provider_instance == self.instance_id for x in track.provider_mappings)
781 ]
782
783 async def get_playlist_tracks(self, prov_playlist_id: str, page: int = 0) -> list[Track]:
784 """Get playlist tracks."""
785 result: list[Track] = []
786 if page > 0:
787 # paging not (yet) supported
788 return result
789 if not await self.exists(prov_playlist_id):
790 msg = f"Playlist path does not exist: {prov_playlist_id}"
791 raise MediaNotFoundError(msg)
792
793 file_item = await self.resolve(prov_playlist_id)
794 # We are using the checksum of the playlist file here to invalidate the cache
795 # when a change has been made to the playlist file (ie track addition/deletion)
796 cache_checksum = file_item.checksum
797
798 cache_key = f"get_playlist_tracks.{prov_playlist_id}"
799 cached_data = await self.mass.cache.get(
800 cache_key,
801 provider=self.instance_id,
802 checksum=cache_checksum,
803 category=0,
804 base_class=Track,
805 )
806 if cached_data is not None:
807 return cached_data # type: ignore[no-any-return]
808
809 _, ext = prov_playlist_id.rsplit(".", 1)
810 try:
811 # get playlist file contents
812 playlist_data_raw = await self._read_file(prov_playlist_id)
813 encoding = await detect_charset(playlist_data_raw)
814 playlist_data = playlist_data_raw.decode(encoding, errors="replace")
815
816 if ext in ("m3u", "m3u8"):
817 playlist_lines = parse_m3u(playlist_data)
818 else:
819 playlist_lines = parse_pls(playlist_data)
820
821 for idx, playlist_line in enumerate(playlist_lines, 1):
822 if "#EXT" in playlist_line.path:
823 continue
824 if track := await self._parse_playlist_line(
825 playlist_line.path, os.path.dirname(prov_playlist_id)
826 ):
827 track.position = idx
828 result.append(track)
829
830 except Exception as err:
831 self.logger.warning(
832 "Error while parsing playlist %s: %s",
833 prov_playlist_id,
834 str(err),
835 exc_info=err if self.logger.isEnabledFor(10) else None,
836 )
837
838 await self.mass.cache.set(
839 key=cache_key,
840 data=[track.to_dict() for track in result],
841 expiration=3600 * 24 * 365, # File timestamp checksum handles invalidation
842 provider=self.instance_id,
843 checksum=cache_checksum,
844 category=0,
845 )
846
847 return result
848
849 async def get_podcast_episodes(self, prov_podcast_id: str) -> AsyncGenerator[PodcastEpisode]:
850 """Get podcast episodes for given podcast id."""
851 folder_items = [item for item in await self._scandir(prov_podcast_id) if not item.is_dir]
852 episode_files = [x for x in folder_items if x.ext in PODCAST_EPISODE_EXTENSIONS]
853 # artwork and metadata.json count towards the signature too, because the parse embeds
854 # them into every episode. Case-insensitive, matching _get_podcast_metadata's exists()
855 signature_files = [
856 x
857 for x in folder_items
858 if x.ext in PODCAST_EPISODE_EXTENSIONS
859 or x.ext in IMAGE_EXTENSIONS
860 or x.filename.lower() == "metadata.json"
861 ]
862 cache_key = f"podcast_episodes.{prov_podcast_id}"
863 cache_checksum = get_folder_signature(signature_files)
864 if (
865 cached_episodes := await self.mass.cache.get(
866 cache_key,
867 provider=self.instance_id,
868 category=CACHE_CATEGORY_PODCAST_EPISODES,
869 checksum=cache_checksum,
870 base_class=PodcastEpisode,
871 )
872 ) is not None:
873 for episode in cached_episodes:
874 yield episode
875 return
876
877 # these caches have no checksum of their own, so drop them before parsing or the new
878 # entry gets the values the signature just invalidated. Refill once, or every parse
879 # task below misses at the same time and repeats the same scandir and file read
880 for stale_category in (CACHE_CATEGORY_FOLDER_IMAGES, CACHE_CATEGORY_PODCAST_METADATA):
881 await self.mass.cache.delete(
882 prov_podcast_id, category=stale_category, provider=self.instance_id
883 )
884 await self._get_local_images(prov_podcast_id)
885 await self._get_podcast_metadata(prov_podcast_id)
886
887 # collected by index so the listing keeps scandir order, not parse completion order
888 parsed: list[PodcastEpisode | None] = [None] * len(episode_files)
889
890 async def _process_podcast_episode(index: int, item: FileSystemItem) -> None:
891 try:
892 tags = await async_parse_tags(item.absolute_path, item.file_size)
893 parsed[index] = await self._parse_podcast_episode(item, tags)
894 except MusicAssistantError as err:
895 self.logger.warning(
896 "Could not parse uri/file %s to podcast episode: %s",
897 item.relative_path,
898 str(err),
899 )
900
901 # reuse the per-sync worker limit: the slowest filesystems to parse are exactly the
902 # ones that lower it
903 async with TaskManager(self.mass, self._SYNC_CONCURRENCY) as tm:
904 for index, item in enumerate(episode_files):
905 await tm.create_task_with_limit(_process_podcast_episode(index, item))
906
907 episodes = [episode for episode in parsed if episode is not None]
908 # cache an incomplete listing briefly rather than not at all, so one unreadable file
909 # cannot make every request re-parse the whole folder
910 complete = len(episodes) == len(episode_files)
911 await self.mass.cache.set(
912 key=cache_key,
913 data=[episode.to_dict() for episode in episodes],
914 # a complete listing is invalidated by the folder signature instead
915 expiration=3600 * 24 * 365 if complete else PARTIAL_LISTING_CACHE_EXPIRATION,
916 provider=self.instance_id,
917 category=CACHE_CATEGORY_PODCAST_EPISODES,
918 checksum=cache_checksum,
919 )
920
921 for episode in episodes:
922 yield episode
923
924 async def add_playlist_tracks(self, prov_playlist_id: str, prov_track_ids: list[str]) -> None:
925 """Add track(s) to playlist."""
926 if not await self.exists(prov_playlist_id):
927 msg = f"Playlist path does not exist: {prov_playlist_id}"
928 raise MediaNotFoundError(msg)
929 playlist_filename = self.get_absolute_path(prov_playlist_id)
930 async with aiofiles.open(playlist_filename, encoding="utf-8") as _file:
931 playlist_data = await _file.read()
932 for file_path in prov_track_ids:
933 track = await self.get_track(file_path)
934 playlist_data += f"\n#EXTINF:{track.duration or 0},{track.name}\n{file_path}\n"
935
936 # write playlist file (always in utf-8)
937 async with aiofiles.open(playlist_filename, "w", encoding="utf-8") as _file:
938 await _file.write(playlist_data)
939
940 async def remove_playlist_tracks(
941 self, prov_playlist_id: str, positions_to_remove: tuple[int, ...]
942 ) -> None:
943 """Remove track(s) from playlist."""
944 if not await self.exists(prov_playlist_id):
945 msg = f"Playlist path does not exist: {prov_playlist_id}"
946 raise MediaNotFoundError(msg)
947 _, ext = prov_playlist_id.rsplit(".", 1)
948 # get playlist file contents
949 playlist_filename = self.get_absolute_path(prov_playlist_id)
950 async with aiofiles.open(playlist_filename, encoding="utf-8") as _file:
951 playlist_data = await _file.read()
952 # get current contents first
953 if ext in ("m3u", "m3u8"):
954 playlist_items = parse_m3u(playlist_data)
955 else:
956 playlist_items = parse_pls(playlist_data)
957 # remove items by index
958 for i in sorted(positions_to_remove, reverse=True):
959 # position = index + 1
960 del playlist_items[i - 1]
961 # build new playlist data
962 new_playlist_data = "#EXTM3U\n"
963 for item in playlist_items:
964 new_playlist_data += f"\n#EXTINF:{item.length or 0},{item.title}\n{item.path}\n"
965 async with aiofiles.open(playlist_filename, "w", encoding="utf-8") as _file:
966 await _file.write(new_playlist_data)
967
968 async def create_playlist(self, name: str, media_types: set[MediaType]) -> Playlist:
969 """Create a new playlist on provider with given name."""
970 # creating a new playlist on the filesystem is as easy
971 # as creating a new (empty) file with the m3u extension...
972 # filename = await self.resolve(f"{name}.m3u")
973 filename = f"{name}.m3u"
974 playlist_filename = self.get_absolute_path(filename)
975 async with aiofiles.open(playlist_filename, "w", encoding="utf-8") as _file:
976 await _file.write("#EXTM3U\n")
977 return await self.get_playlist(filename)
978
979 async def get_stream_details(self, item_id: str, media_type: MediaType) -> StreamDetails:
980 """Return the content details for the given track when it will be streamed."""
981 try:
982 if media_type == MediaType.AUDIOBOOK:
983 return await self._get_stream_details_for_audiobook(item_id)
984 if media_type == MediaType.PODCAST_EPISODE:
985 return await self._get_stream_details_for_podcast_episode(item_id)
986 if media_type == MediaType.SOUND_EFFECT:
987 return await self._get_stream_details_for_sound_effect(item_id)
988 return await self._get_stream_details_for_track(item_id)
989 except FileNotFoundError:
990 self.logger.warning(
991 "File not found for media item %s",
992 item_id,
993 )
994 msg = f"Media file not found: {item_id}"
995 raise MediaNotFoundError(msg)
996
997 async def get_audio_stream(
998 self, streamdetails: StreamDetails, seek_position: int = 0
999 ) -> AsyncGenerator[bytes]:
1000 """Return the custom audio stream for the provider item."""
1001 # only CUE-derived tracks use StreamType.CUSTOM in this provider
1002 async for chunk in self._cue.get_audio_stream(streamdetails, seek_position):
1003 yield chunk
1004
1005 async def resolve_image(self, path: str) -> str | bytes:
1006 """
1007 Resolve an image from an image path.
1008
1009 This either returns (a generator to get) raw bytes of the image or
1010 a string with an http(s) URL or local path that is accessible from the server.
1011 """
1012 # drop the cache-busting suffix appended by _versioned_image_path
1013 try:
1014 file_item = await self.resolve(path.split("?cs=", 1)[0])
1015 except FileNotFoundError as err:
1016 # the referenced image file was removed from disk; surface a typed
1017 # not-found so the image layer treats it as a missing image
1018 raise MediaNotFoundError(f"Image not found: {path}") from err
1019 if file_item.is_dir:
1020 # handing the path back would have the image layer run an ffmpeg
1021 # embedded-artwork extraction on the directory before giving up
1022 raise MediaNotFoundError(f"Image path is a directory: {path}")
1023 return file_item.absolute_path
1024
1025 async def check_write_access(self) -> None:
1026 """Perform check if we have write access."""
1027 # verify write access to determine we have playlist create/edit support
1028 # overwrite with provider specific implementation if needed
1029 temp_file_name = self.get_absolute_path(f"{shortuuid.random(8)}.txt")
1030 try:
1031 async with aiofiles.open(temp_file_name, "w") as _file:
1032 await _file.write("test")
1033 await asyncio.to_thread(os.remove, temp_file_name)
1034 self.write_access = True
1035 except Exception as err:
1036 self.logger.debug("Write access disabled: %s", str(err))
1037
1038 async def resolve(self, file_path: str) -> FileSystemItem:
1039 """Resolve (absolute or relative) path to FileSystemItem."""
1040 absolute_path = self.get_absolute_path(file_path)
1041
1042 def _create_item() -> FileSystemItem:
1043 if Path(absolute_path).is_dir():
1044 return FileSystemItem(
1045 filename=Path(file_path).name,
1046 relative_path=get_relative_path(self.base_path, file_path),
1047 absolute_path=absolute_path,
1048 is_dir=True,
1049 )
1050 stat_info = Path(absolute_path).stat(follow_symlinks=False)
1051 return FileSystemItem(
1052 filename=Path(file_path).name,
1053 relative_path=get_relative_path(self.base_path, file_path),
1054 absolute_path=absolute_path,
1055 is_dir=False,
1056 checksum=str(int(stat_info.st_mtime)),
1057 file_size=stat_info.st_size,
1058 )
1059
1060 return await asyncio.to_thread(_create_item)
1061
1062 async def exists(self, file_path: str) -> bool:
1063 """Return bool is this FileSystem musicprovider has given file/dir."""
1064 if not file_path:
1065 return False
1066 try:
1067 abs_path = self.get_absolute_path(file_path)
1068 except MediaNotFoundError:
1069 # a path that escapes the base directory simply does not exist here
1070 return False
1071 return bool(await exists(abs_path))
1072
1073 def get_absolute_path(self, file_path: str) -> str:
1074 """Return absolute path for given file path."""
1075 return get_absolute_path(self.base_path, file_path)
1076
1077 async def _enumerate_files_for_sync(
1078 self,
1079 *,
1080 file_checksums: dict[str, str],
1081 cue_file_checksums: dict[str, set[str]],
1082 cur_filenames: set[str],
1083 items_to_process: list[tuple[FileSystemItem, str | None]],
1084 unchanged_cue_items: list[FileSystemItem],
1085 cue_stems: set[str],
1086 scan_errors: ScanErrors,
1087 ) -> None:
1088 """
1089 Walk every supported file under the provider root and populate the sync buckets.
1090
1091 Override in subclasses that cannot use a local ``os.scandir`` walk.
1092 Implementations must route each discovered file through
1093 :meth:`_classify_scan_item`, report every unreadable directory to
1094 ``scan_errors`` and stop the walk once it reports ``aborted``.
1095
1096 :param file_checksums: Previously stored checksum per provider item id.
1097 :param cue_file_checksums: Previously stored track checksums keyed by CUE relative_path.
1098 :param cur_filenames: Receives the ids/paths present in this scan.
1099 :param items_to_process: Receives changed or new items to process.
1100 :param unchanged_cue_items: Receives CUE sheets whose checksum matches.
1101 :param cue_stems: Receives absolute paths (minus extension) of CUE sheets.
1102 :param scan_errors: Receives the errors raised while walking the tree.
1103 """
1104 ignore_album_playlists = self.media_content_type == "music" and bool(
1105 self.config.get_value(CONF_ENTRY_IGNORE_ALBUM_PLAYLISTS.key)
1106 )
1107
1108 def _walk() -> None:
1109 for scanned, item in enumerate(
1110 recursive_iter(
1111 self.base_path,
1112 self.base_path,
1113 SUPPORTED_EXTENSIONS,
1114 self.logger,
1115 scan_errors=scan_errors,
1116 ),
1117 start=1,
1118 ):
1119 if scanned % 500 == 0:
1120 update_current_task_progress_text(f"Scanning files: {scanned} found")
1121 self._classify_scan_item(
1122 item,
1123 file_checksums=file_checksums,
1124 cue_file_checksums=cue_file_checksums,
1125 cur_filenames=cur_filenames,
1126 items_to_process=items_to_process,
1127 unchanged_cue_items=unchanged_cue_items,
1128 cue_stems=cue_stems,
1129 ignore_album_playlists=ignore_album_playlists,
1130 )
1131
1132 await asyncio.to_thread(_walk)
1133
1134 def _classify_scan_item(
1135 self,
1136 item: FileSystemItem,
1137 *,
1138 file_checksums: dict[str, str],
1139 cue_file_checksums: dict[str, set[str]],
1140 cur_filenames: set[str],
1141 items_to_process: list[tuple[FileSystemItem, str | None]],
1142 unchanged_cue_items: list[FileSystemItem],
1143 cue_stems: set[str],
1144 ignore_album_playlists: bool,
1145 ) -> None:
1146 """
1147 Route a single scanned file into the correct sync bucket.
1148
1149 :param item: The file to classify.
1150 :param file_checksums: Previously stored checksum per provider item id.
1151 :param cue_file_checksums: Previously stored track checksums keyed by CUE relative_path.
1152 :param cur_filenames: Receives the ids/paths present in this scan.
1153 :param items_to_process: Receives changed or new items to process.
1154 :param unchanged_cue_items: Receives CUE sheets whose checksum matches.
1155 :param cue_stems: Receives absolute paths (minus extension) of CUE sheets.
1156 :param ignore_album_playlists: When True, skip playlists nested inside
1157 album directories.
1158 """
1159 # a file this provider never imports gets no mapping, so it would flag as
1160 # changed on every sync; it is still on disk, so record it as present
1161 if not self._is_imported_file(item):
1162 cur_filenames.add(item.relative_path)
1163 return
1164 # skip playlists in album directories if configured
1165 if (
1166 item.ext in PLAYLIST_EXTENSIONS
1167 and ignore_album_playlists
1168 and len(item.relative_path.split("/")) > 2
1169 ):
1170 return
1171 is_cue = item.ext in CUE_EXTENSIONS and self.media_content_type == "music"
1172 item_checksum = item.checksum
1173 if is_cue:
1174 cue_stems.add(item.absolute_path.rsplit(".", 1)[0])
1175 item_checksum = cue_metadata_checksum(item.checksum)
1176 prev_checksums = cue_file_checksums.get(item.relative_path, set())
1177 prev_checksum = min(prev_checksums, default=None)
1178 checksum_matches = prev_checksums == {item_checksum}
1179 else:
1180 prev_checksum = file_checksums.get(item.relative_path)
1181 checksum_matches = item_checksum == prev_checksum
1182 if checksum_matches:
1183 # unchanged, just record it as still present
1184 cur_filenames.add(item.relative_path)
1185 if is_cue:
1186 unchanged_cue_items.append(item)
1187 else:
1188 items_to_process.append((item, prev_checksum))
1189
1190 def _is_imported_file(self, item: FileSystemItem) -> bool:
1191 """Return True when this provider imports the given file into the library."""
1192 if self.media_content_type == "music":
1193 if item.ext in CUE_EXTENSIONS:
1194 return True
1195 if item.ext in TRACK_EXTENSIONS:
1196 return self._sync_tracks
1197 if item.ext in PLAYLIST_EXTENSIONS:
1198 return self._sync_playlists
1199 return False
1200 if self.media_content_type == "audiobooks":
1201 return item.ext in AUDIOBOOK_EXTENSIONS
1202 if self.media_content_type == "podcasts":
1203 return item.ext in PODCAST_EPISODE_EXTENSIONS
1204 return False
1205
1206 def _set_available(self, available: bool) -> None:
1207 """Update the provider availability and notify listeners on change."""
1208 if self.available == available:
1209 return
1210 self.available = available
1211 if available:
1212 self._cancel_availability_probe()
1213 else:
1214 self._schedule_availability_probe()
1215 self.mass.signal_event(EventType.PROVIDERS_UPDATED, data=self.mass.get_providers())
1216
1217 async def _is_reachable(self) -> bool:
1218 """Return whether the storage backing this provider can be read."""
1219 return bool(await isdir(self.base_path))
1220
1221 @property
1222 def _availability_probe_id(self) -> str:
1223 """Return the timer id of this provider's reachability checks."""
1224 return f"filesystem_availability_probe_{self.instance_id}"
1225
1226 def _schedule_availability_probe(self) -> None:
1227 """Arm the next reachability check."""
1228 self.mass.call_later(
1229 AVAILABILITY_PROBE_INTERVAL,
1230 self._probe_availability,
1231 task_id=self._availability_probe_id,
1232 )
1233
1234 def _cancel_availability_probe(self) -> None:
1235 """Stop checking for the storage coming back."""
1236 self.mass.cancel_timer(self._availability_probe_id)
1237
1238 async def _probe_availability(self) -> None:
1239 """Mark the provider available again once its storage can be read."""
1240 try:
1241 reachable = await self._is_reachable()
1242 except MusicAssistantError as err:
1243 # storage that is simply still gone, which is what this loop waits for
1244 self.logger.debug("%s is still unreachable: %s", self.name, err)
1245 reachable = False
1246 except Exception:
1247 # an unexpected failure must not end the loop, since it is what brings the
1248 # provider back, but it is a defect rather than an outage so it is logged loudly
1249 self.logger.exception("Reachability check for %s failed", self.name)
1250 reachable = False
1251 if self.unloading:
1252 # the provider was torn down while this check was running; re-arming here
1253 # would leave a timer firing against an instance nothing owns anymore
1254 return
1255 if reachable:
1256 self.logger.info("%s is reachable again", self.name)
1257 self._set_available(True)
1258 return
1259 self._schedule_availability_probe()
1260
1261 async def _process_item_async(
1262 self,
1263 item: FileSystemItem,
1264 prev_checksum: str | None,
1265 cur_filenames: set[str] | None = None,
1266 cue_stems: set[str] | None = None,
1267 prev_filenames: set[str] | None = None,
1268 ) -> bool:
1269 """
1270 Process a single item asynchronously.
1271
1272 :param item: The filesystem item to process.
1273 :param prev_checksum: Previous checksum from the database, or None for new items.
1274 :param cur_filenames: Set of current filenames being tracked (for CUE track IDs).
1275 :param cue_stems: Absolute paths (without extension) of CUE sheets in this scan,
1276 used to detect companion-CUE audio files without a filesystem stat.
1277 :param prev_filenames: The ids/paths the previous scan found, used to keep the
1278 ids of a CUE sheet that fails to parse.
1279 """
1280 try:
1281 self.logger.log(VERBOSE_LOG_LEVEL, "Processing: %s", item.relative_path)
1282
1283 if prev_checksum is not None:
1284 # the file changed on disk: drop cached artwork derived from it
1285 # (thumbnails, source bytes, palette) so re-read embedded art is
1286 # served fresh, for both reference forms of the image path
1287 await self.mass.metadata.invalidate_image_cache(
1288 self.instance_id, item.relative_path
1289 )
1290 await self.mass.metadata.invalidate_image_cache(
1291 self.instance_id, self._versioned_image_path(item.relative_path, prev_checksum)
1292 )
1293
1294 if item.ext in CUE_EXTENSIONS and self.media_content_type == "music":
1295 tracks = await self._cue.parse_tracks(item)
1296 for track in tracks:
1297 track.favorite = False
1298 await self.mass.music.tracks.add_item_to_library(
1299 track, overwrite_existing=prev_checksum is not None
1300 )
1301 if cur_filenames is not None:
1302 cur_filenames.add(track.item_id)
1303 return True
1304
1305 if item.ext in TRACK_EXTENSIONS and self.media_content_type == "music":
1306 if not self._sync_tracks:
1307 return False
1308 # skip audio files that have a companion CUE sheet
1309 if cue_stems is not None and item.absolute_path.rsplit(".", 1)[0] in cue_stems:
1310 return False
1311 tags = await async_parse_tags(item.absolute_path, item.file_size)
1312 track = await self._parse_track(item, tags)
1313 track.favorite = False # TODO: implement favorite status based on rating ?
1314 await self.mass.music.tracks.add_item_to_library(
1315 track, overwrite_existing=prev_checksum is not None
1316 )
1317 return True
1318
1319 if item.ext in AUDIOBOOK_EXTENSIONS and self.media_content_type == "audiobooks":
1320 tags = await async_parse_tags(item.absolute_path, item.file_size)
1321 try:
1322 audiobook = await self._parse_audiobook(item, tags)
1323 except IsChapterFile:
1324 return True
1325 await self.mass.music.audiobooks.add_item_to_library(
1326 audiobook, overwrite_existing=prev_checksum is not None
1327 )
1328 return True
1329
1330 if item.ext in PODCAST_EPISODE_EXTENSIONS and self.media_content_type == "podcasts":
1331 tags = await async_parse_tags(item.absolute_path, item.file_size)
1332 episode = await self._parse_podcast_episode(item, tags)
1333 assert isinstance(episode.podcast, Podcast)
1334 await self.mass.music.podcasts.add_item_to_library(
1335 episode.podcast, overwrite_existing=prev_checksum is not None
1336 )
1337 return True
1338
1339 if item.ext in PLAYLIST_EXTENSIONS and self.media_content_type == "music":
1340 if not self._sync_playlists:
1341 return False
1342 playlist = await self.get_playlist(item.relative_path)
1343 await self.mass.music.playlists.add_item_to_library(
1344 playlist, overwrite_existing=prev_checksum is not None
1345 )
1346 return True
1347
1348 except Exception as err:
1349 # we don't want the whole sync to crash on one file so we catch all exceptions here
1350 self.logger.error(
1351 "Error processing %s - %s",
1352 item.relative_path,
1353 str(err),
1354 exc_info=err if self.logger.isEnabledFor(logging.DEBUG) else None,
1355 )
1356 report_current_task_failure(f"Failed to process {item.relative_path}: {err}")
1357 # the file is still on the storage, so keep it in the scan result:
1358 # leaving it out makes the deletion step treat it as removed
1359 self._keep_failed_item(item, cur_filenames, prev_filenames)
1360 return False
1361
1362 def _keep_failed_item(
1363 self,
1364 item: FileSystemItem,
1365 cur_filenames: set[str] | None,
1366 prev_filenames: set[str] | None,
1367 ) -> None:
1368 """
1369 Keep an item that could not be processed in the scan result.
1370
1371 :param item: The item that failed to process.
1372 :param cur_filenames: Receives the ids/paths present in this scan.
1373 :param prev_filenames: The ids/paths the previous scan found.
1374 """
1375 if cur_filenames is None:
1376 return
1377 cur_filenames.add(item.relative_path)
1378 if (
1379 item.ext not in CUE_EXTENSIONS
1380 or self.media_content_type != "music"
1381 or not prev_filenames
1382 ):
1383 return
1384 # a CUE sheet stands in for one id per track it describes and those cannot be
1385 # rebuilt without parsing it, so carry over the ids of the previous scan
1386 cur_filenames.update(
1387 item_id
1388 for item_id in prev_filenames
1389 if (parsed := parse_cue_track_id(item_id)) and parsed[0] == item.relative_path
1390 )
1391
1392 async def _process_orphaned_albums_and_artists(self) -> None:
1393 """Process deletion of orphaned albums and artists."""
1394 assert self.mass.music.database
1395 # Remove albums without any tracks
1396 query = (
1397 f"SELECT item_id FROM {DB_TABLE_ALBUMS} "
1398 f"WHERE item_id not in ( SELECT album_id from {DB_TABLE_ALBUM_TRACKS}) "
1399 f"AND item_id in ( SELECT item_id from {DB_TABLE_PROVIDER_MAPPINGS} "
1400 f"WHERE provider_instance = '{self.instance_id}' and media_type = 'album' )"
1401 )
1402 for db_row in await self.mass.music.database.get_rows_from_query(
1403 query,
1404 limit=100000,
1405 ):
1406 await self.mass.music.albums.remove_item_from_library(db_row["item_id"])
1407
1408 # Remove artists without any tracks or albums
1409 query = (
1410 f"SELECT item_id FROM {DB_TABLE_ARTISTS} "
1411 f"WHERE item_id not in "
1412 f"( select artist_id from {DB_TABLE_TRACK_ARTISTS} "
1413 f"UNION SELECT artist_id from {DB_TABLE_ALBUM_ARTISTS} ) "
1414 f"AND item_id in ( SELECT item_id from {DB_TABLE_PROVIDER_MAPPINGS} "
1415 f"WHERE provider_instance = '{self.instance_id}' and media_type = 'artist' )"
1416 )
1417 for db_row in await self.mass.music.database.get_rows_from_query(
1418 query,
1419 limit=100000,
1420 ):
1421 await self.mass.music.artists.remove_item_from_library(db_row["item_id"])
1422
1423 async def _process_deletions(self, deleted_files: set[str]) -> None:
1424 """Process all deletions."""
1425 # process deleted tracks/playlists
1426 album_ids = set()
1427 artist_ids = set()
1428 for file_path in deleted_files:
1429 if parse_cue_track_id(file_path) is not None and self.media_content_type == "music":
1430 controller = self.mass.music.get_controller(MediaType.TRACK)
1431 elif "." not in file_path:
1432 continue
1433 else:
1434 _, ext = file_path.rsplit(".", 1)
1435 if ext in PODCAST_EPISODE_EXTENSIONS and self.media_content_type == "podcasts":
1436 controller = self.mass.music.get_controller(MediaType.PODCAST_EPISODE)
1437 elif ext in AUDIOBOOK_EXTENSIONS and self.media_content_type == "audiobooks":
1438 controller = self.mass.music.get_controller(MediaType.AUDIOBOOK)
1439 elif ext in PLAYLIST_EXTENSIONS and self.media_content_type == "music":
1440 controller = self.mass.music.get_controller(MediaType.PLAYLIST)
1441 elif ext in TRACK_EXTENSIONS and self.media_content_type == "music":
1442 controller = self.mass.music.get_controller(MediaType.TRACK)
1443 else:
1444 # unsupported file extension?
1445 continue
1446
1447 if library_item := await controller.get_library_item_by_prov_id(
1448 file_path, self.instance_id
1449 ):
1450 if is_track(library_item):
1451 if library_item.album:
1452 album_ids.add(library_item.album.item_id)
1453 # need to fetch the library album to resolve the itemmapping
1454 db_album = await self.mass.music.albums.get_library_item(
1455 library_item.album.item_id
1456 )
1457 for artist in db_album.artists:
1458 artist_ids.add(artist.item_id)
1459 for artist in library_item.artists:
1460 artist_ids.add(artist.item_id)
1461 await controller.remove_item_from_library(library_item.item_id)
1462 # check if any albums need to be cleaned up
1463 for album_id in album_ids:
1464 if not await self.mass.music.albums.tracks(album_id, "library"):
1465 await self.mass.music.albums.remove_item_from_library(album_id)
1466 # check if any artists need to be cleaned up
1467 for artist_id in artist_ids:
1468 artist_albums = await self.mass.music.artists.albums(artist_id, "library")
1469 artist_tracks = await self.mass.music.artists.tracks(artist_id, "library")
1470 if not (artist_albums or artist_tracks):
1471 await self.mass.music.artists.remove_item_from_library(artist_id)
1472
1473 async def _get_playlist_local_image(self, file_item: FileSystemItem) -> MediaItemImage | None:
1474 """Return a local image alongside the playlist file (matching basename) if any."""
1475 cache_key = f"playlist_image.{file_item.relative_path}"
1476 cached = await self.cache.get(
1477 key=cache_key,
1478 provider=self.instance_id,
1479 category=CACHE_CATEGORY_FOLDER_IMAGES,
1480 base_class=MediaItemImage,
1481 )
1482 if cached is not None:
1483 return cached[0] if cached else None
1484 try:
1485 folder_files = await self._scandir(file_item.relative_parent_path)
1486 except OSError, MusicAssistantError:
1487 return None
1488 target = file_item.name.lower()
1489 result: MediaItemImage | None = None
1490 for item in folder_files:
1491 if item.is_dir or not item.ext:
1492 continue
1493 if item.ext.lower() not in IMAGE_EXTENSIONS:
1494 continue
1495 if item.name.lower() != target:
1496 continue
1497 result = MediaItemImage(
1498 type=ImageType.THUMB,
1499 path=item.relative_path,
1500 provider=self.instance_id,
1501 remotely_accessible=False,
1502 )
1503 break
1504 await self.cache.set(
1505 key=cache_key,
1506 data=[result.to_dict()] if result is not None else [],
1507 provider=self.instance_id,
1508 category=CACHE_CATEGORY_FOLDER_IMAGES,
1509 expiration=120,
1510 )
1511 return result
1512
1513 async def _parse_playlist_line(self, line: str, playlist_path: str) -> Track | None:
1514 """Try to parse a track from a playlist line."""
1515 try:
1516 line = line.replace("file://", "").strip()
1517 # try to resolve the filename (both normal and url decoded):
1518 # - relative to the playlist folder (normpath resolves parent .. references)
1519 # - as-is: an absolute path, or relative to our base path
1520 # candidates stay relative so subclasses with virtual paths (cloud,
1521 # webdav) resolve them too, instead of leaking the server CWD
1522 for _line in (line, urllib.parse.unquote(line)):
1523 if playlist_path:
1524 normalized = posixpath.normpath(f"{playlist_path}/{_line}")
1525 with contextlib.suppress(FileNotFoundError, MediaNotFoundError):
1526 file_item = await self.resolve(normalized)
1527 tags = await async_parse_tags(file_item.absolute_path, file_item.file_size)
1528 return await self._parse_track(file_item, tags)
1529 with contextlib.suppress(FileNotFoundError, MediaNotFoundError):
1530 file_item = await self.resolve(_line)
1531 tags = await async_parse_tags(file_item.absolute_path, file_item.file_size)
1532 return await self._parse_track(file_item, tags)
1533 # all attempts failed
1534 raise MediaNotFoundError("Invalid path/uri")
1535
1536 except MusicAssistantError as err:
1537 self.logger.warning("Could not parse %s to track: %s", line, str(err))
1538
1539 return None
1540
1541 @staticmethod
1542 def _versioned_image_path(relative_path: str, checksum: str | None) -> str:
1543 """Append the file checksum so the image cache busts when the file is replaced."""
1544 if checksum:
1545 return f"{relative_path}?cs={checksum}"
1546 return relative_path
1547
1548 async def _parse_track(
1549 self, file_item: FileSystemItem, tags: AudioTags, full_album_metadata: bool = False
1550 ) -> Track:
1551 """Parse full track details from file tags."""
1552 # ruff: noqa: PLR0915
1553 name, version = parse_title_and_version(tags.title, tags.version)
1554 track = Track(
1555 item_id=file_item.relative_path,
1556 provider=self.instance_id,
1557 name=name,
1558 sort_name=tags.title_sort,
1559 version=version,
1560 provider_mappings={
1561 ProviderMapping(
1562 item_id=file_item.relative_path,
1563 provider_domain=self.domain,
1564 provider_instance=self.instance_id,
1565 audio_format=AudioFormat(
1566 content_type=ContentType.try_parse(file_item.ext or tags.format),
1567 sample_rate=tags.sample_rate,
1568 bit_depth=tags.bits_per_sample,
1569 channels=tags.channels,
1570 bit_rate=tags.bit_rate,
1571 ),
1572 details=file_item.checksum,
1573 in_library=True,
1574 )
1575 },
1576 disc_number=tags.disc or 0,
1577 track_number=tags.track or 0,
1578 date_added=(
1579 datetime.fromtimestamp(file_item.created_at, tz=UTC)
1580 if file_item.created_at
1581 else None
1582 ),
1583 )
1584
1585 if isrc_tags := tags.isrc:
1586 for isrsc in isrc_tags:
1587 track.external_ids.add((ExternalID.ISRC, isrsc))
1588
1589 if acoustid := tags.get("acoustid"):
1590 track.external_ids.add((ExternalID.ACOUSTID, acoustid))
1591
1592 # album
1593 album = track.album = (
1594 await self._parse_album(
1595 track_path=file_item.relative_path,
1596 track_tags=tags,
1597 track_created_at=file_item.created_at,
1598 )
1599 if tags.album
1600 else None
1601 )
1602
1603 # track artist(s)
1604 resolved_track_artists = await self._resolve_artists_with_mbids(
1605 tags.artists,
1606 tags.musicbrainz_artistids,
1607 tags.artist_sort_names,
1608 log_label="ARTISTS tag",
1609 )
1610 for name, mbid, sort_name in resolved_track_artists:
1611 # prefer the existing album artist object when it's the same artist
1612 if album_artist_match := self._match_album_artist(album, name, mbid):
1613 track.artists.append(album_artist_match)
1614 continue
1615 artist = await self._parse_artist(name, sort_name=sort_name, mbid=mbid)
1616 track.artists.append(artist)
1617
1618 # handle embedded cover image
1619 if tags.has_cover_image:
1620 # we do not actually embed the image in the metadata because that would consume too
1621 # much space and bandwidth. Instead we set the filename as value so the image can
1622 # be retrieved later in realtime.
1623 track.metadata.images = UniqueList(
1624 [
1625 MediaItemImage(
1626 type=ImageType.THUMB,
1627 path=file_item.relative_path,
1628 provider=self.instance_id,
1629 remotely_accessible=False,
1630 )
1631 ]
1632 )
1633
1634 # copy (embedded) album image from track (if the album itself doesn't have an image)
1635 if album and not album.image and track.image:
1636 album.metadata.images = UniqueList([track.image])
1637
1638 # parse other info
1639 track.duration = int(tags.duration or 0)
1640 track.metadata.genres = set(tags.genres)
1641 if tags.disc:
1642 track.disc_number = tags.disc
1643 if tags.track:
1644 track.track_number = tags.track
1645 track.metadata.copyright = tags.get("copyright")
1646 track.metadata.lyrics = tags.lyrics
1647 track.metadata.grouping = tags.get("grouping")
1648 track.metadata.description = tags.get("comment")
1649 explicit_tag = tags.get("itunesadvisory")
1650 if explicit_tag is not None:
1651 track.metadata.explicit = explicit_tag == "1"
1652 if recording_mbid := clean_mbid(tags.musicbrainz_recordingid, tags.filename):
1653 track.mbid = recording_mbid
1654
1655 # handle (optional) loudness measurement tag(s)
1656 if tags.track_loudness is not None:
1657 self.mass.create_task(
1658 self.mass.streams.audio_analysis.set_track_loudness(
1659 track.item_id,
1660 self.instance_id,
1661 tags.track_loudness,
1662 tags.track_album_loudness,
1663 )
1664 )
1665
1666 # possible lrclib metadata
1667 # synced lyrics are saved as "filename.lrc" by lrcget alongside
1668 # the actual file location - just change the file extension
1669 assert file_item.ext is not None # for type checking
1670 lrc_path = f"{file_item.relative_path.removesuffix(file_item.ext)}lrc"
1671 if await self.exists(lrc_path):
1672 try:
1673 raw = await self._read_file(lrc_path)
1674 track.metadata.lrc_lyrics = raw.decode("utf-8")
1675 except Exception as err:
1676 self.logger.warning(
1677 "Failed to read lyrics file %s: %s",
1678 lrc_path,
1679 str(err),
1680 )
1681 elif syn_lyrics := tags.synchronized_lyrics:
1682 track.metadata.lrc_lyrics = lyrics.convert_to_lrc_lyrics(syn_lyrics)
1683
1684 return track
1685
1686 async def _resolve_artists_with_mbids(
1687 self,
1688 parsed_names: tuple[str, ...],
1689 mbids: tuple[str, ...],
1690 sort_names: tuple[str, ...],
1691 log_label: str,
1692 ) -> list[tuple[str, str | None, str | None]]:
1693 """
1694 Return ``(name, mbid, sort_name)`` triples for a track's or album's artists.
1695
1696 When the parsed name count and the MBID count disagree, canonical names
1697 are looked up from MusicBrainz; otherwise the tag-parsed names are used.
1698
1699 :param parsed_names: Tag-parsed artist names.
1700 :param mbids: MusicBrainz artist IDs from the tag.
1701 :param sort_names: Sort names from the corresponding *sort tag.
1702 :param log_label: Tag name used in warning messages (e.g. "ARTISTS tag").
1703 """
1704
1705 def _sort_name(index: int) -> str | None:
1706 return sort_names[index] if index < len(sort_names) else None
1707
1708 def _from_tags() -> list[tuple[str, str | None, str | None]]:
1709 return [
1710 (
1711 name,
1712 mbids[i] if i < len(mbids) else None,
1713 _sort_name(i),
1714 )
1715 for i, name in enumerate(parsed_names)
1716 ]
1717
1718 if not mbids or len(parsed_names) == len(mbids):
1719 return _from_tags()
1720
1721 mb_provider = cast("MusicbrainzProvider | None", self.mass.get_provider("musicbrainz"))
1722 if mb_provider is None:
1723 self.logger.warning(
1724 "%s count (%d) doesn't match MBID count (%d) and MusicBrainz "
1725 "provider is not loaded; using tag-parsed names: %s",
1726 log_label,
1727 len(parsed_names),
1728 len(mbids),
1729 parsed_names,
1730 )
1731 return _from_tags()
1732
1733 mb_results = await mb_provider.resolve_artists_from_mbids(mbids)
1734 # counts disagree, so positional fallback to a tag name is unreliable;
1735 # drop any MBID whose lookup failed (already logged per-MBID)
1736 resolved: list[tuple[str, str | None, str | None]] = [
1737 mb_result for mb_result in mb_results if mb_result is not None
1738 ]
1739 if not resolved:
1740 self.logger.warning(
1741 "%s count (%d) didn't match MBID count (%d) and every MusicBrainz "
1742 "lookup failed; falling back to tag-parsed names: %s",
1743 log_label,
1744 len(parsed_names),
1745 len(mbids),
1746 parsed_names,
1747 )
1748 return _from_tags()
1749 self.logger.info(
1750 "%s count (%d) didn't match MBID count (%d); resolved canonical names "
1751 "via MusicBrainz: %s",
1752 log_label,
1753 len(parsed_names),
1754 len(mbids),
1755 [r[0] for r in resolved],
1756 )
1757 return resolved
1758
1759 def _match_album_artist(
1760 self, album: Album | None, name: str, mbid: str | None
1761 ) -> Artist | ItemMapping | None:
1762 """
1763 Return an existing album artist representing the same artist, if any.
1764
1765 Matches on MusicBrainz ID when available (names may differ when only one
1766 side was resolved against MusicBrainz), otherwise on exact name.
1767
1768 :param album: The track's album, if known.
1769 :param name: Resolved track-artist name.
1770 :param mbid: Resolved track-artist MusicBrainz ID, if any.
1771 """
1772 if not album:
1773 return None
1774 return next(
1775 (x for x in album.artists if (mbid and x.mbid == mbid) or x.name == name),
1776 None,
1777 )
1778
1779 async def _parse_artist(
1780 self,
1781 name: str,
1782 album_dir: str | None = None,
1783 sort_name: str | None = None,
1784 mbid: str | None = None,
1785 artist_path: str | None = None,
1786 ) -> Artist:
1787 """Parse full (album) Artist."""
1788 if not artist_path:
1789 # we need to hunt for the artist (metadata) path on disk
1790 # this can either be relative to the album path or at root level
1791 # check if we have an artist folder for this artist at root level
1792 safe_artist_name = create_safe_string(name, lowercase=False, replace_space=False)
1793 if await self.exists(name):
1794 artist_path = name
1795 elif await self.exists(safe_artist_name):
1796 artist_path = safe_artist_name
1797 elif album_dir and (foldermatch := get_artist_dir(name, album_dir=album_dir)):
1798 # try to find (album)artist folder based on album path
1799 artist_path = foldermatch
1800 else:
1801 # check if we have an existing item to retrieve the artist path
1802 async for item in self.mass.music.artists.iter_library_items(
1803 search=name, provider=self.instance_id
1804 ):
1805 if not compare_strings(name, item.name):
1806 continue
1807 for prov_mapping in item.provider_mappings:
1808 if prov_mapping.provider_instance != self.instance_id:
1809 continue
1810 if prov_mapping.url:
1811 artist_path = prov_mapping.url
1812 break
1813 if artist_path:
1814 break
1815
1816 # prefer (short lived) cache for a bit more speed
1817 if artist_path and (
1818 cache := await self.cache.get(
1819 key=artist_path,
1820 provider=self.instance_id,
1821 category=CACHE_CATEGORY_ARTIST_INFO,
1822 base_class=Artist,
1823 )
1824 ):
1825 return cache # type: ignore[no-any-return]
1826
1827 prov_artist_id = artist_path or name
1828 artist = Artist(
1829 item_id=prov_artist_id,
1830 provider=self.instance_id,
1831 name=name,
1832 sort_name=sort_name,
1833 provider_mappings={
1834 ProviderMapping(
1835 item_id=prov_artist_id,
1836 provider_domain=self.domain,
1837 provider_instance=self.instance_id,
1838 url=artist_path,
1839 in_library=True,
1840 )
1841 },
1842 )
1843 if mbid := clean_mbid(mbid, f"tags of artist {name}"):
1844 artist.mbid = mbid
1845 if not artist_path or not await self.exists(artist_path):
1846 return artist
1847
1848 # grab additional metadata within the Artist's folder
1849 nfo_file = os.path.join(artist_path, "artist.nfo")
1850 if await self.exists(nfo_file):
1851 # found NFO file with metadata
1852 # https://kodi.wiki/view/NFO_files/Artists
1853 try:
1854 data = (await self._read_file(nfo_file)).decode("utf-8")
1855 info = await asyncio.to_thread(xmltodict.parse, data)
1856 info = info["artist"]
1857 artist.name = info.get("title", info.get("name", name))
1858 if sort_name := info.get("sortname"):
1859 artist.sort_name = sort_name
1860 if mbid := clean_mbid(info.get("musicbrainzartistid"), nfo_file):
1861 artist.mbid = mbid
1862 if description := info.get("biography"):
1863 artist.metadata.description = description
1864 if genre := info.get("genre"):
1865 artist.metadata.genres = set(split_items(genre))
1866 except (ExpatError, KeyError) as err:
1867 self.logger.warning(
1868 "Failed to parse artist NFO file %s: %s",
1869 nfo_file,
1870 str(err),
1871 )
1872 # find local images
1873 if images := await self._get_local_images(artist_path, extra_thumb_names=("artist",)):
1874 artist.metadata.images = UniqueList(images)
1875
1876 await self.cache.set(
1877 key=artist_path,
1878 data=artist.to_dict(),
1879 provider=self.instance_id,
1880 category=CACHE_CATEGORY_ARTIST_INFO,
1881 expiration=120,
1882 )
1883
1884 return artist
1885
1886 async def _parse_audiobook(self, file_item: FileSystemItem, tags: AudioTags) -> Audiobook:
1887 """
1888 Parse Audiobook details from file tags.
1889
1890 Audiobooks can be single files with embedded chapters or multiple files per folder.
1891 Only the first file (by track number or alphabetically) is processed as the audiobook.
1892 """
1893 # Skip files that aren't the first chapter.
1894 # A file carrying its own embedded chapter markers is a standalone audiobook,
1895 # so it should never be treated as a chapter file of another book.
1896 track_tag = tags.tags.get("track")
1897 if track_tag:
1898 track_num = try_parse_int(str(track_tag).split("/")[0], None)
1899 if track_num and track_num > 1 and not tags.chapters:
1900 raise IsChapterFile
1901 elif not tags.chapters:
1902 # No track tag and no embedded chapters -
1903 # assume part of a multi-file audiobook, only process the first file alphabetically
1904 items = await self._scandir(file_item.relative_parent_path)
1905 # Sort by filename for alphabetical ordering
1906 items.sort(key=lambda x: x.filename.lower())
1907 for item in items:
1908 if item.is_dir or item.ext not in AUDIOBOOK_EXTENSIONS:
1909 continue
1910 if item.absolute_path != file_item.absolute_path:
1911 raise IsChapterFile
1912 break
1913
1914 # For multi-file audiobooks, album tag is the book name, title is the chapter name
1915 if tags.album:
1916 book_name = tags.album
1917 sort_name = tags.album_sort
1918 elif (title := tags.tags.get("title")) and tags.track is None:
1919 book_name = title
1920 sort_name = tags.title_sort
1921 else:
1922 # file(s) without tags, use foldername
1923 book_name = file_item.parent_name
1924 sort_name = None
1925
1926 # collect all chapters
1927 total_duration, chapters = await self._get_chapters_for_audiobook(file_item, tags)
1928
1929 audio_book = Audiobook(
1930 item_id=file_item.relative_path,
1931 provider=self.instance_id,
1932 name=book_name,
1933 sort_name=sort_name,
1934 version=tags.version,
1935 duration=total_duration or int(tags.duration or 0),
1936 provider_mappings={
1937 ProviderMapping(
1938 item_id=file_item.relative_path,
1939 provider_domain=self.domain,
1940 provider_instance=self.instance_id,
1941 audio_format=AudioFormat(
1942 content_type=ContentType.try_parse(file_item.ext or tags.format),
1943 sample_rate=tags.sample_rate,
1944 bit_depth=tags.bits_per_sample,
1945 channels=tags.channels,
1946 bit_rate=tags.bit_rate,
1947 ),
1948 details=file_item.checksum,
1949 in_library=True,
1950 )
1951 },
1952 )
1953 audio_book.metadata.chapters = chapters
1954
1955 # handle embedded cover image
1956 if tags.has_cover_image:
1957 # we do not actually embed the image in the metadata because that would consume too
1958 # much space and bandwidth. Instead we set the filename as value so the image can
1959 # be retrieved later in realtime.
1960 audio_book.metadata.add_image(
1961 MediaItemImage(
1962 type=ImageType.THUMB,
1963 path=self._versioned_image_path(file_item.relative_path, file_item.checksum),
1964 provider=self.instance_id,
1965 remotely_accessible=False,
1966 )
1967 )
1968
1969 # parse other info
1970 audio_book.authors.set(tags.writers or tags.album_artists or tags.artists)
1971 audio_book.metadata.genres = (
1972 set(tags.genres) if tags.genres else {DEFAULT_AUDIOBOOK_PODCAST_GENRE}
1973 )
1974 audio_book.metadata.copyright = tags.get("copyright")
1975 audio_book.metadata.lyrics = tags.lyrics
1976 audio_book.metadata.description = tags.get("comment")
1977 explicit_tag = tags.get("itunesadvisory")
1978 if explicit_tag is not None:
1979 audio_book.metadata.explicit = explicit_tag == "1"
1980 if recording_mbid := clean_mbid(tags.musicbrainz_recordingid, tags.filename):
1981 audio_book.mbid = recording_mbid
1982
1983 # try to fetch additional metadata from the folder
1984 if not audio_book.image or not audio_book.metadata.description:
1985 # try to get an image by traversing files in the same folder
1986 for _item in await self._scandir(file_item.relative_parent_path):
1987 if "." not in _item.relative_path or _item.is_dir:
1988 continue
1989 if _item.ext in IMAGE_EXTENSIONS and not audio_book.image:
1990 audio_book.metadata.add_image(
1991 MediaItemImage(
1992 type=ImageType.THUMB,
1993 path=self._versioned_image_path(_item.relative_path, _item.checksum),
1994 provider=self.instance_id,
1995 remotely_accessible=False,
1996 )
1997 )
1998 if _item.ext == "txt" and not audio_book.metadata.description:
1999 # try to parse a description from a text file
2000 try:
2001 raw = await self._read_file(_item.relative_path)
2002 audio_book.metadata.description = raw.decode("utf-8")
2003 except Exception as err:
2004 self.logger.warning(
2005 "Could not read description from file %s: %s",
2006 _item.relative_path,
2007 str(err),
2008 )
2009
2010 # handle (optional) loudness measurement tag(s)
2011 if tags.track_loudness is not None:
2012 self.mass.create_task(
2013 self.mass.streams.audio_analysis.set_track_loudness(
2014 audio_book.item_id,
2015 self.instance_id,
2016 tags.track_loudness,
2017 tags.track_album_loudness,
2018 media_type=MediaType.AUDIOBOOK,
2019 )
2020 )
2021 return audio_book
2022
2023 async def _parse_podcast_episode(
2024 self, file_item: FileSystemItem, tags: AudioTags
2025 ) -> PodcastEpisode:
2026 """Parse full PodcastEpisode details from file tags."""
2027 # ruff: noqa: PLR0915
2028 podcast_name = tags.album or file_item.parent_name
2029 podcast_path = file_item.relative_parent_path
2030 episode = PodcastEpisode(
2031 item_id=file_item.relative_path,
2032 provider=self.instance_id,
2033 name=tags.title,
2034 sort_name=tags.title_sort,
2035 provider_mappings={
2036 ProviderMapping(
2037 item_id=file_item.relative_path,
2038 provider_domain=self.domain,
2039 provider_instance=self.instance_id,
2040 audio_format=AudioFormat(
2041 content_type=ContentType.try_parse(file_item.ext or tags.format),
2042 sample_rate=tags.sample_rate,
2043 bit_depth=tags.bits_per_sample,
2044 channels=tags.channels,
2045 bit_rate=tags.bit_rate,
2046 ),
2047 details=file_item.checksum,
2048 in_library=True,
2049 )
2050 },
2051 position=tags.track or 0,
2052 duration=try_parse_int(tags.duration) or 0,
2053 podcast=Podcast(
2054 item_id=podcast_path,
2055 provider=self.instance_id,
2056 name=podcast_name,
2057 sort_name=tags.album_sort,
2058 publisher=tags.tags.get("publisher"),
2059 provider_mappings={
2060 ProviderMapping(
2061 item_id=podcast_path,
2062 provider_domain=self.domain,
2063 provider_instance=self.instance_id,
2064 in_library=True,
2065 )
2066 },
2067 ),
2068 )
2069 # handle embedded cover image
2070 if tags.has_cover_image:
2071 # we do not actually embed the image in the metadata because that would consume too
2072 # much space and bandwidth. Instead we set the filename as value so the image can
2073 # be retrieved later in realtime.
2074 episode.metadata.add_image(
2075 MediaItemImage(
2076 type=ImageType.THUMB,
2077 path=file_item.relative_path,
2078 provider=self.instance_id,
2079 remotely_accessible=False,
2080 )
2081 )
2082 # parse other info
2083 episode.metadata.genres = (
2084 set(tags.genres) if tags.genres else {DEFAULT_AUDIOBOOK_PODCAST_GENRE}
2085 )
2086 episode.metadata.copyright = tags.get("copyright")
2087 episode.metadata.lyrics = tags.lyrics
2088 episode.metadata.description = tags.get("comment")
2089 explicit_tag = tags.get("itunesadvisory")
2090 if explicit_tag is not None:
2091 episode.metadata.explicit = explicit_tag == "1"
2092
2093 # handle (optional) chapters
2094 if tags.chapters:
2095 episode.metadata.chapters = [
2096 MediaItemChapter(
2097 position=chapter.chapter_id,
2098 name=chapter.title or f"Chapter {chapter.chapter_id}",
2099 start=chapter.position_start,
2100 end=chapter.position_end,
2101 )
2102 for chapter in tags.chapters
2103 ]
2104
2105 # try to fetch additional Podcast metadata from the folder
2106 assert isinstance(episode.podcast, Podcast)
2107 if images := await self._get_local_images(file_item.relative_parent_path):
2108 episode.podcast.metadata.images = images
2109 if metadata := await self._get_podcast_metadata(file_item.relative_parent_path):
2110 if title := metadata.get("title"):
2111 episode.podcast.name = title
2112 if sort_name := metadata.get("sorttitle"):
2113 episode.podcast.sort_name = sort_name
2114 if description := metadata.get("description"):
2115 episode.podcast.metadata.description = description
2116 if genres := metadata.get("genres"):
2117 episode.podcast.metadata.genres = set(genres)
2118 if publisher := metadata.get("publisher"):
2119 episode.podcast.publisher = publisher
2120 if image := metadata.get("imageURL"):
2121 episode.podcast.metadata.add_image(
2122 MediaItemImage(
2123 type=ImageType.THUMB,
2124 path=image,
2125 provider=self.instance_id,
2126 remotely_accessible=True,
2127 )
2128 )
2129 # copy (embedded) image from episode (or vice versa)
2130 if not episode.podcast.image and episode.image:
2131 episode.podcast.metadata.add_image(episode.image)
2132 elif not episode.image and episode.podcast.image:
2133 episode.metadata.add_image(episode.podcast.image)
2134 # ensure podcast has a default genre if none set
2135 if not episode.podcast.metadata.genres:
2136 episode.podcast.metadata.genres = {DEFAULT_AUDIOBOOK_PODCAST_GENRE}
2137
2138 # handle (optional) loudness measurement tag(s)
2139 if tags.track_loudness is not None:
2140 self.mass.create_task(
2141 self.mass.streams.audio_analysis.set_track_loudness(
2142 episode.item_id,
2143 self.instance_id,
2144 tags.track_loudness,
2145 tags.track_album_loudness,
2146 media_type=MediaType.PODCAST_EPISODE,
2147 )
2148 )
2149 return episode
2150
2151 async def _parse_sound_effect(self, file_item: FileSystemItem, tags: AudioTags) -> SoundEffect:
2152 """Parse full sound effect details from file tags."""
2153 sound_effect = SoundEffect(
2154 item_id=file_item.relative_path,
2155 provider=self.instance_id,
2156 name=tags.title,
2157 sort_name=tags.title_sort,
2158 duration=int(tags.duration or 0),
2159 provider_mappings={
2160 ProviderMapping(
2161 item_id=file_item.relative_path,
2162 provider_domain=self.domain,
2163 provider_instance=self.instance_id,
2164 audio_format=AudioFormat(
2165 content_type=ContentType.try_parse(file_item.ext or tags.format),
2166 sample_rate=tags.sample_rate,
2167 bit_depth=tags.bits_per_sample,
2168 channels=tags.channels,
2169 bit_rate=tags.bit_rate,
2170 ),
2171 details=file_item.checksum,
2172 in_library=True,
2173 )
2174 },
2175 )
2176 sound_effect.metadata.description = tags.get("comment")
2177 # handle embedded cover image
2178 if tags.has_cover_image:
2179 # we do not actually embed the image in the metadata because that would consume too
2180 # much space and bandwidth. Instead we set the filename as value so the image can
2181 # be retrieved later in realtime.
2182 sound_effect.metadata.add_image(
2183 MediaItemImage(
2184 type=ImageType.THUMB,
2185 path=file_item.relative_path,
2186 provider=self.instance_id,
2187 remotely_accessible=False,
2188 )
2189 )
2190 return sound_effect
2191
2192 async def _get_or_parse_sound_effect(self, file_item: FileSystemItem) -> SoundEffect:
2193 """Return the (cached) SoundEffect for the given file, parsing tags when needed."""
2194 cache_key = f"sound_effect.{file_item.relative_path}"
2195 cached_data: SoundEffect | None = await self.cache.get(
2196 cache_key,
2197 provider=self.instance_id,
2198 checksum=file_item.checksum,
2199 category=CACHE_CATEGORY_SOUND_EFFECTS,
2200 base_class=SoundEffect,
2201 )
2202 if cached_data is not None:
2203 return cached_data
2204 tags = await async_parse_tags(file_item.absolute_path, file_item.file_size)
2205 sound_effect = await self._parse_sound_effect(file_item, tags)
2206 await self.cache.set(
2207 cache_key,
2208 sound_effect.to_dict(),
2209 expiration=3600 * 24 * 365, # File timestamp checksum handles invalidation
2210 provider=self.instance_id,
2211 checksum=file_item.checksum,
2212 category=CACHE_CATEGORY_SOUND_EFFECTS,
2213 )
2214 return sound_effect
2215
2216 async def _parse_album(
2217 self, track_path: str, track_tags: AudioTags, track_created_at: int | None = None
2218 ) -> Album:
2219 """
2220 Parse Album metadata from Track tags.
2221
2222 :param track_path: Path to the track file.
2223 :param track_tags: Audio tags from the track.
2224 :param track_created_at: Creation timestamp of the track file (Unix epoch).
2225 """
2226 assert track_tags.album
2227 # work out if we have an album and/or disc folder
2228 # track_dir is the folder level where the tracks are located
2229 # this may be a separate disc folder (Disc 1, Disc 2 etc) underneath the album folder
2230 # or this is an album folder with the disc attached
2231 track_dir = os.path.dirname(track_path)
2232 album_dir = get_album_dir(track_dir, track_tags.album)
2233
2234 if album_dir and (
2235 cache := await self.cache.get(
2236 key=album_dir,
2237 provider=self.instance_id,
2238 category=CACHE_CATEGORY_ALBUM_INFO,
2239 base_class=Album,
2240 )
2241 ):
2242 return cache # type: ignore[no-any-return]
2243
2244 # album artist(s)
2245 album_artists: UniqueList[Artist | ItemMapping] = UniqueList()
2246 if track_tags.album_artists:
2247 resolved_album_artists = await self._resolve_artists_with_mbids(
2248 track_tags.album_artists,
2249 track_tags.musicbrainz_albumartistids,
2250 track_tags.album_artist_sort_names,
2251 log_label="ALBUMARTIST tag",
2252 )
2253 for name, mbid, sort_name in resolved_album_artists:
2254 artist = await self._parse_artist(
2255 name, album_dir=album_dir, sort_name=sort_name, mbid=mbid
2256 )
2257 album_artists.append(artist)
2258 else:
2259 # album artist tag is missing, determine fallback
2260 fallback_action = self.config.get_value(CONF_ENTRY_MISSING_ALBUM_ARTIST.key)
2261 if fallback_action == "folder_name" and album_dir:
2262 possible_artist_folder = os.path.dirname(album_dir)
2263 self.logger.warning(
2264 "%s is missing ID3 tag [albumartist], using foldername %s as fallback",
2265 track_path,
2266 possible_artist_folder,
2267 )
2268 album_artist_str = Path(possible_artist_folder).name
2269 album_artists = UniqueList(
2270 [await self._parse_artist(name=album_artist_str, album_dir=album_dir)]
2271 )
2272 # fallback to track artists (if defined by user)
2273 elif fallback_action == "track_artist":
2274 self.logger.warning(
2275 "%s is missing ID3 tag [albumartist], using track artist(s) as fallback",
2276 track_path,
2277 )
2278 album_artists = UniqueList(
2279 [
2280 await self._parse_artist(name=track_artist_str, album_dir=album_dir)
2281 for track_artist_str in track_tags.artists
2282 ]
2283 )
2284 # all other: fallback to various artists
2285 else:
2286 self.logger.warning(
2287 "%s is missing ID3 tag [albumartist], using %s as fallback",
2288 track_path,
2289 VARIOUS_ARTISTS_NAME,
2290 )
2291 album_artists = UniqueList(
2292 [await self._parse_artist(name=VARIOUS_ARTISTS_NAME, mbid=VARIOUS_ARTISTS_MBID)]
2293 )
2294
2295 if album_dir: # noqa: SIM108
2296 # prefer the path as id
2297 item_id = album_dir
2298 else:
2299 # create fake item_id based on artist + album
2300 item_id = album_artists[0].name + os.sep + track_tags.album
2301
2302 name, version = parse_title_and_version(track_tags.album)
2303 album = Album(
2304 item_id=item_id,
2305 provider=self.instance_id,
2306 name=name,
2307 version=version,
2308 sort_name=track_tags.album_sort,
2309 artists=album_artists,
2310 provider_mappings={
2311 ProviderMapping(
2312 item_id=item_id,
2313 provider_domain=self.domain,
2314 provider_instance=self.instance_id,
2315 url=album_dir,
2316 in_library=True,
2317 )
2318 },
2319 date_added=(
2320 datetime.fromtimestamp(track_created_at, tz=UTC) if track_created_at else None
2321 ),
2322 )
2323 if track_tags.barcode:
2324 album.external_ids.add((ExternalID.BARCODE, track_tags.barcode))
2325
2326 if album_mbid := clean_mbid(track_tags.musicbrainz_albumid, track_tags.filename):
2327 album.mbid = album_mbid
2328 if releasegroup_mbid := clean_mbid(
2329 track_tags.musicbrainz_releasegroupid, track_tags.filename
2330 ):
2331 album.add_external_id(ExternalID.MB_RELEASEGROUP, releasegroup_mbid)
2332 if track_tags.year:
2333 album.year = track_tags.year
2334 album.album_type = track_tags.album_type
2335
2336 # hunt for additional metadata and images in the folder structure
2337 if not album_dir:
2338 return album
2339
2340 for folder_path in (track_dir, album_dir):
2341 if not folder_path or not await self.exists(folder_path):
2342 continue
2343 nfo_file = os.path.join(folder_path, "album.nfo")
2344 if await self.exists(nfo_file):
2345 # found NFO file with metadata
2346 # https://kodi.wiki/view/NFO_files/Artists
2347 try:
2348 data = (await self._read_file(nfo_file)).decode("utf-8")
2349 info = await asyncio.to_thread(xmltodict.parse, data)
2350 parse_album_nfo(album, info["album"], nfo_file)
2351 except (ExpatError, KeyError) as err:
2352 self.logger.warning(
2353 "Failed to parse album NFO file %s: %s",
2354 nfo_file,
2355 str(err),
2356 )
2357
2358 # find local images
2359 if images := await self._get_local_images(folder_path, extra_thumb_names=("album",)):
2360 if album.metadata.images is None:
2361 album.metadata.images = UniqueList(images)
2362 else:
2363 album.metadata.images += images
2364
2365 await self.cache.set(
2366 key=album_dir,
2367 data=album.to_dict(),
2368 provider=self.instance_id,
2369 category=CACHE_CATEGORY_ALBUM_INFO,
2370 expiration=120,
2371 )
2372 return album
2373
2374 async def _get_local_images(
2375 self, folder: str, extra_thumb_names: tuple[str, ...] | None = None
2376 ) -> UniqueList[MediaItemImage]:
2377 """Return local images found in a given folderpath."""
2378 if (
2379 cache := await self.cache.get(
2380 key=folder,
2381 provider=self.instance_id,
2382 category=CACHE_CATEGORY_FOLDER_IMAGES,
2383 base_class=MediaItemImage,
2384 )
2385 ) is not None:
2386 return UniqueList(cache)
2387 if extra_thumb_names is None:
2388 extra_thumb_names = ()
2389 images: UniqueList[MediaItemImage] = UniqueList()
2390 folder_files = await self._scandir(folder)
2391 for item in folder_files:
2392 if "." not in item.relative_path or item.is_dir or not item.ext:
2393 continue
2394 if item.ext.lower() not in IMAGE_EXTENSIONS:
2395 continue
2396 # try match on filename = one of our imagetypes
2397 if item.name.lower() in ImageType:
2398 images.append(
2399 MediaItemImage(
2400 type=ImageType(item.name),
2401 path=item.relative_path,
2402 provider=self.instance_id,
2403 remotely_accessible=False,
2404 )
2405 )
2406
2407 # try alternative names for thumbs
2408 extra_thumb_names = ("folder", "cover", *extra_thumb_names)
2409 for item in folder_files:
2410 if "." not in item.relative_path or item.is_dir or not item.ext:
2411 continue
2412 if item.ext.lower() not in IMAGE_EXTENSIONS:
2413 continue
2414 if item.name.lower() not in extra_thumb_names:
2415 continue
2416 images.append(
2417 MediaItemImage(
2418 type=ImageType.THUMB,
2419 path=item.relative_path,
2420 provider=self.instance_id,
2421 remotely_accessible=False,
2422 )
2423 )
2424
2425 await self.cache.set(
2426 key=folder,
2427 data=[img.to_dict() for img in images],
2428 provider=self.instance_id,
2429 category=CACHE_CATEGORY_FOLDER_IMAGES,
2430 expiration=120,
2431 )
2432 return images
2433
2434 async def _get_stream_details_for_track(self, item_id: str) -> StreamDetails:
2435 """Return the streamdetails for a track/song."""
2436 if parse_cue_track_id(item_id) is not None:
2437 return await self._cue.get_stream_details(item_id)
2438
2439 library_item = await self.mass.music.tracks.get_library_item_by_prov_id(
2440 item_id, self.instance_id
2441 )
2442 if library_item is None:
2443 # this could be a file that has just been added, try parsing it
2444 file_item = await self.resolve(item_id)
2445 tags = await async_parse_tags(file_item.absolute_path, file_item.file_size)
2446 if not (library_item := await self._parse_track(file_item, tags)):
2447 msg = f"Item not found: {item_id}"
2448 raise MediaNotFoundError(msg)
2449
2450 prov_mapping = next(x for x in library_item.provider_mappings if x.item_id == item_id)
2451 file_item = await self.resolve(item_id)
2452
2453 return StreamDetails(
2454 provider=self.instance_id,
2455 item_id=item_id,
2456 audio_format=prov_mapping.audio_format,
2457 media_type=MediaType.TRACK,
2458 stream_type=StreamType.LOCAL_FILE,
2459 duration=library_item.duration,
2460 size=file_item.file_size,
2461 data=file_item,
2462 path=file_item.absolute_path,
2463 can_seek=True,
2464 allow_seek=True,
2465 )
2466
2467 async def _get_stream_details_for_podcast_episode(self, item_id: str) -> StreamDetails:
2468 """Return the streamdetails for a podcast episode."""
2469 # podcasts episodes are never stored in the library so we need to parse the file
2470 file_item = await self.resolve(item_id)
2471 tags = await async_parse_tags(file_item.absolute_path, file_item.file_size)
2472 return StreamDetails(
2473 provider=self.instance_id,
2474 item_id=item_id,
2475 audio_format=AudioFormat(
2476 content_type=ContentType.try_parse(file_item.ext or tags.format),
2477 sample_rate=tags.sample_rate,
2478 bit_depth=tags.bits_per_sample,
2479 channels=tags.channels,
2480 bit_rate=tags.bit_rate,
2481 ),
2482 media_type=MediaType.PODCAST_EPISODE,
2483 stream_type=StreamType.LOCAL_FILE,
2484 duration=try_parse_int(tags.duration or 0),
2485 size=file_item.file_size,
2486 data=file_item,
2487 path=file_item.absolute_path,
2488 allow_seek=True,
2489 can_seek=True,
2490 )
2491
2492 async def _get_stream_details_for_sound_effect(self, item_id: str) -> StreamDetails:
2493 """Return the streamdetails for a sound effect."""
2494 # sound effects are never stored in the library so we parse the file,
2495 # served from cache unless the file changed on disk
2496 file_item = await self.resolve(item_id)
2497 sound_effect = await self._get_or_parse_sound_effect(file_item)
2498 prov_mapping = next(x for x in sound_effect.provider_mappings if x.item_id == item_id)
2499 return StreamDetails(
2500 provider=self.instance_id,
2501 item_id=item_id,
2502 audio_format=prov_mapping.audio_format,
2503 media_type=MediaType.SOUND_EFFECT,
2504 stream_type=StreamType.LOCAL_FILE,
2505 duration=sound_effect.duration,
2506 size=file_item.file_size,
2507 data=file_item,
2508 path=file_item.absolute_path,
2509 allow_seek=True,
2510 can_seek=True,
2511 )
2512
2513 async def _get_stream_details_for_audiobook(self, item_id: str) -> StreamDetails:
2514 """Return the streamdetails for an audiobook."""
2515 library_item = await self.mass.music.audiobooks.get_library_item_by_prov_id(
2516 item_id, self.instance_id
2517 )
2518 if library_item is None:
2519 # this could be a file that has just been added, try parsing it
2520 file_item = await self.resolve(item_id)
2521 tags = await async_parse_tags(file_item.absolute_path, file_item.file_size)
2522 if not (library_item := await self._parse_audiobook(file_item, tags)):
2523 msg = f"Item not found: {item_id}"
2524 raise MediaNotFoundError(msg)
2525
2526 prov_mapping = next(x for x in library_item.provider_mappings if x.item_id == item_id)
2527 file_item = await self.resolve(item_id)
2528 duration = library_item.duration
2529 file_based_chapters: list[tuple[str, float]] | None = await self.cache.get(
2530 key=file_item.relative_path,
2531 provider=self.instance_id,
2532 category=CACHE_CATEGORY_AUDIOBOOK_CHAPTERS,
2533 )
2534 if file_based_chapters is None:
2535 # no cache available for this audiobook, we need to parse the chapters
2536 tags = await async_parse_tags(file_item.absolute_path, file_item.file_size)
2537 await self._parse_audiobook(file_item, tags)
2538 file_based_chapters = await self.cache.get(
2539 key=file_item.relative_path,
2540 provider=self.instance_id,
2541 category=CACHE_CATEGORY_AUDIOBOOK_CHAPTERS,
2542 )
2543
2544 if file_based_chapters:
2545 # this is a multi-file audiobook
2546 return StreamDetails(
2547 provider=self.instance_id,
2548 item_id=item_id,
2549 audio_format=prov_mapping.audio_format,
2550 media_type=MediaType.AUDIOBOOK,
2551 stream_type=StreamType.LOCAL_FILE,
2552 duration=duration,
2553 path=[
2554 MultiPartPath(
2555 path=self._get_chapter_path(chapter_path),
2556 duration=chapter_duration,
2557 )
2558 for chapter_path, chapter_duration in file_based_chapters
2559 ],
2560 allow_seek=True,
2561 )
2562
2563 # regular single-file streaming, simply let ffmpeg deal with the file directly
2564 return StreamDetails(
2565 provider=self.instance_id,
2566 item_id=item_id,
2567 audio_format=prov_mapping.audio_format,
2568 media_type=MediaType.AUDIOBOOK,
2569 stream_type=StreamType.LOCAL_FILE,
2570 duration=library_item.duration,
2571 size=file_item.file_size,
2572 data=file_item,
2573 path=file_item.absolute_path,
2574 allow_seek=True,
2575 can_seek=True,
2576 )
2577
2578 def _get_chapter_path(self, relative_path: str) -> str:
2579 """Return absolute path for a chapter file. Override for network storage."""
2580 return self.get_absolute_path(relative_path)
2581
2582 async def _get_chapters_for_audiobook(
2583 self, audiobook_file_item: FileSystemItem, tags: AudioTags
2584 ) -> tuple[int, list[MediaItemChapter]]:
2585 """
2586 Return chapters for an audiobook.
2587
2588 Chapter sources in order of preference:
2589 1. Multiple files with track tags - sorted by track number
2590 2. Single file with embedded chapters - use embedded chapter markers
2591 3. Multiple files without track tags - sorted alphabetically (fallback)
2592 """
2593 chapters: list[MediaItemChapter] = []
2594 all_chapter_files: list[tuple[str, float]] = []
2595 total_duration = 0.0
2596
2597 # Scan folder for chapter files, separating tagged from untagged
2598 chapter_file_items: list[tuple[FileSystemItem, AudioTags]] = []
2599 untagged_file_items: list[tuple[FileSystemItem, AudioTags]] = []
2600
2601 items = await self._scandir(audiobook_file_item.relative_parent_path)
2602 # Sort by filename for consistent alphabetical ordering
2603 items.sort(key=lambda x: x.filename.lower())
2604
2605 for item in items:
2606 if "." not in item.relative_path or item.is_dir:
2607 continue
2608 if item.ext not in AUDIOBOOK_EXTENSIONS:
2609 continue
2610 item_tags = await async_parse_tags(item.absolute_path, item.file_size)
2611 if not (tags.album == item_tags.album or (item_tags.tags.get("title") is None)):
2612 continue
2613 if item_tags.tags.get("track") is None:
2614 untagged_file_items.append((item, item_tags))
2615 else:
2616 chapter_file_items.append((item, item_tags))
2617
2618 # Determine chapter source
2619 use_embedded = False
2620 use_alphabetical = False
2621
2622 if len(chapter_file_items) > 1:
2623 chapter_file_items.sort(key=lambda x: (x[1].disc or 0, x[1].track or 0))
2624 elif len(chapter_file_items) <= 1 and tags.chapters:
2625 use_embedded = True
2626 elif len(untagged_file_items) > 1:
2627 use_alphabetical = True
2628 chapter_file_items = untagged_file_items
2629 self.logger.info(
2630 "Audiobook files have no track tags, using alphabetical order: %s",
2631 tags.album,
2632 )
2633
2634 if use_embedded:
2635 chapters = [
2636 MediaItemChapter(
2637 position=chapter.chapter_id,
2638 name=chapter.title or f"Chapter {chapter.chapter_id}",
2639 start=chapter.position_start,
2640 end=chapter.position_end,
2641 )
2642 for chapter in tags.chapters
2643 ]
2644 total_duration = try_parse_int(tags.duration) or 0
2645 self.logger.log(
2646 VERBOSE_LOG_LEVEL,
2647 "Audiobook '%s': %d embedded chapters, duration=%d",
2648 tags.album,
2649 len(chapters),
2650 int(total_duration),
2651 )
2652 else:
2653 for position, (chapter_item, chapter_tags) in enumerate(chapter_file_items, start=1):
2654 if chapter_tags.duration is None:
2655 self.logger.warning(
2656 "Chapter file has no duration, skipping: %s",
2657 chapter_item.relative_path,
2658 )
2659 continue
2660 self.logger.debug("Chapter filename: %s", chapter_item.relative_path)
2661 chapters.append(
2662 MediaItemChapter(
2663 position=position,
2664 name=chapter_tags.title,
2665 start=total_duration,
2666 end=total_duration + chapter_tags.duration,
2667 )
2668 )
2669 all_chapter_files.append(
2670 (
2671 chapter_item.relative_path,
2672 chapter_tags.duration,
2673 )
2674 )
2675 total_duration += chapter_tags.duration
2676 sort_method = "alphabetical" if use_alphabetical else "track"
2677 self.logger.log(
2678 VERBOSE_LOG_LEVEL,
2679 "Audiobook '%s': %d files (%s order), duration=%d",
2680 tags.album,
2681 len(chapters),
2682 sort_method,
2683 int(total_duration),
2684 )
2685 # Cache chapter files for streaming
2686 await self.cache.set(
2687 key=audiobook_file_item.relative_path,
2688 data=all_chapter_files,
2689 provider=self.instance_id,
2690 category=CACHE_CATEGORY_AUDIOBOOK_CHAPTERS,
2691 )
2692 return int(total_duration), chapters
2693
2694 async def _get_podcast_metadata(self, podcast_folder: str) -> dict[str, Any]:
2695 """Return metadata for a podcast."""
2696 if (
2697 cache := await self.cache.get(
2698 key=podcast_folder,
2699 provider=self.instance_id,
2700 category=CACHE_CATEGORY_PODCAST_METADATA,
2701 )
2702 ) is not None:
2703 return cast("dict[str, Any]", cache)
2704 data: dict[str, Any] = {}
2705 metadata_file = os.path.join(podcast_folder, "metadata.json")
2706 if await self.exists(metadata_file):
2707 # found json file with metadata
2708 raw = await self._read_file(metadata_file)
2709 data.update(json_loads(raw.decode("utf-8")))
2710 await self.cache.set(
2711 key=podcast_folder,
2712 data=data,
2713 provider=self.instance_id,
2714 category=CACHE_CATEGORY_PODCAST_METADATA,
2715 )
2716 return data
2717
2718 async def _scandir(self, path: str) -> list[FileSystemItem]:
2719 """List directory contents in natural sort order."""
2720 # raw scandir order depends on the underlying filesystem (e.g. hash order
2721 # on ext4) so sort to make browse and folder playback order deterministic
2722 abs_path = self.get_absolute_path(path)
2723 return await asyncio.to_thread(sorted_scandir, self.base_path, abs_path, sort=True)
2724
2725 async def _read_file(self, path: str) -> bytes:
2726 """Read file contents. Override for network storage."""
2727 async with aiofiles.open(self.get_absolute_path(path), mode="rb") as f:
2728 return cast("bytes", await f.read())
2729