/
/
1"""Filesystem musicprovider support for MusicAssistant."""
2
3from __future__ import annotations
4
5import asyncio
6import contextlib
7import logging
8import os
9import os.path
10import posixpath
11import urllib.parse
12from collections.abc import AsyncGenerator, Sequence
13from datetime import UTC, datetime
14from pathlib import Path
15from typing import TYPE_CHECKING, Any, ClassVar, cast
16from xml.parsers.expat import ExpatError
17
18import aiofiles
19import shortuuid
20import xmltodict
21from aiofiles.os import wrap
22from music_assistant_models.enums import (
23 ContentType,
24 EventType,
25 ExternalID,
26 ImageType,
27 MediaType,
28 ProviderFeature,
29 StreamType,
30)
31from music_assistant_models.errors import (
32 InvalidDataError,
33 MediaNotFoundError,
34 MusicAssistantError,
35 SetupFailedError,
36)
37from music_assistant_models.helpers import create_safe_string
38from music_assistant_models.media_items import (
39 Album,
40 Artist,
41 Audiobook,
42 AudioFormat,
43 BrowseFolder,
44 ItemMapping,
45 MediaItemChapter,
46 MediaItemImage,
47 MediaItemType,
48 Playlist,
49 Podcast,
50 PodcastEpisode,
51 ProviderMapping,
52 SearchResults,
53 SoundEffect,
54 Track,
55 UniqueList,
56 is_track,
57)
58from music_assistant_models.streamdetails import MultiPartPath, StreamDetails
59
60from music_assistant.constants import (
61 CONF_PATH,
62 DB_TABLE_ALBUM_ARTISTS,
63 DB_TABLE_ALBUM_TRACKS,
64 DB_TABLE_ALBUMS,
65 DB_TABLE_ARTISTS,
66 DB_TABLE_PROVIDER_MAPPINGS,
67 DB_TABLE_TRACK_ARTISTS,
68 VARIOUS_ARTISTS_MBID,
69 VARIOUS_ARTISTS_NAME,
70 VERBOSE_LOG_LEVEL,
71)
72from music_assistant.controllers.tasks.context import (
73 report_current_task_failure,
74 update_current_task_progress_from_index,
75 update_current_task_progress_text,
76)
77from music_assistant.helpers import lyrics
78from music_assistant.helpers.compare import compare_strings
79from music_assistant.helpers.json import SerializableType, json_loads
80from music_assistant.helpers.playlists import parse_m3u, parse_pls
81from music_assistant.helpers.tags import AudioTags, async_parse_tags, clean_mbid, split_items
82from music_assistant.helpers.uri import create_uri
83from music_assistant.helpers.util import (
84 TaskManager,
85 detect_charset,
86 parse_title_and_version,
87 try_parse_int,
88)
89from music_assistant.models.music_provider import MusicProvider
90
91from .constants import (
92 AUDIOBOOK_EXTENSIONS,
93 AVAILABILITY_PROBE_INTERVAL,
94 CACHE_CATEGORY_ALBUM_INFO,
95 CACHE_CATEGORY_ARTIST_INFO,
96 CACHE_CATEGORY_AUDIOBOOK_CHAPTERS,
97 CACHE_CATEGORY_FOLDER_IMAGES,
98 CACHE_CATEGORY_PODCAST_EPISODES,
99 CACHE_CATEGORY_PODCAST_METADATA,
100 CACHE_CATEGORY_SOUND_EFFECTS,
101 CONF_CONTENT_TYPE,
102 CONF_ENTRY_CONTENT_TYPE,
103 CONF_ENTRY_IGNORE_ALBUM_PLAYLISTS,
104 CONF_ENTRY_LIBRARY_SYNC_AUDIOBOOKS,
105 CONF_ENTRY_LIBRARY_SYNC_PLAYLISTS,
106 CONF_ENTRY_LIBRARY_SYNC_PODCASTS,
107 CONF_ENTRY_LIBRARY_SYNC_TRACKS,
108 CONF_ENTRY_MISSING_ALBUM_ARTIST,
109 CONF_ENTRY_PROPAGATE_GENRES,
110 CUE_EXTENSIONS,
111 DEFAULT_AUDIOBOOK_PODCAST_GENRE,
112 IMAGE_EXTENSIONS,
113 PARTIAL_LISTING_CACHE_EXPIRATION,
114 PLAYLIST_EXTENSIONS,
115 PODCAST_EPISODE_EXTENSIONS,
116 SOUND_EFFECT_EXTENSIONS,
117 SUPPORTED_EXTENSIONS,
118 TRACK_EXTENSIONS,
119 IsChapterFile,
120 content_type_config_entry,
121)
122from .cue import (
123 CueSheetHandler,
124 cue_metadata_checksum,
125 cue_referenced_audio_stem,
126 make_cue_track_id,
127 parse_cue_track_id,
128)
129from .helpers import (
130 FileSystemItem,
131 ScanErrors,
132 get_absolute_path,
133 get_album_dir,
134 get_artist_dir,
135 get_folder_signature,
136 get_relative_path,
137 recursive_iter,
138 sorted_scandir,
139)
140from .parsers import parse_album_nfo
141
142if TYPE_CHECKING:
143 from music_assistant_models.config_entries import ConfigEntry, ProviderConfig
144 from music_assistant_models.provider import ProviderManifest
145
146 from music_assistant.mass import MusicAssistant
147 from music_assistant.models import ProviderInstanceType
148 from music_assistant.providers.musicbrainz import MusicbrainzProvider
149
150
151isdir = wrap(os.path.isdir)
152isfile = wrap(os.path.isfile)
153ismount = wrap(os.path.ismount)
154exists = wrap(os.path.exists)
155makedirs = wrap(os.makedirs)
156
157SUPPORTED_FEATURES = {
158 ProviderFeature.BROWSE,
159 ProviderFeature.SEARCH,
160}
161
162
163async def setup(
164 mass: MusicAssistant, manifest: ProviderManifest, config: ProviderConfig
165) -> ProviderInstanceType:
166 """Initialize provider(instance) with given configuration."""
167 return LocalFileSystemProvider(mass, manifest, config)
168
169
170class LocalFileSystemProvider(MusicProvider):
171 """
172 Implementation of a musicprovider for (local) files.
173
174 Reads ID3 tags from file and falls back to parsing filename.
175 Optionally reads metadata from nfo files and images in folder structure <artist>/<album>.
176 Supports m3u files for playlists.
177 """
178
179 # parallel workers per sync; subclasses lower this for slower transports
180 _SYNC_CONCURRENCY: ClassVar[int] = 16
181 _sync_tracks: bool = True
182 _sync_playlists: bool = True
183
184 def __init__(
185 self,
186 mass: MusicAssistant,
187 manifest: ProviderManifest,
188 config: ProviderConfig,
189 base_path: str | None = None,
190 ) -> None:
191 """Initialize MusicProvider."""
192 super().__init__(mass, manifest, config, SUPPORTED_FEATURES)
193 # subclasses (NFS/SMB/...) mount elsewhere and pass their own base_path;
194 # the plain local provider reads its scan directory from the setup data
195 self.base_path: str = (
196 base_path if base_path is not None else cast("str", self.get_setup_value(CONF_PATH))
197 )
198 self.write_access: bool = False
199 self.sync_running: bool = False
200 self.media_content_type = cast(
201 "str", self.get_setup_value(CONF_CONTENT_TYPE, CONF_ENTRY_CONTENT_TYPE.default_value)
202 )
203 self._cue = CueSheetHandler(self)
204
205 async def get_config_entries(self) -> tuple[ConfigEntry, ...]:
206 """Return Config entries to configure this provider."""
207 # content type and path are collected by the setup flow; surface the (immutable)
208 # content type read-only so the sync options' depends_on chains still resolve
209 content_type = str(
210 self.get_setup_value(CONF_CONTENT_TYPE, CONF_ENTRY_CONTENT_TYPE.default_value)
211 )
212 return (
213 content_type_config_entry(content_type),
214 CONF_ENTRY_MISSING_ALBUM_ARTIST,
215 CONF_ENTRY_IGNORE_ALBUM_PLAYLISTS,
216 CONF_ENTRY_LIBRARY_SYNC_TRACKS,
217 CONF_ENTRY_LIBRARY_SYNC_PLAYLISTS,
218 CONF_ENTRY_LIBRARY_SYNC_PODCASTS,
219 CONF_ENTRY_LIBRARY_SYNC_AUDIOBOOKS,
220 CONF_ENTRY_PROPAGATE_GENRES,
221 )
222
223 @property
224 def supported_features(self) -> set[ProviderFeature]:
225 """Return the features supported by this Provider."""
226 base_features = {*SUPPORTED_FEATURES}
227 if self.media_content_type == "audiobooks":
228 return {ProviderFeature.LIBRARY_AUDIOBOOKS, *base_features}
229 if self.media_content_type == "podcasts":
230 return {ProviderFeature.LIBRARY_PODCASTS, *base_features}
231 if self.media_content_type == "sound_effects":
232 # sound effects are live-fetched content, never synced into the library
233 return {ProviderFeature.SOUND_EFFECTS, *base_features}
234 music_features = {
235 ProviderFeature.LIBRARY_ALBUMS,
236 ProviderFeature.LIBRARY_ARTISTS,
237 ProviderFeature.LIBRARY_TRACKS,
238 ProviderFeature.LIBRARY_PLAYLISTS,
239 *base_features,
240 }
241 if self.write_access:
242 music_features.add(ProviderFeature.PLAYLIST_TRACKS_EDIT)
243 music_features.add(ProviderFeature.PLAYLIST_CREATE)
244 return music_features
245
246 @property
247 def is_streaming_provider(self) -> bool:
248 """Return True if the provider is a streaming provider."""
249 return False
250
251 @property
252 def instance_name_postfix(self) -> str | None:
253 """Return a (default) instance name postfix for this provider instance."""
254 return Path(self.base_path).name
255
256 async def handle_async_init(self) -> None:
257 """Handle async initialization of the provider."""
258 if not await isdir(self.base_path):
259 msg = f"Music Directory {self.base_path} does not exist"
260 raise SetupFailedError(
261 msg,
262 translation_key="music_directory_not_found",
263 translation_owner=self.translation_owner,
264 translation_args=[self.base_path],
265 )
266 await self.check_write_access()
267
268 async def unload(self, is_removed: bool = False) -> None:
269 """Handle unload/close of the provider."""
270 self._cancel_availability_probe()
271 # a check that already started runs as a task under the same id, and it would
272 # otherwise keep talking to storage this unload is in the middle of tearing down
273 self.mass.cancel_task(self._availability_probe_id)
274
275 async def get_diagnostics(self) -> dict[str, SerializableType]:
276 """Return diagnostics info for this provider to include in diagnostics reports."""
277 return {
278 "sync_running": self.sync_running,
279 "write_access": self.write_access,
280 "content_type": self.media_content_type,
281 }
282
283 async def search(
284 self,
285 search_query: str,
286 media_types: list[MediaType] | None,
287 limit: int = 5,
288 ) -> SearchResults:
289 """Perform search on this file based musicprovider."""
290 result = SearchResults()
291 # searching the filesystem is slow and unreliable,
292 # so instead we just query the db...
293 if media_types is None or MediaType.TRACK in media_types:
294 result.tracks = await self.mass.music.tracks.get_library_items_by_query(
295 search=search_query, provider_filter=[self.instance_id], limit=limit
296 )
297
298 if media_types is None or MediaType.ALBUM in media_types:
299 result.albums = await self.mass.music.albums.get_library_items_by_query(
300 search=search_query,
301 provider_filter=[self.instance_id],
302 limit=limit,
303 )
304
305 if media_types is None or MediaType.ARTIST in media_types:
306 result.artists = await self.mass.music.artists.get_library_items_by_query(
307 search=search_query,
308 provider_filter=[self.instance_id],
309 limit=limit,
310 )
311 if media_types is None or MediaType.PLAYLIST in media_types:
312 result.playlists = await self.mass.music.playlists.get_library_items_by_query(
313 search=search_query,
314 provider_filter=[self.instance_id],
315 limit=limit,
316 )
317 if media_types is None or MediaType.AUDIOBOOK in media_types:
318 result.audiobooks = await self.mass.music.audiobooks.get_library_items_by_query(
319 search=search_query,
320 provider_filter=[self.instance_id],
321 limit=limit,
322 )
323 if media_types is None or MediaType.PODCAST in media_types:
324 result.podcasts = await self.mass.music.podcasts.get_library_items_by_query(
325 search=search_query,
326 provider_filter=[self.instance_id],
327 limit=limit,
328 )
329 return result
330
331 async def browse(self, path: str) -> Sequence[MediaItemType | ItemMapping | BrowseFolder]:
332 """
333 Browse this provider's items.
334
335 :param path: The path to browse, (e.g. provid://artists).
336 """
337 # for audiobooks and podcasts we just return all library items
338 if self.media_content_type == "podcasts":
339 return await self.mass.music.podcasts.library_items(
340 provider=self.instance_id, summary=False
341 )
342 if self.media_content_type == "audiobooks":
343 return await self.mass.music.audiobooks.library_items(
344 provider=self.instance_id, summary=False
345 )
346 items: list[MediaItemType | ItemMapping | BrowseFolder] = []
347 item_path = path.split("://", 1)[1]
348 if not item_path:
349 item_path = ""
350 scanned = await self._scandir(item_path)
351 # expand CUE sheets into per-track entries and hide the companion audio;
352 # synthetic ids match those minted during sync so get_track resolves them
353 cue_stems: set[str] = set()
354 if self.media_content_type == "music":
355 for item in scanned:
356 if item.ext not in CUE_EXTENSIONS:
357 continue
358 cue_stems.add(item.absolute_path.rsplit(".", 1)[0])
359 try:
360 cue_sheet = await self._cue.load_cue_sheet(item)
361 except InvalidDataError as err:
362 self.logger.warning("Unable to parse CUE sheet %s: %s", item.relative_path, err)
363 continue
364 # also hide the audio file named in the CUE (may differ from its stem)
365 if companion_stem := cue_referenced_audio_stem(item, cue_sheet):
366 cue_stems.add(companion_stem)
367 for cue_track in cue_sheet.tracks:
368 items.append(
369 ItemMapping(
370 media_type=MediaType.TRACK,
371 item_id=make_cue_track_id(item.relative_path, cue_track.number),
372 provider=self.instance_id,
373 name=cue_track.title or f"Track {cue_track.number}",
374 )
375 )
376 for item in scanned:
377 if not item.is_dir and ("." not in item.filename or not item.ext):
378 # skip system files and files without extension
379 continue
380
381 if item.is_dir:
382 items.append(
383 BrowseFolder(
384 item_id=item.relative_path,
385 provider=self.instance_id,
386 path=f"{self.instance_id}://{item.relative_path}",
387 name=item.filename,
388 # mark folder as playable, assuming it contains tracks underneath
389 is_playable=True,
390 )
391 )
392 elif item.ext in TRACK_EXTENSIONS:
393 if item.absolute_path.rsplit(".", 1)[0] in cue_stems:
394 continue
395 items.append(
396 ItemMapping(
397 media_type=(
398 MediaType.SOUND_EFFECT
399 if self.media_content_type == "sound_effects"
400 else MediaType.TRACK
401 ),
402 item_id=item.relative_path,
403 provider=self.instance_id,
404 name=item.filename,
405 )
406 )
407 elif item.ext in PLAYLIST_EXTENSIONS and self.media_content_type == "music":
408 items.append(
409 ItemMapping(
410 media_type=MediaType.PLAYLIST,
411 item_id=item.relative_path,
412 provider=self.instance_id,
413 name=item.filename,
414 )
415 )
416 if self.media_content_type == "music":
417 track_indexes = [
418 index
419 for index, item in enumerate(items)
420 if isinstance(item, ItemMapping) and item.media_type == MediaType.TRACK
421 ]
422 library_tracks = await asyncio.gather(
423 *(
424 self.mass.music.tracks.get_library_item_by_prov_id(
425 items[index].item_id, self.instance_id
426 )
427 for index in track_indexes
428 )
429 )
430 for index, library_track in zip(track_indexes, library_tracks, strict=True):
431 if library_track:
432 items[index] = library_track
433 return items
434
435 async def sync_library(self, media_type: MediaType) -> None:
436 """Run library sync for this provider."""
437 if media_type in (MediaType.ARTIST, MediaType.ALBUM):
438 # artists and albums are synced as part of track sync
439 return
440 if self.media_content_type == "sound_effects":
441 # sound effects are live-fetched content, never synced into the library
442 return
443 # check if any sync options are enabled for this content type
444 # the filesystem provider processes all file types in one scan,
445 # so we can return early if nothing needs syncing
446 if self.media_content_type == "music":
447 self._sync_tracks = bool(self.config.get_value(CONF_ENTRY_LIBRARY_SYNC_TRACKS.key))
448 self._sync_playlists = bool(
449 self.config.get_value(CONF_ENTRY_LIBRARY_SYNC_PLAYLISTS.key)
450 )
451 if not self._sync_tracks and not self._sync_playlists:
452 return
453 elif self.media_content_type == "audiobooks":
454 if not self.config.get_value(CONF_ENTRY_LIBRARY_SYNC_AUDIOBOOKS.key):
455 return
456 elif self.media_content_type == "podcasts":
457 if not self.config.get_value(CONF_ENTRY_LIBRARY_SYNC_PODCASTS.key):
458 return
459 assert self.mass.music.database
460 if self.sync_running:
461 self.logger.warning("Library sync already running for %s", self.name)
462 return
463 file_checksums: dict[str, str] = {}
464 # NOTE: we always run a scan of the entire library, as we need to detect changes
465 # we ignore any given mediatype(s) and just scan all supported files
466 query = (
467 f"SELECT provider_item_id, details FROM {DB_TABLE_PROVIDER_MAPPINGS} "
468 f"WHERE provider_instance = '{self.instance_id}' "
469 f"AND media_type in ('track', 'playlist', 'audiobook', 'podcast_episode')"
470 )
471 for db_row in await self.mass.music.database.get_rows_from_query(query, limit=0):
472 file_checksums[db_row["provider_item_id"]] = str(db_row["details"])
473 # provider_mappings stores synthetic per-track ids for CUE sheets, not the
474 # CUE path, so collect every track checksum per path for the scan classifier
475 cue_file_checksums: dict[str, set[str]] = {}
476 for prov_item_id, checksum in file_checksums.items():
477 parsed = parse_cue_track_id(prov_item_id)
478 if parsed is not None:
479 cue_file_checksums.setdefault(parsed[0], set()).add(checksum)
480 # find all supported files in the base directory and all subfolders
481 # we work bottom up, as-in we derive all info from the tracks
482 cur_filenames: set[str] = set()
483 prev_filenames = set(file_checksums.keys())
484
485 items_to_process: list[tuple[FileSystemItem, str | None]] = []
486 unchanged_cue_items: list[FileSystemItem] = []
487 # absolute paths of every CUE sheet in this scan with the ".cue" stripped,
488 # used for O(1) companion-CUE lookups per audio file
489 cue_stems: set[str] = set()
490 # collects the errors raised while walking the tree; any error means the
491 # scan is incomplete, a fatal one means the provider is unreachable
492 scan_errors = ScanErrors()
493
494 self.sync_running = True
495 try:
496 await self._enumerate_files_for_sync(
497 file_checksums=file_checksums,
498 cue_file_checksums=cue_file_checksums,
499 cur_filenames=cur_filenames,
500 items_to_process=items_to_process,
501 unchanged_cue_items=unchanged_cue_items,
502 cue_stems=cue_stems,
503 scan_errors=scan_errors,
504 )
505 if scan_errors.fatal:
506 # the storage is gone, so reading the files collected before it went
507 # away would only add a timeout each
508 self.logger.error("Aborting sync for %s: %s", self.name, scan_errors.fatal)
509 report_current_task_failure("Sync aborted: filesystem unavailable during scan")
510 self._set_available(False)
511 return
512 # a CUE may name an audio file other than its own; hide that companion too
513 if self.media_content_type == "music":
514 for cue_item in (
515 *unchanged_cue_items,
516 *(item for item, _ in items_to_process if item.ext in CUE_EXTENSIONS),
517 ):
518 try:
519 cue_sheet = await self._cue.load_cue_sheet(cue_item)
520 except InvalidDataError:
521 continue
522 if companion_stem := cue_referenced_audio_stem(cue_item, cue_sheet):
523 cue_stems.add(companion_stem)
524 # drop CUE companion audio: absorbed into CUE tracks and not tracked in
525 # provider_mappings, so they would otherwise flag as changed every sync
526 items_to_process = [
527 (item, prev)
528 for item, prev in items_to_process
529 if not (
530 item.ext in TRACK_EXTENSIONS
531 and item.absolute_path.rsplit(".", 1)[0] in cue_stems
532 )
533 ]
534 # register synthetic track IDs for unchanged CUE files so the
535 # deletion pass does not treat them as removed
536 for cue_item in unchanged_cue_items:
537 try:
538 cue_sheet = await self._cue.load_cue_sheet(cue_item)
539 except InvalidDataError as err:
540 self.logger.warning(
541 "Unable to parse CUE sheet %s: %s", cue_item.relative_path, err
542 )
543 continue
544 for cue_track in cue_sheet.tracks:
545 cur_filenames.add(make_cue_track_id(cue_item.relative_path, cue_track.number))
546 total_items = len(items_to_process)
547 self.logger.info(
548 "Found %d changed/new items to process for %s",
549 total_items,
550 self.name,
551 )
552
553 # _SYNC_CONCURRENCY caps parallelism per provider (NFS/SMB/WebDAV friendly)
554 processed_count = 0
555
556 async def _process(item: FileSystemItem, prev_checksum: str | None) -> None:
557 nonlocal processed_count
558 if await self._process_item_async(
559 item, prev_checksum, cur_filenames, cue_stems, prev_filenames
560 ):
561 cur_filenames.add(item.relative_path)
562 processed_count += 1
563 if processed_count % 50 == 0 or processed_count == total_items:
564 update_current_task_progress_from_index(
565 processed_count,
566 total_items,
567 f"Processed {processed_count}/{total_items} files",
568 )
569
570 async with TaskManager(self.mass, self._SYNC_CONCURRENCY) as tm:
571 for item, prev_checksum in items_to_process:
572 await tm.create_task_with_limit(_process(item, prev_checksum))
573 finally:
574 self.sync_running = False
575
576 # do not run deletions on a clean but empty scan of a previously non-empty library
577 # (wrong share mounted, empty backup mount, ...)
578 if prev_filenames and not cur_filenames:
579 self.logger.error(
580 "Aborting sync for %s: scan found no files but %d were previously indexed",
581 self.name,
582 len(prev_filenames),
583 )
584 report_current_task_failure(
585 f"Sync aborted: scan found no files but {len(prev_filenames)} "
586 "were previously indexed"
587 )
588 return
589
590 # a scan that skipped folders or files is incomplete: what it missed is still
591 # there, so deleting it from the library would throw away valid content
592 if scan_errors.incomplete:
593 summary = scan_errors.describe()
594 self.logger.warning("Skipping deletions for %s: %s", self.name, summary)
595 report_current_task_failure(f"Deletions skipped: {summary}")
596 else:
597 deleted_files = prev_filenames - cur_filenames
598 await self._process_deletions(deleted_files)
599 await self._process_orphaned_albums_and_artists()
600
601 # flag provider as available again if an earlier sync had marked it down
602 self._set_available(True)
603
604 async def get_artist(self, prov_artist_id: str) -> Artist:
605 """Get full artist details by id."""
606 db_artist = await self.mass.music.artists.get_library_item_by_prov_id(
607 prov_artist_id, self.instance_id
608 )
609 if not db_artist:
610 # this may happen if the artist is not in the db yet
611 # e.g. when browsing the filesystem
612 if await self.exists(prov_artist_id):
613 return await self._parse_artist(prov_artist_id, artist_path=prov_artist_id)
614 return await self._parse_artist(prov_artist_id)
615
616 # prov_artist_id is either an actual (relative) path or a name (as fallback)
617 safe_artist_name = create_safe_string(prov_artist_id, lowercase=False, replace_space=False)
618 if await self.exists(prov_artist_id):
619 artist_path = prov_artist_id
620 elif await self.exists(safe_artist_name):
621 artist_path = safe_artist_name
622 else:
623 for prov_mapping in db_artist.provider_mappings:
624 if prov_mapping.provider_instance != self.instance_id:
625 continue
626 if prov_mapping.url:
627 artist_path = prov_mapping.url
628 break
629 else:
630 # this is an artist without an actual path on disk
631 # return the info we already have in the db
632 return db_artist
633 return await self._parse_artist(
634 db_artist.name,
635 sort_name=db_artist.sort_name,
636 mbid=db_artist.mbid,
637 artist_path=artist_path,
638 )
639
640 async def get_album(self, prov_album_id: str) -> Album:
641 """Get full album details by id."""
642 parsed_cue_paths: set[str] = set()
643 for track in await self.get_album_tracks(prov_album_id):
644 for prov_mapping in track.provider_mappings:
645 if prov_mapping.provider_instance != self.instance_id:
646 continue
647 if parsed := parse_cue_track_id(prov_mapping.item_id):
648 # every track from the same CUE shares the same album; only parse once
649 if parsed[0] in parsed_cue_paths:
650 continue
651 parsed_cue_paths.add(parsed[0])
652 cue_item = await self.resolve(parsed[0])
653 for cue_track in await self._cue.parse_tracks(cue_item):
654 if isinstance(cue_track.album, Album):
655 return cue_track.album
656 continue
657 file_item = await self.resolve(prov_mapping.item_id)
658 tags = await async_parse_tags(file_item.absolute_path, file_item.file_size)
659 full_track = await self._parse_track(file_item, tags)
660 assert isinstance(full_track.album, Album)
661 return full_track.album
662 msg = f"Album not found: {prov_album_id}"
663 raise MediaNotFoundError(msg)
664
665 async def get_track(self, prov_track_id: str) -> Track:
666 """Get full track details by id."""
667 # ruff: noqa: PLR0915
668 if parsed := parse_cue_track_id(prov_track_id):
669 cue_item = await self.resolve(parsed[0])
670 for cue_track in await self._cue.parse_tracks(cue_item):
671 if cue_track.item_id == prov_track_id:
672 return cue_track
673 msg = f"CUE track not found: {prov_track_id}"
674 raise MediaNotFoundError(msg)
675
676 if not await self.exists(prov_track_id):
677 msg = f"Track path does not exist: {prov_track_id}"
678 raise MediaNotFoundError(msg)
679
680 file_item = await self.resolve(prov_track_id)
681 tags = await async_parse_tags(file_item.absolute_path, file_item.file_size)
682 return await self._parse_track(file_item, tags=tags, full_album_metadata=True)
683
684 async def get_podcast_episode(self, prov_episode_id: str) -> PodcastEpisode:
685 """Get (full) podcast episode details by id."""
686 if not await self.exists(prov_episode_id):
687 msg = f"Episode path does not exist: {prov_episode_id}"
688 raise MediaNotFoundError(msg)
689 file_item = await self.resolve(prov_episode_id)
690 tags = await async_parse_tags(file_item.absolute_path, file_item.file_size)
691 return await self._parse_podcast_episode(file_item, tags=tags)
692
693 async def get_playlist(self, prov_playlist_id: str) -> Playlist:
694 """Get full playlist details by id."""
695 if not await self.exists(prov_playlist_id):
696 msg = f"Playlist path does not exist: {prov_playlist_id}"
697 raise MediaNotFoundError(msg)
698
699 file_item = await self.resolve(prov_playlist_id)
700 playlist = Playlist(
701 item_id=file_item.relative_path,
702 provider=self.instance_id,
703 name=file_item.name,
704 provider_mappings={
705 ProviderMapping(
706 item_id=file_item.relative_path,
707 provider_domain=self.domain,
708 provider_instance=self.instance_id,
709 details=file_item.checksum,
710 in_library=True,
711 )
712 },
713 )
714 playlist.is_editable = ProviderFeature.PLAYLIST_TRACKS_EDIT in self.supported_features
715 # only playlists in the root are editable - all other are read only
716 if "/" in prov_playlist_id or "\\" in prov_playlist_id:
717 playlist.is_editable = False
718 # we do not (yet) have support to edit/create pls playlists, only m3u files can be edited
719 if file_item.ext == "pls":
720 playlist.is_editable = False
721 playlist.owner = self.name
722 # Check for local image with the same basename
723 if local_image := await self._get_playlist_local_image(file_item):
724 playlist.metadata.images = UniqueList([local_image])
725 return playlist
726
727 async def get_audiobook(self, prov_audiobook_id: str) -> Audiobook:
728 """Get full audiobook details by id."""
729 # ruff: noqa: PLR0915
730 if not await self.exists(prov_audiobook_id):
731 msg = f"Audiobook path does not exist: {prov_audiobook_id}"
732 raise MediaNotFoundError(msg)
733
734 file_item = await self.resolve(prov_audiobook_id)
735 tags = await async_parse_tags(file_item.absolute_path, file_item.file_size)
736 return await self._parse_audiobook(file_item, tags=tags)
737
738 async def get_podcast(self, prov_podcast_id: str) -> Podcast:
739 """Get full podcast details by id."""
740 async for episode in self.get_podcast_episodes(prov_podcast_id):
741 assert isinstance(episode.podcast, Podcast)
742 return episode.podcast
743 msg = f"Podcast not found: {prov_podcast_id}"
744 raise MediaNotFoundError(msg)
745
746 async def get_sound_effect(self, prov_sound_effect_id: str) -> SoundEffect:
747 """Get full sound effect details by id."""
748 if not await self.exists(prov_sound_effect_id):
749 msg = f"Sound effect path does not exist: {prov_sound_effect_id}"
750 raise MediaNotFoundError(msg)
751 file_item = await self.resolve(prov_sound_effect_id)
752 return await self._get_or_parse_sound_effect(file_item)
753
754 async def get_sound_effects(self) -> AsyncGenerator[SoundEffect]:
755 """Get all sound effect items this provider offers."""
756
757 def _walk() -> list[FileSystemItem]:
758 return sorted(
759 recursive_iter(
760 self.base_path, self.base_path, SOUND_EFFECT_EXTENSIONS, self.logger
761 ),
762 key=lambda x: x.relative_path,
763 )
764
765 for file_item in await asyncio.to_thread(_walk):
766 yield await self._get_or_parse_sound_effect(file_item)
767
768 async def get_album_tracks(self, prov_album_id: str) -> list[Track]:
769 """Get album tracks for given album id."""
770 # filesystem items are always stored in db so we can query the database
771 db_album = await self.mass.music.albums.get_library_item_by_prov_id(
772 prov_album_id, self.instance_id
773 )
774 if db_album is None:
775 msg = f"Album not found: {prov_album_id}"
776 raise MediaNotFoundError(msg)
777 album_tracks = await self.mass.music.albums.get_library_album_tracks(db_album.item_id)
778 return [
779 track
780 for track in album_tracks
781 if any(x.provider_instance == self.instance_id for x in track.provider_mappings)
782 ]
783
784 async def get_playlist_tracks(self, prov_playlist_id: str, page: int = 0) -> list[Track]:
785 """Get playlist tracks."""
786 result: list[Track] = []
787 if page > 0:
788 # paging not (yet) supported
789 return result
790 if not await self.exists(prov_playlist_id):
791 msg = f"Playlist path does not exist: {prov_playlist_id}"
792 raise MediaNotFoundError(msg)
793
794 file_item = await self.resolve(prov_playlist_id)
795 # We are using the checksum of the playlist file here to invalidate the cache
796 # when a change has been made to the playlist file (ie track addition/deletion)
797 cache_checksum = file_item.checksum
798
799 cache_key = f"get_playlist_tracks.{prov_playlist_id}"
800 cached_data = await self.mass.cache.get(
801 cache_key,
802 provider=self.instance_id,
803 checksum=cache_checksum,
804 category=0,
805 base_class=Track,
806 )
807 if cached_data is not None:
808 return cached_data # type: ignore[no-any-return]
809
810 _, ext = prov_playlist_id.rsplit(".", 1)
811 try:
812 # get playlist file contents
813 playlist_data_raw = await self._read_file(prov_playlist_id)
814 encoding = await detect_charset(playlist_data_raw)
815 playlist_data = playlist_data_raw.decode(encoding, errors="replace")
816
817 if ext in ("m3u", "m3u8"):
818 playlist_lines = parse_m3u(playlist_data)
819 else:
820 playlist_lines = parse_pls(playlist_data)
821
822 for idx, playlist_line in enumerate(playlist_lines, 1):
823 if "#EXT" in playlist_line.path:
824 continue
825 if track := await self._parse_playlist_line(
826 playlist_line.path, os.path.dirname(prov_playlist_id)
827 ):
828 track.position = idx
829 result.append(track)
830
831 except Exception as err:
832 self.logger.warning(
833 "Error while parsing playlist %s: %s",
834 prov_playlist_id,
835 str(err),
836 exc_info=err if self.logger.isEnabledFor(10) else None,
837 )
838
839 await self.mass.cache.set(
840 key=cache_key,
841 data=[track.to_dict() for track in result],
842 expiration=3600 * 24 * 365, # File timestamp checksum handles invalidation
843 provider=self.instance_id,
844 checksum=cache_checksum,
845 category=0,
846 )
847
848 return result
849
850 async def get_podcast_episodes(self, prov_podcast_id: str) -> AsyncGenerator[PodcastEpisode]:
851 """Get podcast episodes for given podcast id."""
852 folder_items = [item for item in await self._scandir(prov_podcast_id) if not item.is_dir]
853 episode_files = [x for x in folder_items if x.ext in PODCAST_EPISODE_EXTENSIONS]
854 # artwork and metadata.json count towards the signature too, because the parse embeds
855 # them into every episode. Case-insensitive, matching _get_podcast_metadata's exists()
856 signature_files = [
857 x
858 for x in folder_items
859 if x.ext in PODCAST_EPISODE_EXTENSIONS
860 or x.ext in IMAGE_EXTENSIONS
861 or x.filename.lower() == "metadata.json"
862 ]
863 cache_key = f"podcast_episodes.{prov_podcast_id}"
864 cache_checksum = get_folder_signature(signature_files)
865 if (
866 cached_episodes := await self.mass.cache.get(
867 cache_key,
868 provider=self.instance_id,
869 category=CACHE_CATEGORY_PODCAST_EPISODES,
870 checksum=cache_checksum,
871 base_class=PodcastEpisode,
872 )
873 ) is not None:
874 for episode in cached_episodes:
875 yield episode
876 return
877
878 # these caches have no checksum of their own, so drop them before parsing or the new
879 # entry gets the values the signature just invalidated. Refill once, or every parse
880 # task below misses at the same time and repeats the same scandir and file read
881 for stale_category in (CACHE_CATEGORY_FOLDER_IMAGES, CACHE_CATEGORY_PODCAST_METADATA):
882 await self.mass.cache.delete(
883 prov_podcast_id, category=stale_category, provider=self.instance_id
884 )
885 await self._get_local_images(prov_podcast_id)
886 await self._get_podcast_metadata(prov_podcast_id)
887
888 # collected by index so the listing keeps scandir order, not parse completion order
889 parsed: list[PodcastEpisode | None] = [None] * len(episode_files)
890
891 async def _process_podcast_episode(index: int, item: FileSystemItem) -> None:
892 try:
893 tags = await async_parse_tags(item.absolute_path, item.file_size)
894 parsed[index] = await self._parse_podcast_episode(item, tags)
895 except MusicAssistantError as err:
896 self.logger.warning(
897 "Could not parse uri/file %s to podcast episode: %s",
898 item.relative_path,
899 str(err),
900 )
901
902 # reuse the per-sync worker limit: the slowest filesystems to parse are exactly the
903 # ones that lower it
904 async with TaskManager(self.mass, self._SYNC_CONCURRENCY) as tm:
905 for index, item in enumerate(episode_files):
906 await tm.create_task_with_limit(_process_podcast_episode(index, item))
907
908 episodes = [episode for episode in parsed if episode is not None]
909 # cache an incomplete listing briefly rather than not at all, so one unreadable file
910 # cannot make every request re-parse the whole folder
911 complete = len(episodes) == len(episode_files)
912 await self.mass.cache.set(
913 key=cache_key,
914 data=[episode.to_dict() for episode in episodes],
915 # a complete listing is invalidated by the folder signature instead
916 expiration=3600 * 24 * 365 if complete else PARTIAL_LISTING_CACHE_EXPIRATION,
917 provider=self.instance_id,
918 category=CACHE_CATEGORY_PODCAST_EPISODES,
919 checksum=cache_checksum,
920 )
921
922 for episode in episodes:
923 yield episode
924
925 async def add_playlist_tracks(self, prov_playlist_id: str, prov_track_ids: list[str]) -> None:
926 """Add track(s) to playlist."""
927 if not await self.exists(prov_playlist_id):
928 msg = f"Playlist path does not exist: {prov_playlist_id}"
929 raise MediaNotFoundError(msg)
930 playlist_filename = self.get_absolute_path(prov_playlist_id)
931 async with aiofiles.open(playlist_filename, encoding="utf-8") as _file:
932 playlist_data = await _file.read()
933 for file_path in prov_track_ids:
934 track = await self.get_track(file_path)
935 playlist_data += f"\n#EXTINF:{track.duration or 0},{track.name}\n{file_path}\n"
936
937 # write playlist file (always in utf-8)
938 async with aiofiles.open(playlist_filename, "w", encoding="utf-8") as _file:
939 await _file.write(playlist_data)
940
941 async def remove_playlist_tracks(
942 self, prov_playlist_id: str, positions_to_remove: tuple[int, ...]
943 ) -> None:
944 """Remove track(s) from playlist."""
945 if not await self.exists(prov_playlist_id):
946 msg = f"Playlist path does not exist: {prov_playlist_id}"
947 raise MediaNotFoundError(msg)
948 _, ext = prov_playlist_id.rsplit(".", 1)
949 # get playlist file contents
950 playlist_filename = self.get_absolute_path(prov_playlist_id)
951 async with aiofiles.open(playlist_filename, encoding="utf-8") as _file:
952 playlist_data = await _file.read()
953 # get current contents first
954 if ext in ("m3u", "m3u8"):
955 playlist_items = parse_m3u(playlist_data)
956 else:
957 playlist_items = parse_pls(playlist_data)
958 # remove items by index
959 for i in sorted(positions_to_remove, reverse=True):
960 # position = index + 1
961 del playlist_items[i - 1]
962 # build new playlist data
963 new_playlist_data = "#EXTM3U\n"
964 for item in playlist_items:
965 new_playlist_data += f"\n#EXTINF:{item.length or 0},{item.title}\n{item.path}\n"
966 async with aiofiles.open(playlist_filename, "w", encoding="utf-8") as _file:
967 await _file.write(new_playlist_data)
968
969 async def create_playlist(self, name: str, media_types: set[MediaType]) -> Playlist:
970 """Create a new playlist on provider with given name."""
971 # creating a new playlist on the filesystem is as easy
972 # as creating a new (empty) file with the m3u extension...
973 # filename = await self.resolve(f"{name}.m3u")
974 filename = f"{name}.m3u"
975 playlist_filename = self.get_absolute_path(filename)
976 async with aiofiles.open(playlist_filename, "w", encoding="utf-8") as _file:
977 await _file.write("#EXTM3U\n")
978 return await self.get_playlist(filename)
979
980 async def get_stream_details(self, item_id: str, media_type: MediaType) -> StreamDetails:
981 """Return the content details for the given track when it will be streamed."""
982 try:
983 if media_type == MediaType.AUDIOBOOK:
984 return await self._get_stream_details_for_audiobook(item_id)
985 if media_type == MediaType.PODCAST_EPISODE:
986 return await self._get_stream_details_for_podcast_episode(item_id)
987 if media_type == MediaType.SOUND_EFFECT:
988 return await self._get_stream_details_for_sound_effect(item_id)
989 return await self._get_stream_details_for_track(item_id)
990 except FileNotFoundError:
991 self.logger.warning(
992 "File not found for media item %s",
993 item_id,
994 )
995 msg = f"Media file not found: {item_id}"
996 raise MediaNotFoundError(msg)
997
998 async def get_audio_stream(
999 self, streamdetails: StreamDetails, seek_position: int = 0
1000 ) -> AsyncGenerator[bytes]:
1001 """Return the custom audio stream for the provider item."""
1002 # only CUE-derived tracks use StreamType.CUSTOM in this provider
1003 async for chunk in self._cue.get_audio_stream(streamdetails, seek_position):
1004 yield chunk
1005
1006 async def resolve_image(self, path: str) -> str | bytes:
1007 """
1008 Resolve an image from an image path.
1009
1010 This either returns (a generator to get) raw bytes of the image or
1011 a string with an http(s) URL or local path that is accessible from the server.
1012 """
1013 # drop the cache-busting suffix appended by _versioned_image_path
1014 try:
1015 file_item = await self.resolve(path.split("?cs=", 1)[0])
1016 except FileNotFoundError as err:
1017 # the referenced image file was removed from disk; surface a typed
1018 # not-found so the image layer treats it as a missing image
1019 raise MediaNotFoundError(f"Image not found: {path}") from err
1020 if file_item.is_dir:
1021 # handing the path back would have the image layer run an ffmpeg
1022 # embedded-artwork extraction on the directory before giving up
1023 raise MediaNotFoundError(f"Image path is a directory: {path}")
1024 return file_item.absolute_path
1025
1026 async def check_write_access(self) -> None:
1027 """Perform check if we have write access."""
1028 # verify write access to determine we have playlist create/edit support
1029 # overwrite with provider specific implementation if needed
1030 temp_file_name = self.get_absolute_path(f"{shortuuid.random(8)}.txt")
1031 try:
1032 async with aiofiles.open(temp_file_name, "w") as _file:
1033 await _file.write("test")
1034 await asyncio.to_thread(os.remove, temp_file_name)
1035 self.write_access = True
1036 except Exception as err:
1037 self.logger.debug("Write access disabled: %s", str(err))
1038
1039 async def resolve(self, file_path: str) -> FileSystemItem:
1040 """Resolve (absolute or relative) path to FileSystemItem."""
1041 absolute_path = self.get_absolute_path(file_path)
1042
1043 def _create_item() -> FileSystemItem:
1044 if Path(absolute_path).is_dir():
1045 return FileSystemItem(
1046 filename=Path(file_path).name,
1047 relative_path=get_relative_path(self.base_path, file_path),
1048 absolute_path=absolute_path,
1049 is_dir=True,
1050 )
1051 stat_info = Path(absolute_path).stat(follow_symlinks=False)
1052 return FileSystemItem(
1053 filename=Path(file_path).name,
1054 relative_path=get_relative_path(self.base_path, file_path),
1055 absolute_path=absolute_path,
1056 is_dir=False,
1057 checksum=str(int(stat_info.st_mtime)),
1058 file_size=stat_info.st_size,
1059 )
1060
1061 return await asyncio.to_thread(_create_item)
1062
1063 async def exists(self, file_path: str) -> bool:
1064 """Return bool is this FileSystem musicprovider has given file/dir."""
1065 if not file_path:
1066 return False
1067 try:
1068 abs_path = self.get_absolute_path(file_path)
1069 except MediaNotFoundError:
1070 # a path that escapes the base directory simply does not exist here
1071 return False
1072 return bool(await exists(abs_path))
1073
1074 def get_absolute_path(self, file_path: str) -> str:
1075 """Return absolute path for given file path."""
1076 return get_absolute_path(self.base_path, file_path)
1077
1078 async def _enumerate_files_for_sync(
1079 self,
1080 *,
1081 file_checksums: dict[str, str],
1082 cue_file_checksums: dict[str, set[str]],
1083 cur_filenames: set[str],
1084 items_to_process: list[tuple[FileSystemItem, str | None]],
1085 unchanged_cue_items: list[FileSystemItem],
1086 cue_stems: set[str],
1087 scan_errors: ScanErrors,
1088 ) -> None:
1089 """
1090 Walk every supported file under the provider root and populate the sync buckets.
1091
1092 Override in subclasses that cannot use a local ``os.scandir`` walk.
1093 Implementations must route each discovered file through
1094 :meth:`_classify_scan_item`, report every unreadable directory to
1095 ``scan_errors`` and stop the walk once it reports ``aborted``.
1096
1097 :param file_checksums: Previously stored checksum per provider item id.
1098 :param cue_file_checksums: Previously stored track checksums keyed by CUE relative_path.
1099 :param cur_filenames: Receives the ids/paths present in this scan.
1100 :param items_to_process: Receives changed or new items to process.
1101 :param unchanged_cue_items: Receives CUE sheets whose checksum matches.
1102 :param cue_stems: Receives absolute paths (minus extension) of CUE sheets.
1103 :param scan_errors: Receives the errors raised while walking the tree.
1104 """
1105 ignore_album_playlists = self.media_content_type == "music" and bool(
1106 self.config.get_value(CONF_ENTRY_IGNORE_ALBUM_PLAYLISTS.key)
1107 )
1108
1109 def _walk() -> None:
1110 for scanned, item in enumerate(
1111 recursive_iter(
1112 self.base_path,
1113 self.base_path,
1114 SUPPORTED_EXTENSIONS,
1115 self.logger,
1116 scan_errors=scan_errors,
1117 ),
1118 start=1,
1119 ):
1120 if scanned % 500 == 0:
1121 update_current_task_progress_text(f"Scanning files: {scanned} found")
1122 self._classify_scan_item(
1123 item,
1124 file_checksums=file_checksums,
1125 cue_file_checksums=cue_file_checksums,
1126 cur_filenames=cur_filenames,
1127 items_to_process=items_to_process,
1128 unchanged_cue_items=unchanged_cue_items,
1129 cue_stems=cue_stems,
1130 ignore_album_playlists=ignore_album_playlists,
1131 )
1132
1133 await asyncio.to_thread(_walk)
1134
1135 def _classify_scan_item(
1136 self,
1137 item: FileSystemItem,
1138 *,
1139 file_checksums: dict[str, str],
1140 cue_file_checksums: dict[str, set[str]],
1141 cur_filenames: set[str],
1142 items_to_process: list[tuple[FileSystemItem, str | None]],
1143 unchanged_cue_items: list[FileSystemItem],
1144 cue_stems: set[str],
1145 ignore_album_playlists: bool,
1146 ) -> None:
1147 """
1148 Route a single scanned file into the correct sync bucket.
1149
1150 :param item: The file to classify.
1151 :param file_checksums: Previously stored checksum per provider item id.
1152 :param cue_file_checksums: Previously stored track checksums keyed by CUE relative_path.
1153 :param cur_filenames: Receives the ids/paths present in this scan.
1154 :param items_to_process: Receives changed or new items to process.
1155 :param unchanged_cue_items: Receives CUE sheets whose checksum matches.
1156 :param cue_stems: Receives absolute paths (minus extension) of CUE sheets.
1157 :param ignore_album_playlists: When True, skip playlists nested inside
1158 album directories.
1159 """
1160 # a file this provider never imports gets no mapping, so it would flag as
1161 # changed on every sync; it is still on disk, so record it as present
1162 if not self._is_imported_file(item):
1163 cur_filenames.add(item.relative_path)
1164 return
1165 # skip playlists in album directories if configured
1166 if (
1167 item.ext in PLAYLIST_EXTENSIONS
1168 and ignore_album_playlists
1169 and len(item.relative_path.split("/")) > 2
1170 ):
1171 return
1172 is_cue = item.ext in CUE_EXTENSIONS and self.media_content_type == "music"
1173 item_checksum = item.checksum
1174 if is_cue:
1175 cue_stems.add(item.absolute_path.rsplit(".", 1)[0])
1176 item_checksum = cue_metadata_checksum(item.checksum)
1177 prev_checksums = cue_file_checksums.get(item.relative_path, set())
1178 prev_checksum = min(prev_checksums, default=None)
1179 checksum_matches = prev_checksums == {item_checksum}
1180 else:
1181 prev_checksum = file_checksums.get(item.relative_path)
1182 checksum_matches = item_checksum == prev_checksum
1183 if checksum_matches:
1184 # unchanged, just record it as still present
1185 cur_filenames.add(item.relative_path)
1186 if is_cue:
1187 unchanged_cue_items.append(item)
1188 else:
1189 items_to_process.append((item, prev_checksum))
1190
1191 def _is_imported_file(self, item: FileSystemItem) -> bool:
1192 """Return True when this provider imports the given file into the library."""
1193 if self.media_content_type == "music":
1194 if item.ext in CUE_EXTENSIONS:
1195 return True
1196 if item.ext in TRACK_EXTENSIONS:
1197 return self._sync_tracks
1198 if item.ext in PLAYLIST_EXTENSIONS:
1199 return self._sync_playlists
1200 return False
1201 if self.media_content_type == "audiobooks":
1202 return item.ext in AUDIOBOOK_EXTENSIONS
1203 if self.media_content_type == "podcasts":
1204 return item.ext in PODCAST_EPISODE_EXTENSIONS
1205 return False
1206
1207 def _set_available(self, available: bool) -> None:
1208 """Update the provider availability and notify listeners on change."""
1209 if self.available == available:
1210 return
1211 self.available = available
1212 if available:
1213 self._cancel_availability_probe()
1214 else:
1215 self._schedule_availability_probe()
1216 self.mass.signal_event(EventType.PROVIDERS_UPDATED, data=self.mass.get_providers())
1217
1218 async def _is_reachable(self) -> bool:
1219 """Return whether the storage backing this provider can be read."""
1220 return bool(await isdir(self.base_path))
1221
1222 @property
1223 def _availability_probe_id(self) -> str:
1224 """Return the timer id of this provider's reachability checks."""
1225 return f"filesystem_availability_probe_{self.instance_id}"
1226
1227 def _schedule_availability_probe(self) -> None:
1228 """Arm the next reachability check."""
1229 self.mass.call_later(
1230 AVAILABILITY_PROBE_INTERVAL,
1231 self._probe_availability,
1232 task_id=self._availability_probe_id,
1233 )
1234
1235 def _cancel_availability_probe(self) -> None:
1236 """Stop checking for the storage coming back."""
1237 self.mass.cancel_timer(self._availability_probe_id)
1238
1239 async def _probe_availability(self) -> None:
1240 """Mark the provider available again once its storage can be read."""
1241 try:
1242 reachable = await self._is_reachable()
1243 except MusicAssistantError as err:
1244 # storage that is simply still gone, which is what this loop waits for
1245 self.logger.debug("%s is still unreachable: %s", self.name, err)
1246 reachable = False
1247 except Exception:
1248 # an unexpected failure must not end the loop, since it is what brings the
1249 # provider back, but it is a defect rather than an outage so it is logged loudly
1250 self.logger.exception("Reachability check for %s failed", self.name)
1251 reachable = False
1252 if self.unloading:
1253 # the provider was torn down while this check was running; re-arming here
1254 # would leave a timer firing against an instance nothing owns anymore
1255 return
1256 if reachable:
1257 self.logger.info("%s is reachable again", self.name)
1258 self._set_available(True)
1259 return
1260 self._schedule_availability_probe()
1261
1262 async def _process_item_async(
1263 self,
1264 item: FileSystemItem,
1265 prev_checksum: str | None,
1266 cur_filenames: set[str] | None = None,
1267 cue_stems: set[str] | None = None,
1268 prev_filenames: set[str] | None = None,
1269 ) -> bool:
1270 """
1271 Process a single item asynchronously.
1272
1273 :param item: The filesystem item to process.
1274 :param prev_checksum: Previous checksum from the database, or None for new items.
1275 :param cur_filenames: Set of current filenames being tracked (for CUE track IDs).
1276 :param cue_stems: Absolute paths (without extension) of CUE sheets in this scan,
1277 used to detect companion-CUE audio files without a filesystem stat.
1278 :param prev_filenames: The ids/paths the previous scan found, used to keep the
1279 ids of a CUE sheet that fails to parse.
1280 """
1281 try:
1282 self.logger.log(VERBOSE_LOG_LEVEL, "Processing: %s", item.relative_path)
1283
1284 if prev_checksum is not None:
1285 # the file changed on disk: drop cached artwork derived from it
1286 # (thumbnails, source bytes, palette) so re-read embedded art is
1287 # served fresh, for both reference forms of the image path
1288 await self.mass.metadata.invalidate_image_cache(
1289 self.instance_id, item.relative_path
1290 )
1291 await self.mass.metadata.invalidate_image_cache(
1292 self.instance_id, self._versioned_image_path(item.relative_path, prev_checksum)
1293 )
1294
1295 if item.ext in CUE_EXTENSIONS and self.media_content_type == "music":
1296 tracks = await self._cue.parse_tracks(item)
1297 for track in tracks:
1298 track.favorite = False
1299 await self.mass.music.tracks.add_item_to_library(
1300 track, overwrite_existing=prev_checksum is not None
1301 )
1302 if cur_filenames is not None:
1303 cur_filenames.add(track.item_id)
1304 return True
1305
1306 if item.ext in TRACK_EXTENSIONS and self.media_content_type == "music":
1307 if not self._sync_tracks:
1308 return False
1309 # skip audio files that have a companion CUE sheet
1310 if cue_stems is not None and item.absolute_path.rsplit(".", 1)[0] in cue_stems:
1311 return False
1312 tags = await async_parse_tags(item.absolute_path, item.file_size)
1313 track = await self._parse_track(item, tags)
1314 track.favorite = False # TODO: implement favorite status based on rating ?
1315 await self.mass.music.tracks.add_item_to_library(
1316 track, overwrite_existing=prev_checksum is not None
1317 )
1318 return True
1319
1320 if item.ext in AUDIOBOOK_EXTENSIONS and self.media_content_type == "audiobooks":
1321 tags = await async_parse_tags(item.absolute_path, item.file_size)
1322 try:
1323 audiobook = await self._parse_audiobook(item, tags)
1324 except IsChapterFile:
1325 return True
1326 await self.mass.music.audiobooks.add_item_to_library(
1327 audiobook, overwrite_existing=prev_checksum is not None
1328 )
1329 return True
1330
1331 if item.ext in PODCAST_EPISODE_EXTENSIONS and self.media_content_type == "podcasts":
1332 tags = await async_parse_tags(item.absolute_path, item.file_size)
1333 episode = await self._parse_podcast_episode(item, tags)
1334 assert isinstance(episode.podcast, Podcast)
1335 await self.mass.music.podcasts.add_item_to_library(
1336 episode.podcast, overwrite_existing=prev_checksum is not None
1337 )
1338 return True
1339
1340 if item.ext in PLAYLIST_EXTENSIONS and self.media_content_type == "music":
1341 if not self._sync_playlists:
1342 return False
1343 playlist = await self.get_playlist(item.relative_path)
1344 await self.mass.music.playlists.add_item_to_library(
1345 playlist, overwrite_existing=prev_checksum is not None
1346 )
1347 return True
1348
1349 except Exception as err:
1350 # we don't want the whole sync to crash on one file so we catch all exceptions here
1351 self.logger.error(
1352 "Error processing %s - %s",
1353 item.relative_path,
1354 str(err),
1355 exc_info=err if self.logger.isEnabledFor(logging.DEBUG) else None,
1356 )
1357 report_current_task_failure(f"Failed to process {item.relative_path}: {err}")
1358 # the file is still on the storage, so keep it in the scan result:
1359 # leaving it out makes the deletion step treat it as removed
1360 self._keep_failed_item(item, cur_filenames, prev_filenames)
1361 return False
1362
1363 def _keep_failed_item(
1364 self,
1365 item: FileSystemItem,
1366 cur_filenames: set[str] | None,
1367 prev_filenames: set[str] | None,
1368 ) -> None:
1369 """
1370 Keep an item that could not be processed in the scan result.
1371
1372 :param item: The item that failed to process.
1373 :param cur_filenames: Receives the ids/paths present in this scan.
1374 :param prev_filenames: The ids/paths the previous scan found.
1375 """
1376 if cur_filenames is None:
1377 return
1378 cur_filenames.add(item.relative_path)
1379 if (
1380 item.ext not in CUE_EXTENSIONS
1381 or self.media_content_type != "music"
1382 or not prev_filenames
1383 ):
1384 return
1385 # a CUE sheet stands in for one id per track it describes and those cannot be
1386 # rebuilt without parsing it, so carry over the ids of the previous scan
1387 cur_filenames.update(
1388 item_id
1389 for item_id in prev_filenames
1390 if (parsed := parse_cue_track_id(item_id)) and parsed[0] == item.relative_path
1391 )
1392
1393 async def _process_orphaned_albums_and_artists(self) -> None:
1394 """Process deletion of orphaned albums and artists."""
1395 assert self.mass.music.database
1396 # Remove albums without any tracks
1397 query = (
1398 f"SELECT item_id FROM {DB_TABLE_ALBUMS} "
1399 f"WHERE item_id not in ( SELECT album_id from {DB_TABLE_ALBUM_TRACKS}) "
1400 f"AND item_id in ( SELECT item_id from {DB_TABLE_PROVIDER_MAPPINGS} "
1401 f"WHERE provider_instance = '{self.instance_id}' and media_type = 'album' )"
1402 )
1403 for db_row in await self.mass.music.database.get_rows_from_query(
1404 query,
1405 limit=100000,
1406 ):
1407 await self.mass.music.albums.remove_item_from_library(db_row["item_id"])
1408
1409 # Remove artists without any tracks or albums
1410 query = (
1411 f"SELECT item_id FROM {DB_TABLE_ARTISTS} "
1412 f"WHERE item_id not in "
1413 f"( select artist_id from {DB_TABLE_TRACK_ARTISTS} "
1414 f"UNION SELECT artist_id from {DB_TABLE_ALBUM_ARTISTS} ) "
1415 f"AND item_id in ( SELECT item_id from {DB_TABLE_PROVIDER_MAPPINGS} "
1416 f"WHERE provider_instance = '{self.instance_id}' and media_type = 'artist' )"
1417 )
1418 for db_row in await self.mass.music.database.get_rows_from_query(
1419 query,
1420 limit=100000,
1421 ):
1422 await self.mass.music.artists.remove_item_from_library(db_row["item_id"])
1423
1424 async def _process_deletions(self, deleted_files: set[str]) -> None:
1425 """Process all deletions."""
1426 # process deleted tracks/playlists
1427 album_ids = set()
1428 artist_ids = set()
1429 for file_path in deleted_files:
1430 if parse_cue_track_id(file_path) is not None and self.media_content_type == "music":
1431 controller = self.mass.music.get_controller(MediaType.TRACK)
1432 elif "." not in file_path:
1433 continue
1434 else:
1435 _, ext = file_path.rsplit(".", 1)
1436 if ext in PODCAST_EPISODE_EXTENSIONS and self.media_content_type == "podcasts":
1437 controller = self.mass.music.get_controller(MediaType.PODCAST_EPISODE)
1438 elif ext in AUDIOBOOK_EXTENSIONS and self.media_content_type == "audiobooks":
1439 controller = self.mass.music.get_controller(MediaType.AUDIOBOOK)
1440 elif ext in PLAYLIST_EXTENSIONS and self.media_content_type == "music":
1441 controller = self.mass.music.get_controller(MediaType.PLAYLIST)
1442 elif ext in TRACK_EXTENSIONS and self.media_content_type == "music":
1443 controller = self.mass.music.get_controller(MediaType.TRACK)
1444 else:
1445 # unsupported file extension?
1446 continue
1447
1448 if library_item := await controller.get_library_item_by_prov_id(
1449 file_path, self.instance_id
1450 ):
1451 if is_track(library_item):
1452 if library_item.album:
1453 album_ids.add(library_item.album.item_id)
1454 # need to fetch the library album to resolve the itemmapping
1455 db_album = await self.mass.music.albums.get_library_item(
1456 library_item.album.item_id
1457 )
1458 for artist in db_album.artists:
1459 artist_ids.add(artist.item_id)
1460 for artist in library_item.artists:
1461 artist_ids.add(artist.item_id)
1462 await controller.remove_item_from_library(library_item.item_id)
1463 # check if any albums need to be cleaned up
1464 for album_id in album_ids:
1465 if not await self.mass.music.albums.tracks(album_id, "library"):
1466 await self.mass.music.albums.remove_item_from_library(album_id)
1467 # check if any artists need to be cleaned up
1468 for artist_id in artist_ids:
1469 artist_albums = await self.mass.music.artists.albums(artist_id, "library")
1470 artist_tracks = await self.mass.music.artists.tracks(artist_id, "library")
1471 if not (artist_albums or artist_tracks):
1472 await self.mass.music.artists.remove_item_from_library(artist_id)
1473
1474 async def _get_playlist_local_image(self, file_item: FileSystemItem) -> MediaItemImage | None:
1475 """Return a local image alongside the playlist file (matching basename) if any."""
1476 cache_key = f"playlist_image.{file_item.relative_path}"
1477 cached = await self.cache.get(
1478 key=cache_key,
1479 provider=self.instance_id,
1480 category=CACHE_CATEGORY_FOLDER_IMAGES,
1481 base_class=MediaItemImage,
1482 )
1483 if cached is not None:
1484 return cached[0] if cached else None
1485 try:
1486 folder_files = await self._scandir(file_item.relative_parent_path)
1487 except OSError, MusicAssistantError:
1488 return None
1489 target = file_item.name.lower()
1490 result: MediaItemImage | None = None
1491 for item in folder_files:
1492 if item.is_dir or not item.ext:
1493 continue
1494 if item.ext.lower() not in IMAGE_EXTENSIONS:
1495 continue
1496 if item.name.lower() != target:
1497 continue
1498 result = MediaItemImage(
1499 type=ImageType.THUMB,
1500 path=item.relative_path,
1501 provider=self.instance_id,
1502 remotely_accessible=False,
1503 )
1504 break
1505 await self.cache.set(
1506 key=cache_key,
1507 data=[result.to_dict()] if result is not None else [],
1508 provider=self.instance_id,
1509 category=CACHE_CATEGORY_FOLDER_IMAGES,
1510 expiration=120,
1511 )
1512 return result
1513
1514 async def _parse_playlist_line(self, line: str, playlist_path: str) -> Track | None:
1515 """Try to parse a track from a playlist line."""
1516 try:
1517 line = line.replace("file://", "").strip()
1518 # try to resolve the filename (both normal and url decoded):
1519 # - relative to the playlist folder (normpath resolves parent .. references)
1520 # - as-is: an absolute path, or relative to our base path
1521 # candidates stay relative so subclasses with virtual paths (cloud,
1522 # webdav) resolve them too, instead of leaking the server CWD
1523 for _line in (line, urllib.parse.unquote(line)):
1524 if playlist_path:
1525 normalized = posixpath.normpath(f"{playlist_path}/{_line}")
1526 with contextlib.suppress(FileNotFoundError, MediaNotFoundError):
1527 file_item = await self.resolve(normalized)
1528 return await self._get_playlist_line_track(file_item)
1529 with contextlib.suppress(FileNotFoundError, MediaNotFoundError):
1530 file_item = await self.resolve(_line)
1531 return await self._get_playlist_line_track(file_item)
1532 # all attempts failed
1533 raise MediaNotFoundError("Invalid path/uri")
1534
1535 except MusicAssistantError as err:
1536 self.logger.warning("Could not parse %s to track: %s", line, str(err))
1537
1538 return None
1539
1540 async def _get_playlist_line_track(self, file_item: FileSystemItem) -> Track:
1541 """
1542 Return the track for a resolved playlist entry.
1543
1544 :param file_item: The resolved file the playlist entry points at.
1545 """
1546 # filesystem tracks are synced into the library, so prefer the database over
1547 # (expensive) tag parsing - this keeps loading large playlists fast
1548 library_track = await self.mass.music.tracks.get_library_item_by_prov_id(
1549 file_item.relative_path, self.instance_id
1550 )
1551 # only trust the library item if its mapping for this file is available: the file
1552 # just resolved, so an unavailable mapping is stale (e.g. the file was missing
1553 # during the last scan) and would wrongly exclude the track from playback
1554 if library_track is not None and any(
1555 mapping.provider_instance == self.instance_id
1556 and mapping.item_id == file_item.relative_path
1557 and mapping.available
1558 for mapping in library_track.provider_mappings
1559 ):
1560 # callers expect the provider item identity here (not the library one),
1561 # e.g. for duplicate detection when editing the playlist
1562 library_track.item_id = file_item.relative_path
1563 library_track.provider = self.instance_id
1564 library_track.uri = create_uri(
1565 MediaType.TRACK, self.instance_id, file_item.relative_path
1566 )
1567 return library_track
1568 # not (yet) in the library: parse the file tags
1569 tags = await async_parse_tags(file_item.absolute_path, file_item.file_size)
1570 return await self._parse_track(file_item, tags)
1571
1572 @staticmethod
1573 def _versioned_image_path(relative_path: str, checksum: str | None) -> str:
1574 """Append the file checksum so the image cache busts when the file is replaced."""
1575 if checksum:
1576 return f"{relative_path}?cs={checksum}"
1577 return relative_path
1578
1579 @staticmethod
1580 def _codec_type_from_tags(tags: AudioTags) -> ContentType:
1581 """Return the audio codec detected by ffprobe, if any."""
1582 if tags.raw and (streams := tags.raw.get("streams")):
1583 if codec_name := streams[0].get("codec_name"):
1584 return ContentType.try_parse(codec_name)
1585 return ContentType.UNKNOWN
1586
1587 async def _parse_track(
1588 self, file_item: FileSystemItem, tags: AudioTags, full_album_metadata: bool = False
1589 ) -> Track:
1590 """Parse full track details from file tags."""
1591 # ruff: noqa: PLR0915
1592 name, version = parse_title_and_version(tags.title, tags.version)
1593 track = Track(
1594 item_id=file_item.relative_path,
1595 provider=self.instance_id,
1596 name=name,
1597 sort_name=tags.title_sort,
1598 version=version,
1599 provider_mappings={
1600 ProviderMapping(
1601 item_id=file_item.relative_path,
1602 provider_domain=self.domain,
1603 provider_instance=self.instance_id,
1604 audio_format=AudioFormat(
1605 content_type=ContentType.try_parse(file_item.ext or tags.format),
1606 codec_type=self._codec_type_from_tags(tags),
1607 sample_rate=tags.sample_rate,
1608 bit_depth=tags.bits_per_sample,
1609 channels=tags.channels,
1610 bit_rate=tags.bit_rate,
1611 ),
1612 details=file_item.checksum,
1613 in_library=True,
1614 )
1615 },
1616 disc_number=tags.disc or 0,
1617 track_number=tags.track or 0,
1618 date_added=(
1619 datetime.fromtimestamp(file_item.created_at, tz=UTC)
1620 if file_item.created_at
1621 else None
1622 ),
1623 )
1624
1625 if isrc_tags := tags.isrc:
1626 for isrsc in isrc_tags:
1627 track.external_ids.add((ExternalID.ISRC, isrsc))
1628
1629 if acoustid := tags.get("acoustid"):
1630 track.external_ids.add((ExternalID.ACOUSTID, acoustid))
1631
1632 # album
1633 album = track.album = (
1634 await self._parse_album(
1635 track_path=file_item.relative_path,
1636 track_tags=tags,
1637 track_created_at=file_item.created_at,
1638 )
1639 if tags.album
1640 else None
1641 )
1642
1643 # track artist(s)
1644 resolved_track_artists = await self._resolve_artists_with_mbids(
1645 tags.artists,
1646 tags.musicbrainz_artistids,
1647 tags.artist_sort_names,
1648 log_label="ARTISTS tag",
1649 )
1650 for name, mbid, sort_name in resolved_track_artists:
1651 # prefer the existing album artist object when it's the same artist
1652 if album_artist_match := self._match_album_artist(album, name, mbid):
1653 track.artists.append(album_artist_match)
1654 continue
1655 artist = await self._parse_artist(name, sort_name=sort_name, mbid=mbid)
1656 track.artists.append(artist)
1657
1658 # handle embedded cover image
1659 if tags.has_cover_image:
1660 # we do not actually embed the image in the metadata because that would consume too
1661 # much space and bandwidth. Instead we set the filename as value so the image can
1662 # be retrieved later in realtime.
1663 track.metadata.images = UniqueList(
1664 [
1665 MediaItemImage(
1666 type=ImageType.THUMB,
1667 path=file_item.relative_path,
1668 provider=self.instance_id,
1669 remotely_accessible=False,
1670 )
1671 ]
1672 )
1673
1674 # copy (embedded) album image from track (if the album itself doesn't have an image)
1675 if album and not album.image and track.image:
1676 album.metadata.images = UniqueList([track.image])
1677
1678 # parse other info
1679 track.duration = int(tags.duration or 0)
1680 track.metadata.genres = set(tags.genres)
1681 if tags.disc:
1682 track.disc_number = tags.disc
1683 if tags.track:
1684 track.track_number = tags.track
1685 track.metadata.copyright = tags.get("copyright")
1686 track.metadata.lyrics = tags.lyrics
1687 track.metadata.grouping = tags.get("grouping")
1688 track.metadata.description = tags.get("comment")
1689 explicit_tag = tags.get("itunesadvisory")
1690 if explicit_tag is not None:
1691 track.metadata.explicit = explicit_tag == "1"
1692 if recording_mbid := clean_mbid(tags.musicbrainz_recordingid, tags.filename):
1693 track.mbid = recording_mbid
1694
1695 # handle (optional) loudness measurement tag(s)
1696 if tags.track_loudness is not None:
1697 self.mass.create_task(
1698 self.mass.streams.audio_analysis.set_track_loudness(
1699 track.item_id,
1700 self.instance_id,
1701 tags.track_loudness,
1702 tags.track_album_loudness,
1703 )
1704 )
1705
1706 # possible lrclib metadata
1707 # synced lyrics are saved as "filename.lrc" by lrcget alongside
1708 # the actual file location - just change the file extension
1709 assert file_item.ext is not None # for type checking
1710 lrc_path = f"{file_item.relative_path.removesuffix(file_item.ext)}lrc"
1711 if await self.exists(lrc_path):
1712 try:
1713 raw = await self._read_file(lrc_path)
1714 track.metadata.lrc_lyrics = raw.decode("utf-8")
1715 except Exception as err:
1716 self.logger.warning(
1717 "Failed to read lyrics file %s: %s",
1718 lrc_path,
1719 str(err),
1720 )
1721 elif syn_lyrics := tags.synchronized_lyrics:
1722 track.metadata.lrc_lyrics = lyrics.convert_to_lrc_lyrics(syn_lyrics)
1723
1724 return track
1725
1726 async def _resolve_artists_with_mbids(
1727 self,
1728 parsed_names: tuple[str, ...],
1729 mbids: tuple[str, ...],
1730 sort_names: tuple[str, ...],
1731 log_label: str,
1732 ) -> list[tuple[str, str | None, str | None]]:
1733 """
1734 Return ``(name, mbid, sort_name)`` triples for a track's or album's artists.
1735
1736 When the parsed name count and the MBID count disagree, canonical names
1737 are looked up from MusicBrainz; otherwise the tag-parsed names are used.
1738
1739 :param parsed_names: Tag-parsed artist names.
1740 :param mbids: MusicBrainz artist IDs from the tag.
1741 :param sort_names: Sort names from the corresponding *sort tag.
1742 :param log_label: Tag name used in warning messages (e.g. "ARTISTS tag").
1743 """
1744
1745 def _sort_name(index: int) -> str | None:
1746 return sort_names[index] if index < len(sort_names) else None
1747
1748 def _from_tags() -> list[tuple[str, str | None, str | None]]:
1749 return [
1750 (
1751 name,
1752 mbids[i] if i < len(mbids) else None,
1753 _sort_name(i),
1754 )
1755 for i, name in enumerate(parsed_names)
1756 ]
1757
1758 if not mbids or len(parsed_names) == len(mbids):
1759 return _from_tags()
1760
1761 mb_provider = cast("MusicbrainzProvider | None", self.mass.get_provider("musicbrainz"))
1762 if mb_provider is None:
1763 self.logger.warning(
1764 "%s count (%d) doesn't match MBID count (%d) and MusicBrainz "
1765 "provider is not loaded; using tag-parsed names: %s",
1766 log_label,
1767 len(parsed_names),
1768 len(mbids),
1769 parsed_names,
1770 )
1771 return _from_tags()
1772
1773 mb_results = await mb_provider.resolve_artists_from_mbids(mbids)
1774 # counts disagree, so positional fallback to a tag name is unreliable;
1775 # drop any MBID whose lookup failed (already logged per-MBID)
1776 resolved: list[tuple[str, str | None, str | None]] = [
1777 mb_result for mb_result in mb_results if mb_result is not None
1778 ]
1779 if not resolved:
1780 self.logger.warning(
1781 "%s count (%d) didn't match MBID count (%d) and every MusicBrainz "
1782 "lookup failed; falling back to tag-parsed names: %s",
1783 log_label,
1784 len(parsed_names),
1785 len(mbids),
1786 parsed_names,
1787 )
1788 return _from_tags()
1789 self.logger.info(
1790 "%s count (%d) didn't match MBID count (%d); resolved canonical names "
1791 "via MusicBrainz: %s",
1792 log_label,
1793 len(parsed_names),
1794 len(mbids),
1795 [r[0] for r in resolved],
1796 )
1797 return resolved
1798
1799 def _match_album_artist(
1800 self, album: Album | None, name: str, mbid: str | None
1801 ) -> Artist | ItemMapping | None:
1802 """
1803 Return an existing album artist representing the same artist, if any.
1804
1805 Matches on MusicBrainz ID when available (names may differ when only one
1806 side was resolved against MusicBrainz), otherwise on exact name.
1807
1808 :param album: The track's album, if known.
1809 :param name: Resolved track-artist name.
1810 :param mbid: Resolved track-artist MusicBrainz ID, if any.
1811 """
1812 if not album:
1813 return None
1814 return next(
1815 (x for x in album.artists if (mbid and x.mbid == mbid) or x.name == name),
1816 None,
1817 )
1818
1819 async def _parse_artist(
1820 self,
1821 name: str,
1822 album_dir: str | None = None,
1823 sort_name: str | None = None,
1824 mbid: str | None = None,
1825 artist_path: str | None = None,
1826 ) -> Artist:
1827 """Parse full (album) Artist."""
1828 if not artist_path:
1829 # we need to hunt for the artist (metadata) path on disk
1830 # this can either be relative to the album path or at root level
1831 # check if we have an artist folder for this artist at root level
1832 safe_artist_name = create_safe_string(name, lowercase=False, replace_space=False)
1833 if await self.exists(name):
1834 artist_path = name
1835 elif await self.exists(safe_artist_name):
1836 artist_path = safe_artist_name
1837 elif album_dir and (foldermatch := get_artist_dir(name, album_dir=album_dir)):
1838 # try to find (album)artist folder based on album path
1839 artist_path = foldermatch
1840 else:
1841 # check if we have an existing item to retrieve the artist path
1842 async for item in self.mass.music.artists.iter_library_items(
1843 search=name, provider=self.instance_id
1844 ):
1845 if not compare_strings(name, item.name):
1846 continue
1847 for prov_mapping in item.provider_mappings:
1848 if prov_mapping.provider_instance != self.instance_id:
1849 continue
1850 if prov_mapping.url:
1851 artist_path = prov_mapping.url
1852 break
1853 if artist_path:
1854 break
1855
1856 # prefer (short lived) cache for a bit more speed
1857 if artist_path and (
1858 cache := await self.cache.get(
1859 key=artist_path,
1860 provider=self.instance_id,
1861 category=CACHE_CATEGORY_ARTIST_INFO,
1862 base_class=Artist,
1863 )
1864 ):
1865 return cache # type: ignore[no-any-return]
1866
1867 prov_artist_id = artist_path or name
1868 artist = Artist(
1869 item_id=prov_artist_id,
1870 provider=self.instance_id,
1871 name=name,
1872 sort_name=sort_name,
1873 provider_mappings={
1874 ProviderMapping(
1875 item_id=prov_artist_id,
1876 provider_domain=self.domain,
1877 provider_instance=self.instance_id,
1878 url=artist_path,
1879 in_library=True,
1880 )
1881 },
1882 )
1883 if mbid := clean_mbid(mbid, f"tags of artist {name}"):
1884 artist.mbid = mbid
1885 if not artist_path or not await self.exists(artist_path):
1886 return artist
1887
1888 # grab additional metadata within the Artist's folder
1889 nfo_file = os.path.join(artist_path, "artist.nfo")
1890 if await self.exists(nfo_file):
1891 # found NFO file with metadata
1892 # https://kodi.wiki/view/NFO_files/Artists
1893 try:
1894 data = (await self._read_file(nfo_file)).decode("utf-8")
1895 info = await asyncio.to_thread(xmltodict.parse, data)
1896 info = info["artist"]
1897 artist.name = info.get("title", info.get("name", name))
1898 if sort_name := info.get("sortname"):
1899 artist.sort_name = sort_name
1900 if mbid := clean_mbid(info.get("musicbrainzartistid"), nfo_file):
1901 artist.mbid = mbid
1902 if description := info.get("biography"):
1903 artist.metadata.description = description
1904 if genre := info.get("genre"):
1905 artist.metadata.genres = set(split_items(genre))
1906 except (ExpatError, KeyError) as err:
1907 self.logger.warning(
1908 "Failed to parse artist NFO file %s: %s",
1909 nfo_file,
1910 str(err),
1911 )
1912 # find local images
1913 if images := await self._get_local_images(artist_path, extra_thumb_names=("artist",)):
1914 artist.metadata.images = UniqueList(images)
1915
1916 await self.cache.set(
1917 key=artist_path,
1918 data=artist.to_dict(),
1919 provider=self.instance_id,
1920 category=CACHE_CATEGORY_ARTIST_INFO,
1921 expiration=120,
1922 )
1923
1924 return artist
1925
1926 async def _parse_audiobook(self, file_item: FileSystemItem, tags: AudioTags) -> Audiobook:
1927 """
1928 Parse Audiobook details from file tags.
1929
1930 Audiobooks can be single files with embedded chapters or multiple files per folder.
1931 Only the first file (by track number or alphabetically) is processed as the audiobook.
1932 """
1933 # Skip files that aren't the first chapter.
1934 # A file carrying its own embedded chapter markers is a standalone audiobook,
1935 # so it should never be treated as a chapter file of another book.
1936 track_tag = tags.tags.get("track")
1937 if track_tag:
1938 track_num = try_parse_int(str(track_tag).split("/")[0], None)
1939 if track_num and track_num > 1 and not tags.chapters:
1940 raise IsChapterFile
1941 elif not tags.chapters:
1942 # No track tag and no embedded chapters -
1943 # assume part of a multi-file audiobook, only process the first file alphabetically
1944 items = await self._scandir(file_item.relative_parent_path)
1945 # Sort by filename for alphabetical ordering
1946 items.sort(key=lambda x: x.filename.lower())
1947 for item in items:
1948 if item.is_dir or item.ext not in AUDIOBOOK_EXTENSIONS:
1949 continue
1950 if item.absolute_path != file_item.absolute_path:
1951 raise IsChapterFile
1952 break
1953
1954 # For multi-file audiobooks, album tag is the book name, title is the chapter name
1955 if tags.album:
1956 book_name = tags.album
1957 sort_name = tags.album_sort
1958 elif (title := tags.tags.get("title")) and tags.track is None:
1959 book_name = title
1960 sort_name = tags.title_sort
1961 else:
1962 # file(s) without tags, use foldername
1963 book_name = file_item.parent_name
1964 sort_name = None
1965
1966 # collect all chapters
1967 total_duration, chapters = await self._get_chapters_for_audiobook(file_item, tags)
1968
1969 audio_book = Audiobook(
1970 item_id=file_item.relative_path,
1971 provider=self.instance_id,
1972 name=book_name,
1973 sort_name=sort_name,
1974 version=tags.version,
1975 duration=total_duration or int(tags.duration or 0),
1976 provider_mappings={
1977 ProviderMapping(
1978 item_id=file_item.relative_path,
1979 provider_domain=self.domain,
1980 provider_instance=self.instance_id,
1981 audio_format=AudioFormat(
1982 content_type=ContentType.try_parse(file_item.ext or tags.format),
1983 codec_type=self._codec_type_from_tags(tags),
1984 sample_rate=tags.sample_rate,
1985 bit_depth=tags.bits_per_sample,
1986 channels=tags.channels,
1987 bit_rate=tags.bit_rate,
1988 ),
1989 details=file_item.checksum,
1990 in_library=True,
1991 )
1992 },
1993 )
1994 audio_book.metadata.chapters = chapters
1995
1996 # handle embedded cover image
1997 if tags.has_cover_image:
1998 # we do not actually embed the image in the metadata because that would consume too
1999 # much space and bandwidth. Instead we set the filename as value so the image can
2000 # be retrieved later in realtime.
2001 audio_book.metadata.add_image(
2002 MediaItemImage(
2003 type=ImageType.THUMB,
2004 path=self._versioned_image_path(file_item.relative_path, file_item.checksum),
2005 provider=self.instance_id,
2006 remotely_accessible=False,
2007 )
2008 )
2009
2010 # parse other info
2011 audio_book.authors.set(tags.writers or tags.album_artists or tags.artists)
2012 audio_book.metadata.genres = (
2013 set(tags.genres) if tags.genres else {DEFAULT_AUDIOBOOK_PODCAST_GENRE}
2014 )
2015 audio_book.metadata.copyright = tags.get("copyright")
2016 audio_book.metadata.lyrics = tags.lyrics
2017 audio_book.metadata.description = tags.get("comment")
2018 explicit_tag = tags.get("itunesadvisory")
2019 if explicit_tag is not None:
2020 audio_book.metadata.explicit = explicit_tag == "1"
2021 if recording_mbid := clean_mbid(tags.musicbrainz_recordingid, tags.filename):
2022 audio_book.mbid = recording_mbid
2023
2024 # try to fetch additional metadata from the folder
2025 if not audio_book.image or not audio_book.metadata.description:
2026 # try to get an image by traversing files in the same folder
2027 for _item in await self._scandir(file_item.relative_parent_path):
2028 if "." not in _item.relative_path or _item.is_dir:
2029 continue
2030 if _item.ext in IMAGE_EXTENSIONS and not audio_book.image:
2031 audio_book.metadata.add_image(
2032 MediaItemImage(
2033 type=ImageType.THUMB,
2034 path=self._versioned_image_path(_item.relative_path, _item.checksum),
2035 provider=self.instance_id,
2036 remotely_accessible=False,
2037 )
2038 )
2039 if _item.ext == "txt" and not audio_book.metadata.description:
2040 # try to parse a description from a text file
2041 try:
2042 raw = await self._read_file(_item.relative_path)
2043 audio_book.metadata.description = raw.decode("utf-8")
2044 except Exception as err:
2045 self.logger.warning(
2046 "Could not read description from file %s: %s",
2047 _item.relative_path,
2048 str(err),
2049 )
2050
2051 # handle (optional) loudness measurement tag(s)
2052 if tags.track_loudness is not None:
2053 self.mass.create_task(
2054 self.mass.streams.audio_analysis.set_track_loudness(
2055 audio_book.item_id,
2056 self.instance_id,
2057 tags.track_loudness,
2058 tags.track_album_loudness,
2059 media_type=MediaType.AUDIOBOOK,
2060 )
2061 )
2062 return audio_book
2063
2064 async def _parse_podcast_episode(
2065 self, file_item: FileSystemItem, tags: AudioTags
2066 ) -> PodcastEpisode:
2067 """Parse full PodcastEpisode details from file tags."""
2068 # ruff: noqa: PLR0915
2069 podcast_name = tags.album or file_item.parent_name
2070 podcast_path = file_item.relative_parent_path
2071 episode = PodcastEpisode(
2072 item_id=file_item.relative_path,
2073 provider=self.instance_id,
2074 name=tags.title,
2075 sort_name=tags.title_sort,
2076 provider_mappings={
2077 ProviderMapping(
2078 item_id=file_item.relative_path,
2079 provider_domain=self.domain,
2080 provider_instance=self.instance_id,
2081 audio_format=AudioFormat(
2082 content_type=ContentType.try_parse(file_item.ext or tags.format),
2083 codec_type=self._codec_type_from_tags(tags),
2084 sample_rate=tags.sample_rate,
2085 bit_depth=tags.bits_per_sample,
2086 channels=tags.channels,
2087 bit_rate=tags.bit_rate,
2088 ),
2089 details=file_item.checksum,
2090 in_library=True,
2091 )
2092 },
2093 position=tags.track or 0,
2094 duration=try_parse_int(tags.duration) or 0,
2095 podcast=Podcast(
2096 item_id=podcast_path,
2097 provider=self.instance_id,
2098 name=podcast_name,
2099 sort_name=tags.album_sort,
2100 publisher=tags.tags.get("publisher"),
2101 provider_mappings={
2102 ProviderMapping(
2103 item_id=podcast_path,
2104 provider_domain=self.domain,
2105 provider_instance=self.instance_id,
2106 in_library=True,
2107 )
2108 },
2109 ),
2110 )
2111 # handle embedded cover image
2112 if tags.has_cover_image:
2113 # we do not actually embed the image in the metadata because that would consume too
2114 # much space and bandwidth. Instead we set the filename as value so the image can
2115 # be retrieved later in realtime.
2116 episode.metadata.add_image(
2117 MediaItemImage(
2118 type=ImageType.THUMB,
2119 path=file_item.relative_path,
2120 provider=self.instance_id,
2121 remotely_accessible=False,
2122 )
2123 )
2124 # parse other info
2125 episode.metadata.genres = (
2126 set(tags.genres) if tags.genres else {DEFAULT_AUDIOBOOK_PODCAST_GENRE}
2127 )
2128 episode.metadata.copyright = tags.get("copyright")
2129 episode.metadata.lyrics = tags.lyrics
2130 episode.metadata.description = tags.get("comment")
2131 explicit_tag = tags.get("itunesadvisory")
2132 if explicit_tag is not None:
2133 episode.metadata.explicit = explicit_tag == "1"
2134
2135 # handle (optional) chapters
2136 if tags.chapters:
2137 episode.metadata.chapters = [
2138 MediaItemChapter(
2139 position=chapter.chapter_id,
2140 name=chapter.title or f"Chapter {chapter.chapter_id}",
2141 start=chapter.position_start,
2142 end=chapter.position_end,
2143 )
2144 for chapter in tags.chapters
2145 ]
2146
2147 # try to fetch additional Podcast metadata from the folder
2148 assert isinstance(episode.podcast, Podcast)
2149 if images := await self._get_local_images(file_item.relative_parent_path):
2150 episode.podcast.metadata.images = images
2151 if metadata := await self._get_podcast_metadata(file_item.relative_parent_path):
2152 if title := metadata.get("title"):
2153 episode.podcast.name = title
2154 if sort_name := metadata.get("sorttitle"):
2155 episode.podcast.sort_name = sort_name
2156 if description := metadata.get("description"):
2157 episode.podcast.metadata.description = description
2158 if genres := metadata.get("genres"):
2159 episode.podcast.metadata.genres = set(genres)
2160 if publisher := metadata.get("publisher"):
2161 episode.podcast.publisher = publisher
2162 if image := metadata.get("imageURL"):
2163 episode.podcast.metadata.add_image(
2164 MediaItemImage(
2165 type=ImageType.THUMB,
2166 path=image,
2167 provider=self.instance_id,
2168 remotely_accessible=True,
2169 )
2170 )
2171 # copy (embedded) image from episode (or vice versa)
2172 if not episode.podcast.image and episode.image:
2173 episode.podcast.metadata.add_image(episode.image)
2174 elif not episode.image and episode.podcast.image:
2175 episode.metadata.add_image(episode.podcast.image)
2176 # ensure podcast has a default genre if none set
2177 if not episode.podcast.metadata.genres:
2178 episode.podcast.metadata.genres = {DEFAULT_AUDIOBOOK_PODCAST_GENRE}
2179
2180 # handle (optional) loudness measurement tag(s)
2181 if tags.track_loudness is not None:
2182 self.mass.create_task(
2183 self.mass.streams.audio_analysis.set_track_loudness(
2184 episode.item_id,
2185 self.instance_id,
2186 tags.track_loudness,
2187 tags.track_album_loudness,
2188 media_type=MediaType.PODCAST_EPISODE,
2189 )
2190 )
2191 return episode
2192
2193 async def _parse_sound_effect(self, file_item: FileSystemItem, tags: AudioTags) -> SoundEffect:
2194 """Parse full sound effect details from file tags."""
2195 sound_effect = SoundEffect(
2196 item_id=file_item.relative_path,
2197 provider=self.instance_id,
2198 name=tags.title,
2199 sort_name=tags.title_sort,
2200 duration=int(tags.duration or 0),
2201 provider_mappings={
2202 ProviderMapping(
2203 item_id=file_item.relative_path,
2204 provider_domain=self.domain,
2205 provider_instance=self.instance_id,
2206 audio_format=AudioFormat(
2207 content_type=ContentType.try_parse(file_item.ext or tags.format),
2208 codec_type=self._codec_type_from_tags(tags),
2209 sample_rate=tags.sample_rate,
2210 bit_depth=tags.bits_per_sample,
2211 channels=tags.channels,
2212 bit_rate=tags.bit_rate,
2213 ),
2214 details=file_item.checksum,
2215 in_library=True,
2216 )
2217 },
2218 )
2219 sound_effect.metadata.description = tags.get("comment")
2220 # handle embedded cover image
2221 if tags.has_cover_image:
2222 # we do not actually embed the image in the metadata because that would consume too
2223 # much space and bandwidth. Instead we set the filename as value so the image can
2224 # be retrieved later in realtime.
2225 sound_effect.metadata.add_image(
2226 MediaItemImage(
2227 type=ImageType.THUMB,
2228 path=file_item.relative_path,
2229 provider=self.instance_id,
2230 remotely_accessible=False,
2231 )
2232 )
2233 return sound_effect
2234
2235 async def _get_or_parse_sound_effect(self, file_item: FileSystemItem) -> SoundEffect:
2236 """Return the (cached) SoundEffect for the given file, parsing tags when needed."""
2237 cache_key = f"sound_effect.{file_item.relative_path}"
2238 cached_data: SoundEffect | None = await self.cache.get(
2239 cache_key,
2240 provider=self.instance_id,
2241 checksum=file_item.checksum,
2242 category=CACHE_CATEGORY_SOUND_EFFECTS,
2243 base_class=SoundEffect,
2244 )
2245 if cached_data is not None:
2246 return cached_data
2247 tags = await async_parse_tags(file_item.absolute_path, file_item.file_size)
2248 sound_effect = await self._parse_sound_effect(file_item, tags)
2249 await self.cache.set(
2250 cache_key,
2251 sound_effect.to_dict(),
2252 expiration=3600 * 24 * 365, # File timestamp checksum handles invalidation
2253 provider=self.instance_id,
2254 checksum=file_item.checksum,
2255 category=CACHE_CATEGORY_SOUND_EFFECTS,
2256 )
2257 return sound_effect
2258
2259 async def _parse_album(
2260 self, track_path: str, track_tags: AudioTags, track_created_at: int | None = None
2261 ) -> Album:
2262 """
2263 Parse Album metadata from Track tags.
2264
2265 :param track_path: Path to the track file.
2266 :param track_tags: Audio tags from the track.
2267 :param track_created_at: Creation timestamp of the track file (Unix epoch).
2268 """
2269 assert track_tags.album
2270 # work out if we have an album and/or disc folder
2271 # track_dir is the folder level where the tracks are located
2272 # this may be a separate disc folder (Disc 1, Disc 2 etc) underneath the album folder
2273 # or this is an album folder with the disc attached
2274 track_dir = os.path.dirname(track_path)
2275 album_dir = get_album_dir(track_dir, track_tags.album)
2276
2277 if album_dir and (
2278 cache := await self.cache.get(
2279 key=album_dir,
2280 provider=self.instance_id,
2281 category=CACHE_CATEGORY_ALBUM_INFO,
2282 base_class=Album,
2283 )
2284 ):
2285 return cache # type: ignore[no-any-return]
2286
2287 # album artist(s)
2288 album_artists: UniqueList[Artist | ItemMapping] = UniqueList()
2289 if track_tags.album_artists:
2290 resolved_album_artists = await self._resolve_artists_with_mbids(
2291 track_tags.album_artists,
2292 track_tags.musicbrainz_albumartistids,
2293 track_tags.album_artist_sort_names,
2294 log_label="ALBUMARTIST tag",
2295 )
2296 for name, mbid, sort_name in resolved_album_artists:
2297 artist = await self._parse_artist(
2298 name, album_dir=album_dir, sort_name=sort_name, mbid=mbid
2299 )
2300 album_artists.append(artist)
2301 else:
2302 # album artist tag is missing, determine fallback
2303 fallback_action = self.config.get_value(CONF_ENTRY_MISSING_ALBUM_ARTIST.key)
2304 if fallback_action == "folder_name" and album_dir:
2305 possible_artist_folder = os.path.dirname(album_dir)
2306 self.logger.warning(
2307 "%s is missing ID3 tag [albumartist], using foldername %s as fallback",
2308 track_path,
2309 possible_artist_folder,
2310 )
2311 album_artist_str = Path(possible_artist_folder).name
2312 album_artists = UniqueList(
2313 [await self._parse_artist(name=album_artist_str, album_dir=album_dir)]
2314 )
2315 # fallback to track artists (if defined by user)
2316 elif fallback_action == "track_artist":
2317 self.logger.warning(
2318 "%s is missing ID3 tag [albumartist], using track artist(s) as fallback",
2319 track_path,
2320 )
2321 album_artists = UniqueList(
2322 [
2323 await self._parse_artist(name=track_artist_str, album_dir=album_dir)
2324 for track_artist_str in track_tags.artists
2325 ]
2326 )
2327 # all other: fallback to various artists
2328 else:
2329 self.logger.warning(
2330 "%s is missing ID3 tag [albumartist], using %s as fallback",
2331 track_path,
2332 VARIOUS_ARTISTS_NAME,
2333 )
2334 album_artists = UniqueList(
2335 [await self._parse_artist(name=VARIOUS_ARTISTS_NAME, mbid=VARIOUS_ARTISTS_MBID)]
2336 )
2337
2338 if album_dir: # noqa: SIM108
2339 # prefer the path as id
2340 item_id = album_dir
2341 else:
2342 # create fake item_id based on artist + album
2343 item_id = album_artists[0].name + os.sep + track_tags.album
2344
2345 name, version = parse_title_and_version(track_tags.album)
2346 album = Album(
2347 item_id=item_id,
2348 provider=self.instance_id,
2349 name=name,
2350 version=version,
2351 sort_name=track_tags.album_sort,
2352 artists=album_artists,
2353 provider_mappings={
2354 ProviderMapping(
2355 item_id=item_id,
2356 provider_domain=self.domain,
2357 provider_instance=self.instance_id,
2358 url=album_dir,
2359 in_library=True,
2360 )
2361 },
2362 date_added=(
2363 datetime.fromtimestamp(track_created_at, tz=UTC) if track_created_at else None
2364 ),
2365 )
2366 if track_tags.barcode:
2367 album.external_ids.add((ExternalID.BARCODE, track_tags.barcode))
2368
2369 if album_mbid := clean_mbid(track_tags.musicbrainz_albumid, track_tags.filename):
2370 album.mbid = album_mbid
2371 if releasegroup_mbid := clean_mbid(
2372 track_tags.musicbrainz_releasegroupid, track_tags.filename
2373 ):
2374 album.add_external_id(ExternalID.MB_RELEASEGROUP, releasegroup_mbid)
2375 if track_tags.year:
2376 album.year = track_tags.year
2377 album.album_type = track_tags.album_type
2378
2379 # hunt for additional metadata and images in the folder structure
2380 if not album_dir:
2381 return album
2382
2383 for folder_path in (track_dir, album_dir):
2384 if not folder_path or not await self.exists(folder_path):
2385 continue
2386 nfo_file = os.path.join(folder_path, "album.nfo")
2387 if await self.exists(nfo_file):
2388 # found NFO file with metadata
2389 # https://kodi.wiki/view/NFO_files/Artists
2390 try:
2391 data = (await self._read_file(nfo_file)).decode("utf-8")
2392 info = await asyncio.to_thread(xmltodict.parse, data)
2393 parse_album_nfo(album, info["album"], nfo_file)
2394 except (ExpatError, KeyError) as err:
2395 self.logger.warning(
2396 "Failed to parse album NFO file %s: %s",
2397 nfo_file,
2398 str(err),
2399 )
2400
2401 # find local images
2402 if images := await self._get_local_images(folder_path, extra_thumb_names=("album",)):
2403 if album.metadata.images is None:
2404 album.metadata.images = UniqueList(images)
2405 else:
2406 album.metadata.images += images
2407
2408 await self.cache.set(
2409 key=album_dir,
2410 data=album.to_dict(),
2411 provider=self.instance_id,
2412 category=CACHE_CATEGORY_ALBUM_INFO,
2413 expiration=120,
2414 )
2415 return album
2416
2417 async def _get_local_images(
2418 self, folder: str, extra_thumb_names: tuple[str, ...] | None = None
2419 ) -> UniqueList[MediaItemImage]:
2420 """Return local images found in a given folderpath."""
2421 if (
2422 cache := await self.cache.get(
2423 key=folder,
2424 provider=self.instance_id,
2425 category=CACHE_CATEGORY_FOLDER_IMAGES,
2426 base_class=MediaItemImage,
2427 )
2428 ) is not None:
2429 return UniqueList(cache)
2430 if extra_thumb_names is None:
2431 extra_thumb_names = ()
2432 images: UniqueList[MediaItemImage] = UniqueList()
2433 folder_files = await self._scandir(folder)
2434 for item in folder_files:
2435 if "." not in item.relative_path or item.is_dir or not item.ext:
2436 continue
2437 if item.ext.lower() not in IMAGE_EXTENSIONS:
2438 continue
2439 # try match on filename = one of our imagetypes
2440 if item.name.lower() in ImageType:
2441 images.append(
2442 MediaItemImage(
2443 type=ImageType(item.name),
2444 path=item.relative_path,
2445 provider=self.instance_id,
2446 remotely_accessible=False,
2447 )
2448 )
2449
2450 # try alternative names for thumbs
2451 extra_thumb_names = ("folder", "cover", *extra_thumb_names)
2452 for item in folder_files:
2453 if "." not in item.relative_path or item.is_dir or not item.ext:
2454 continue
2455 if item.ext.lower() not in IMAGE_EXTENSIONS:
2456 continue
2457 if item.name.lower() not in extra_thumb_names:
2458 continue
2459 images.append(
2460 MediaItemImage(
2461 type=ImageType.THUMB,
2462 path=item.relative_path,
2463 provider=self.instance_id,
2464 remotely_accessible=False,
2465 )
2466 )
2467
2468 await self.cache.set(
2469 key=folder,
2470 data=[img.to_dict() for img in images],
2471 provider=self.instance_id,
2472 category=CACHE_CATEGORY_FOLDER_IMAGES,
2473 expiration=120,
2474 )
2475 return images
2476
2477 async def _get_stream_details_for_track(self, item_id: str) -> StreamDetails:
2478 """Return the streamdetails for a track/song."""
2479 if parse_cue_track_id(item_id) is not None:
2480 return await self._cue.get_stream_details(item_id)
2481
2482 library_item = await self.mass.music.tracks.get_library_item_by_prov_id(
2483 item_id, self.instance_id
2484 )
2485 if library_item is None:
2486 # this could be a file that has just been added, try parsing it
2487 file_item = await self.resolve(item_id)
2488 tags = await async_parse_tags(file_item.absolute_path, file_item.file_size)
2489 if not (library_item := await self._parse_track(file_item, tags)):
2490 msg = f"Item not found: {item_id}"
2491 raise MediaNotFoundError(msg)
2492
2493 prov_mapping = next(x for x in library_item.provider_mappings if x.item_id == item_id)
2494 file_item = await self.resolve(item_id)
2495
2496 return StreamDetails(
2497 provider=self.instance_id,
2498 item_id=item_id,
2499 audio_format=prov_mapping.audio_format,
2500 media_type=MediaType.TRACK,
2501 stream_type=StreamType.LOCAL_FILE,
2502 duration=library_item.duration,
2503 size=file_item.file_size,
2504 data=file_item,
2505 path=file_item.absolute_path,
2506 can_seek=True,
2507 allow_seek=True,
2508 )
2509
2510 async def _get_stream_details_for_podcast_episode(self, item_id: str) -> StreamDetails:
2511 """Return the streamdetails for a podcast episode."""
2512 # podcasts episodes are never stored in the library so we need to parse the file
2513 file_item = await self.resolve(item_id)
2514 tags = await async_parse_tags(file_item.absolute_path, file_item.file_size)
2515 return StreamDetails(
2516 provider=self.instance_id,
2517 item_id=item_id,
2518 audio_format=AudioFormat(
2519 content_type=ContentType.try_parse(file_item.ext or tags.format),
2520 codec_type=self._codec_type_from_tags(tags),
2521 sample_rate=tags.sample_rate,
2522 bit_depth=tags.bits_per_sample,
2523 channels=tags.channels,
2524 bit_rate=tags.bit_rate,
2525 ),
2526 media_type=MediaType.PODCAST_EPISODE,
2527 stream_type=StreamType.LOCAL_FILE,
2528 duration=try_parse_int(tags.duration or 0),
2529 size=file_item.file_size,
2530 data=file_item,
2531 path=file_item.absolute_path,
2532 allow_seek=True,
2533 can_seek=True,
2534 )
2535
2536 async def _get_stream_details_for_sound_effect(self, item_id: str) -> StreamDetails:
2537 """Return the streamdetails for a sound effect."""
2538 # sound effects are never stored in the library so we parse the file,
2539 # served from cache unless the file changed on disk
2540 file_item = await self.resolve(item_id)
2541 sound_effect = await self._get_or_parse_sound_effect(file_item)
2542 prov_mapping = next(x for x in sound_effect.provider_mappings if x.item_id == item_id)
2543 return StreamDetails(
2544 provider=self.instance_id,
2545 item_id=item_id,
2546 audio_format=prov_mapping.audio_format,
2547 media_type=MediaType.SOUND_EFFECT,
2548 stream_type=StreamType.LOCAL_FILE,
2549 duration=sound_effect.duration,
2550 size=file_item.file_size,
2551 data=file_item,
2552 path=file_item.absolute_path,
2553 allow_seek=True,
2554 can_seek=True,
2555 )
2556
2557 async def _get_stream_details_for_audiobook(self, item_id: str) -> StreamDetails:
2558 """Return the streamdetails for an audiobook."""
2559 library_item = await self.mass.music.audiobooks.get_library_item_by_prov_id(
2560 item_id, self.instance_id
2561 )
2562 if library_item is None:
2563 # this could be a file that has just been added, try parsing it
2564 file_item = await self.resolve(item_id)
2565 tags = await async_parse_tags(file_item.absolute_path, file_item.file_size)
2566 if not (library_item := await self._parse_audiobook(file_item, tags)):
2567 msg = f"Item not found: {item_id}"
2568 raise MediaNotFoundError(msg)
2569
2570 prov_mapping = next(x for x in library_item.provider_mappings if x.item_id == item_id)
2571 file_item = await self.resolve(item_id)
2572 duration = library_item.duration
2573 file_based_chapters: list[tuple[str, float]] | None = await self.cache.get(
2574 key=file_item.relative_path,
2575 provider=self.instance_id,
2576 category=CACHE_CATEGORY_AUDIOBOOK_CHAPTERS,
2577 )
2578 if file_based_chapters is None:
2579 # no cache available for this audiobook, we need to parse the chapters
2580 tags = await async_parse_tags(file_item.absolute_path, file_item.file_size)
2581 await self._parse_audiobook(file_item, tags)
2582 file_based_chapters = await self.cache.get(
2583 key=file_item.relative_path,
2584 provider=self.instance_id,
2585 category=CACHE_CATEGORY_AUDIOBOOK_CHAPTERS,
2586 )
2587
2588 if file_based_chapters:
2589 # this is a multi-file audiobook
2590 return StreamDetails(
2591 provider=self.instance_id,
2592 item_id=item_id,
2593 audio_format=prov_mapping.audio_format,
2594 media_type=MediaType.AUDIOBOOK,
2595 stream_type=StreamType.LOCAL_FILE,
2596 duration=duration,
2597 path=[
2598 MultiPartPath(
2599 path=self._get_chapter_path(chapter_path),
2600 duration=chapter_duration,
2601 )
2602 for chapter_path, chapter_duration in file_based_chapters
2603 ],
2604 allow_seek=True,
2605 )
2606
2607 # regular single-file streaming, simply let ffmpeg deal with the file directly
2608 return StreamDetails(
2609 provider=self.instance_id,
2610 item_id=item_id,
2611 audio_format=prov_mapping.audio_format,
2612 media_type=MediaType.AUDIOBOOK,
2613 stream_type=StreamType.LOCAL_FILE,
2614 duration=library_item.duration,
2615 size=file_item.file_size,
2616 data=file_item,
2617 path=file_item.absolute_path,
2618 allow_seek=True,
2619 can_seek=True,
2620 )
2621
2622 def _get_chapter_path(self, relative_path: str) -> str:
2623 """Return absolute path for a chapter file. Override for network storage."""
2624 return self.get_absolute_path(relative_path)
2625
2626 async def _get_chapters_for_audiobook(
2627 self, audiobook_file_item: FileSystemItem, tags: AudioTags
2628 ) -> tuple[int, list[MediaItemChapter]]:
2629 """
2630 Return chapters for an audiobook.
2631
2632 Chapter sources in order of preference:
2633 1. Multiple files with track tags - sorted by track number
2634 2. Single file with embedded chapters - use embedded chapter markers
2635 3. Multiple files without track tags - sorted alphabetically (fallback)
2636 """
2637 chapters: list[MediaItemChapter] = []
2638 all_chapter_files: list[tuple[str, float]] = []
2639 total_duration = 0.0
2640
2641 # Scan folder for chapter files, separating tagged from untagged
2642 chapter_file_items: list[tuple[FileSystemItem, AudioTags]] = []
2643 untagged_file_items: list[tuple[FileSystemItem, AudioTags]] = []
2644
2645 items = await self._scandir(audiobook_file_item.relative_parent_path)
2646 # Sort by filename for consistent alphabetical ordering
2647 items.sort(key=lambda x: x.filename.lower())
2648
2649 for item in items:
2650 if "." not in item.relative_path or item.is_dir:
2651 continue
2652 if item.ext not in AUDIOBOOK_EXTENSIONS:
2653 continue
2654 item_tags = await async_parse_tags(item.absolute_path, item.file_size)
2655 if not (tags.album == item_tags.album or (item_tags.tags.get("title") is None)):
2656 continue
2657 if item_tags.tags.get("track") is None:
2658 untagged_file_items.append((item, item_tags))
2659 else:
2660 chapter_file_items.append((item, item_tags))
2661
2662 # Determine chapter source
2663 use_embedded = False
2664 use_alphabetical = False
2665
2666 if len(chapter_file_items) > 1:
2667 chapter_file_items.sort(key=lambda x: (x[1].disc or 0, x[1].track or 0))
2668 elif len(chapter_file_items) <= 1 and tags.chapters:
2669 use_embedded = True
2670 elif len(untagged_file_items) > 1:
2671 use_alphabetical = True
2672 chapter_file_items = untagged_file_items
2673 self.logger.info(
2674 "Audiobook files have no track tags, using alphabetical order: %s",
2675 tags.album,
2676 )
2677
2678 if use_embedded:
2679 chapters = [
2680 MediaItemChapter(
2681 position=chapter.chapter_id,
2682 name=chapter.title or f"Chapter {chapter.chapter_id}",
2683 start=chapter.position_start,
2684 end=chapter.position_end,
2685 )
2686 for chapter in tags.chapters
2687 ]
2688 total_duration = try_parse_int(tags.duration) or 0
2689 self.logger.log(
2690 VERBOSE_LOG_LEVEL,
2691 "Audiobook '%s': %d embedded chapters, duration=%d",
2692 tags.album,
2693 len(chapters),
2694 int(total_duration),
2695 )
2696 else:
2697 for position, (chapter_item, chapter_tags) in enumerate(chapter_file_items, start=1):
2698 if chapter_tags.duration is None:
2699 self.logger.warning(
2700 "Chapter file has no duration, skipping: %s",
2701 chapter_item.relative_path,
2702 )
2703 continue
2704 self.logger.debug("Chapter filename: %s", chapter_item.relative_path)
2705 chapters.append(
2706 MediaItemChapter(
2707 position=position,
2708 name=chapter_tags.title,
2709 start=total_duration,
2710 end=total_duration + chapter_tags.duration,
2711 )
2712 )
2713 all_chapter_files.append(
2714 (
2715 chapter_item.relative_path,
2716 chapter_tags.duration,
2717 )
2718 )
2719 total_duration += chapter_tags.duration
2720 sort_method = "alphabetical" if use_alphabetical else "track"
2721 self.logger.log(
2722 VERBOSE_LOG_LEVEL,
2723 "Audiobook '%s': %d files (%s order), duration=%d",
2724 tags.album,
2725 len(chapters),
2726 sort_method,
2727 int(total_duration),
2728 )
2729 # Cache chapter files for streaming
2730 await self.cache.set(
2731 key=audiobook_file_item.relative_path,
2732 data=all_chapter_files,
2733 provider=self.instance_id,
2734 category=CACHE_CATEGORY_AUDIOBOOK_CHAPTERS,
2735 )
2736 return int(total_duration), chapters
2737
2738 async def _get_podcast_metadata(self, podcast_folder: str) -> dict[str, Any]:
2739 """Return metadata for a podcast."""
2740 if (
2741 cache := await self.cache.get(
2742 key=podcast_folder,
2743 provider=self.instance_id,
2744 category=CACHE_CATEGORY_PODCAST_METADATA,
2745 )
2746 ) is not None:
2747 return cast("dict[str, Any]", cache)
2748 data: dict[str, Any] = {}
2749 metadata_file = os.path.join(podcast_folder, "metadata.json")
2750 if await self.exists(metadata_file):
2751 # found json file with metadata
2752 raw = await self._read_file(metadata_file)
2753 data.update(json_loads(raw.decode("utf-8")))
2754 await self.cache.set(
2755 key=podcast_folder,
2756 data=data,
2757 provider=self.instance_id,
2758 category=CACHE_CATEGORY_PODCAST_METADATA,
2759 )
2760 return data
2761
2762 async def _scandir(self, path: str) -> list[FileSystemItem]:
2763 """List directory contents in natural sort order."""
2764 # raw scandir order depends on the underlying filesystem (e.g. hash order
2765 # on ext4) so sort to make browse and folder playback order deterministic
2766 abs_path = self.get_absolute_path(path)
2767 return await asyncio.to_thread(sorted_scandir, self.base_path, abs_path, sort=True)
2768
2769 async def _read_file(self, path: str) -> bytes:
2770 """Read file contents. Override for network storage."""
2771 async with aiofiles.open(self.get_absolute_path(path), mode="rb") as f:
2772 return cast("bytes", await f.read())
2773