/
/
1"""Filesystem musicprovider support for MusicAssistant."""
2
3from __future__ import annotations
4
5import asyncio
6import contextlib
7import logging
8import os
9import os.path
10import posixpath
11import urllib.parse
12from collections.abc import AsyncGenerator, Sequence
13from datetime import UTC, datetime
14from pathlib import Path
15from typing import TYPE_CHECKING, Any, ClassVar, cast
16from xml.parsers.expat import ExpatError
17
18import aiofiles
19import shortuuid
20import xmltodict
21from aiofiles.os import wrap
22from music_assistant_models.enums import (
23 ContentType,
24 EventType,
25 ExternalID,
26 ImageType,
27 MediaType,
28 ProviderFeature,
29 StreamType,
30)
31from music_assistant_models.errors import (
32 InvalidDataError,
33 MediaNotFoundError,
34 MusicAssistantError,
35 SetupFailedError,
36)
37from music_assistant_models.helpers import create_safe_string
38from music_assistant_models.media_items import (
39 Album,
40 Artist,
41 Audiobook,
42 AudioFormat,
43 BrowseFolder,
44 ItemMapping,
45 MediaItemChapter,
46 MediaItemImage,
47 MediaItemType,
48 Playlist,
49 Podcast,
50 PodcastEpisode,
51 ProviderMapping,
52 SearchResults,
53 SoundEffect,
54 Track,
55 UniqueList,
56 is_track,
57)
58from music_assistant_models.streamdetails import MultiPartPath, StreamDetails
59
60from music_assistant.constants import (
61 CONF_PATH,
62 DB_TABLE_ALBUM_ARTISTS,
63 DB_TABLE_ALBUM_TRACKS,
64 DB_TABLE_ALBUMS,
65 DB_TABLE_ARTISTS,
66 DB_TABLE_PROVIDER_MAPPINGS,
67 DB_TABLE_TRACK_ARTISTS,
68 VARIOUS_ARTISTS_MBID,
69 VARIOUS_ARTISTS_NAME,
70 VERBOSE_LOG_LEVEL,
71)
72from music_assistant.controllers.tasks.context import (
73 report_current_task_failure,
74 update_current_task_progress_from_index,
75 update_current_task_progress_text,
76)
77from music_assistant.helpers import lyrics
78from music_assistant.helpers.compare import compare_strings
79from music_assistant.helpers.json import SerializableType, json_loads
80from music_assistant.helpers.playlists import parse_m3u, parse_pls
81from music_assistant.helpers.tags import AudioTags, async_parse_tags, clean_mbid, split_items
82from music_assistant.helpers.util import (
83 TaskManager,
84 detect_charset,
85 parse_title_and_version,
86 try_parse_int,
87)
88from music_assistant.models.music_provider import MusicProvider
89
90from .constants import (
91 AUDIOBOOK_EXTENSIONS,
92 AVAILABILITY_PROBE_INTERVAL,
93 CACHE_CATEGORY_ALBUM_INFO,
94 CACHE_CATEGORY_ARTIST_INFO,
95 CACHE_CATEGORY_AUDIOBOOK_CHAPTERS,
96 CACHE_CATEGORY_FOLDER_IMAGES,
97 CACHE_CATEGORY_PODCAST_EPISODES,
98 CACHE_CATEGORY_PODCAST_METADATA,
99 CACHE_CATEGORY_SOUND_EFFECTS,
100 CONF_CONTENT_TYPE,
101 CONF_ENTRY_CONTENT_TYPE,
102 CONF_ENTRY_IGNORE_ALBUM_PLAYLISTS,
103 CONF_ENTRY_LIBRARY_SYNC_AUDIOBOOKS,
104 CONF_ENTRY_LIBRARY_SYNC_PLAYLISTS,
105 CONF_ENTRY_LIBRARY_SYNC_PODCASTS,
106 CONF_ENTRY_LIBRARY_SYNC_TRACKS,
107 CONF_ENTRY_MISSING_ALBUM_ARTIST,
108 CONF_ENTRY_PROPAGATE_GENRES,
109 CUE_EXTENSIONS,
110 DEFAULT_AUDIOBOOK_PODCAST_GENRE,
111 IMAGE_EXTENSIONS,
112 PARTIAL_LISTING_CACHE_EXPIRATION,
113 PLAYLIST_EXTENSIONS,
114 PODCAST_EPISODE_EXTENSIONS,
115 SOUND_EFFECT_EXTENSIONS,
116 SUPPORTED_EXTENSIONS,
117 TRACK_EXTENSIONS,
118 IsChapterFile,
119 content_type_config_entry,
120)
121from .cue import (
122 CueSheetHandler,
123 cue_metadata_checksum,
124 cue_referenced_audio_stem,
125 make_cue_track_id,
126 parse_cue_track_id,
127)
128from .helpers import (
129 FileSystemItem,
130 ScanErrors,
131 get_absolute_path,
132 get_album_dir,
133 get_artist_dir,
134 get_folder_signature,
135 get_relative_path,
136 recursive_iter,
137 sorted_scandir,
138)
139from .parsers import parse_album_nfo
140
141if TYPE_CHECKING:
142 from music_assistant_models.config_entries import ConfigEntry, ProviderConfig
143 from music_assistant_models.provider import ProviderManifest
144
145 from music_assistant.mass import MusicAssistant
146 from music_assistant.models import ProviderInstanceType
147 from music_assistant.providers.musicbrainz import MusicbrainzProvider
148
149
150isdir = wrap(os.path.isdir)
151isfile = wrap(os.path.isfile)
152ismount = wrap(os.path.ismount)
153exists = wrap(os.path.exists)
154makedirs = wrap(os.makedirs)
155
156SUPPORTED_FEATURES = {
157 ProviderFeature.BROWSE,
158 ProviderFeature.SEARCH,
159}
160
161
162async def setup(
163 mass: MusicAssistant, manifest: ProviderManifest, config: ProviderConfig
164) -> ProviderInstanceType:
165 """Initialize provider(instance) with given configuration."""
166 return LocalFileSystemProvider(mass, manifest, config)
167
168
169class LocalFileSystemProvider(MusicProvider):
170 """
171 Implementation of a musicprovider for (local) files.
172
173 Reads ID3 tags from file and falls back to parsing filename.
174 Optionally reads metadata from nfo files and images in folder structure <artist>/<album>.
175 Supports m3u files for playlists.
176 """
177
178 # parallel workers per sync; subclasses lower this for slower transports
179 _SYNC_CONCURRENCY: ClassVar[int] = 16
180 _sync_tracks: bool = True
181 _sync_playlists: bool = True
182
183 def __init__(
184 self,
185 mass: MusicAssistant,
186 manifest: ProviderManifest,
187 config: ProviderConfig,
188 base_path: str | None = None,
189 ) -> None:
190 """Initialize MusicProvider."""
191 super().__init__(mass, manifest, config, SUPPORTED_FEATURES)
192 # subclasses (NFS/SMB/...) mount elsewhere and pass their own base_path;
193 # the plain local provider reads its scan directory from the setup data
194 self.base_path: str = (
195 base_path if base_path is not None else cast("str", self.get_setup_value(CONF_PATH))
196 )
197 self.write_access: bool = False
198 self.sync_running: bool = False
199 self.media_content_type = cast(
200 "str", self.get_setup_value(CONF_CONTENT_TYPE, CONF_ENTRY_CONTENT_TYPE.default_value)
201 )
202 self._cue = CueSheetHandler(self)
203
204 async def get_config_entries(self) -> tuple[ConfigEntry, ...]:
205 """Return Config entries to configure this provider."""
206 # content type and path are collected by the setup flow; surface the (immutable)
207 # content type read-only so the sync options' depends_on chains still resolve
208 content_type = str(
209 self.get_setup_value(CONF_CONTENT_TYPE, CONF_ENTRY_CONTENT_TYPE.default_value)
210 )
211 return (
212 content_type_config_entry(content_type),
213 CONF_ENTRY_MISSING_ALBUM_ARTIST,
214 CONF_ENTRY_IGNORE_ALBUM_PLAYLISTS,
215 CONF_ENTRY_LIBRARY_SYNC_TRACKS,
216 CONF_ENTRY_LIBRARY_SYNC_PLAYLISTS,
217 CONF_ENTRY_LIBRARY_SYNC_PODCASTS,
218 CONF_ENTRY_LIBRARY_SYNC_AUDIOBOOKS,
219 CONF_ENTRY_PROPAGATE_GENRES,
220 )
221
222 @property
223 def supported_features(self) -> set[ProviderFeature]:
224 """Return the features supported by this Provider."""
225 base_features = {*SUPPORTED_FEATURES}
226 if self.media_content_type == "audiobooks":
227 return {ProviderFeature.LIBRARY_AUDIOBOOKS, *base_features}
228 if self.media_content_type == "podcasts":
229 return {ProviderFeature.LIBRARY_PODCASTS, *base_features}
230 if self.media_content_type == "sound_effects":
231 # sound effects are live-fetched content, never synced into the library
232 return {ProviderFeature.SOUND_EFFECTS, *base_features}
233 music_features = {
234 ProviderFeature.LIBRARY_ALBUMS,
235 ProviderFeature.LIBRARY_ARTISTS,
236 ProviderFeature.LIBRARY_TRACKS,
237 ProviderFeature.LIBRARY_PLAYLISTS,
238 *base_features,
239 }
240 if self.write_access:
241 music_features.add(ProviderFeature.PLAYLIST_TRACKS_EDIT)
242 music_features.add(ProviderFeature.PLAYLIST_CREATE)
243 return music_features
244
245 @property
246 def is_streaming_provider(self) -> bool:
247 """Return True if the provider is a streaming provider."""
248 return False
249
250 @property
251 def instance_name_postfix(self) -> str | None:
252 """Return a (default) instance name postfix for this provider instance."""
253 return Path(self.base_path).name
254
255 async def handle_async_init(self) -> None:
256 """Handle async initialization of the provider."""
257 if not await isdir(self.base_path):
258 msg = f"Music Directory {self.base_path} does not exist"
259 raise SetupFailedError(
260 msg,
261 translation_key="music_directory_not_found",
262 translation_owner=self.translation_owner,
263 translation_args=[self.base_path],
264 )
265 await self.check_write_access()
266
267 async def unload(self, is_removed: bool = False) -> None:
268 """Handle unload/close of the provider."""
269 self._cancel_availability_probe()
270 # a check that already started runs as a task under the same id, and it would
271 # otherwise keep talking to storage this unload is in the middle of tearing down
272 self.mass.cancel_task(self._availability_probe_id)
273
274 async def get_diagnostics(self) -> dict[str, SerializableType]:
275 """Return diagnostics info for this provider to include in diagnostics reports."""
276 return {
277 "sync_running": self.sync_running,
278 "write_access": self.write_access,
279 "content_type": self.media_content_type,
280 }
281
282 async def search(
283 self,
284 search_query: str,
285 media_types: list[MediaType] | None,
286 limit: int = 5,
287 ) -> SearchResults:
288 """Perform search on this file based musicprovider."""
289 result = SearchResults()
290 # searching the filesystem is slow and unreliable,
291 # so instead we just query the db...
292 if media_types is None or MediaType.TRACK in media_types:
293 result.tracks = await self.mass.music.tracks.get_library_items_by_query(
294 search=search_query, provider_filter=[self.instance_id], limit=limit
295 )
296
297 if media_types is None or MediaType.ALBUM in media_types:
298 result.albums = await self.mass.music.albums.get_library_items_by_query(
299 search=search_query,
300 provider_filter=[self.instance_id],
301 limit=limit,
302 )
303
304 if media_types is None or MediaType.ARTIST in media_types:
305 result.artists = await self.mass.music.artists.get_library_items_by_query(
306 search=search_query,
307 provider_filter=[self.instance_id],
308 limit=limit,
309 )
310 if media_types is None or MediaType.PLAYLIST in media_types:
311 result.playlists = await self.mass.music.playlists.get_library_items_by_query(
312 search=search_query,
313 provider_filter=[self.instance_id],
314 limit=limit,
315 )
316 if media_types is None or MediaType.AUDIOBOOK in media_types:
317 result.audiobooks = await self.mass.music.audiobooks.get_library_items_by_query(
318 search=search_query,
319 provider_filter=[self.instance_id],
320 limit=limit,
321 )
322 if media_types is None or MediaType.PODCAST in media_types:
323 result.podcasts = await self.mass.music.podcasts.get_library_items_by_query(
324 search=search_query,
325 provider_filter=[self.instance_id],
326 limit=limit,
327 )
328 return result
329
330 async def browse(self, path: str) -> Sequence[MediaItemType | ItemMapping | BrowseFolder]:
331 """
332 Browse this provider's items.
333
334 :param path: The path to browse, (e.g. provid://artists).
335 """
336 # for audiobooks and podcasts we just return all library items
337 if self.media_content_type == "podcasts":
338 return await self.mass.music.podcasts.library_items(
339 provider=self.instance_id, summary=False
340 )
341 if self.media_content_type == "audiobooks":
342 return await self.mass.music.audiobooks.library_items(
343 provider=self.instance_id, summary=False
344 )
345 items: list[MediaItemType | ItemMapping | BrowseFolder] = []
346 item_path = path.split("://", 1)[1]
347 if not item_path:
348 item_path = ""
349 scanned = await self._scandir(item_path)
350 # expand CUE sheets into per-track entries and hide the companion audio;
351 # synthetic ids match those minted during sync so get_track resolves them
352 cue_stems: set[str] = set()
353 if self.media_content_type == "music":
354 for item in scanned:
355 if item.ext not in CUE_EXTENSIONS:
356 continue
357 cue_stems.add(item.absolute_path.rsplit(".", 1)[0])
358 try:
359 cue_sheet = await self._cue.load_cue_sheet(item)
360 except InvalidDataError as err:
361 self.logger.warning("Unable to parse CUE sheet %s: %s", item.relative_path, err)
362 continue
363 # also hide the audio file named in the CUE (may differ from its stem)
364 if companion_stem := cue_referenced_audio_stem(item, cue_sheet):
365 cue_stems.add(companion_stem)
366 for cue_track in cue_sheet.tracks:
367 items.append(
368 ItemMapping(
369 media_type=MediaType.TRACK,
370 item_id=make_cue_track_id(item.relative_path, cue_track.number),
371 provider=self.instance_id,
372 name=cue_track.title or f"Track {cue_track.number}",
373 )
374 )
375 for item in scanned:
376 if not item.is_dir and ("." not in item.filename or not item.ext):
377 # skip system files and files without extension
378 continue
379
380 if item.is_dir:
381 items.append(
382 BrowseFolder(
383 item_id=item.relative_path,
384 provider=self.instance_id,
385 path=f"{self.instance_id}://{item.relative_path}",
386 name=item.filename,
387 # mark folder as playable, assuming it contains tracks underneath
388 is_playable=True,
389 )
390 )
391 elif item.ext in TRACK_EXTENSIONS:
392 if item.absolute_path.rsplit(".", 1)[0] in cue_stems:
393 continue
394 items.append(
395 ItemMapping(
396 media_type=(
397 MediaType.SOUND_EFFECT
398 if self.media_content_type == "sound_effects"
399 else MediaType.TRACK
400 ),
401 item_id=item.relative_path,
402 provider=self.instance_id,
403 name=item.filename,
404 )
405 )
406 elif item.ext in PLAYLIST_EXTENSIONS and self.media_content_type == "music":
407 items.append(
408 ItemMapping(
409 media_type=MediaType.PLAYLIST,
410 item_id=item.relative_path,
411 provider=self.instance_id,
412 name=item.filename,
413 )
414 )
415 if self.media_content_type == "music":
416 track_indexes = [
417 index
418 for index, item in enumerate(items)
419 if isinstance(item, ItemMapping) and item.media_type == MediaType.TRACK
420 ]
421 library_tracks = await asyncio.gather(
422 *(
423 self.mass.music.tracks.get_library_item_by_prov_id(
424 items[index].item_id, self.instance_id
425 )
426 for index in track_indexes
427 )
428 )
429 for index, library_track in zip(track_indexes, library_tracks, strict=True):
430 if library_track:
431 items[index] = library_track
432 return items
433
434 async def sync_library(self, media_type: MediaType) -> None:
435 """Run library sync for this provider."""
436 if media_type in (MediaType.ARTIST, MediaType.ALBUM):
437 # artists and albums are synced as part of track sync
438 return
439 if self.media_content_type == "sound_effects":
440 # sound effects are live-fetched content, never synced into the library
441 return
442 # check if any sync options are enabled for this content type
443 # the filesystem provider processes all file types in one scan,
444 # so we can return early if nothing needs syncing
445 if self.media_content_type == "music":
446 self._sync_tracks = bool(self.config.get_value(CONF_ENTRY_LIBRARY_SYNC_TRACKS.key))
447 self._sync_playlists = bool(
448 self.config.get_value(CONF_ENTRY_LIBRARY_SYNC_PLAYLISTS.key)
449 )
450 if not self._sync_tracks and not self._sync_playlists:
451 return
452 elif self.media_content_type == "audiobooks":
453 if not self.config.get_value(CONF_ENTRY_LIBRARY_SYNC_AUDIOBOOKS.key):
454 return
455 elif self.media_content_type == "podcasts":
456 if not self.config.get_value(CONF_ENTRY_LIBRARY_SYNC_PODCASTS.key):
457 return
458 assert self.mass.music.database
459 if self.sync_running:
460 self.logger.warning("Library sync already running for %s", self.name)
461 return
462 file_checksums: dict[str, str] = {}
463 # NOTE: we always run a scan of the entire library, as we need to detect changes
464 # we ignore any given mediatype(s) and just scan all supported files
465 query = (
466 f"SELECT provider_item_id, details FROM {DB_TABLE_PROVIDER_MAPPINGS} "
467 f"WHERE provider_instance = '{self.instance_id}' "
468 f"AND media_type in ('track', 'playlist', 'audiobook', 'podcast_episode')"
469 )
470 for db_row in await self.mass.music.database.get_rows_from_query(query, limit=0):
471 file_checksums[db_row["provider_item_id"]] = str(db_row["details"])
472 # provider_mappings stores synthetic per-track ids for CUE sheets, not the
473 # CUE path, so collect every track checksum per path for the scan classifier
474 cue_file_checksums: dict[str, set[str]] = {}
475 for prov_item_id, checksum in file_checksums.items():
476 parsed = parse_cue_track_id(prov_item_id)
477 if parsed is not None:
478 cue_file_checksums.setdefault(parsed[0], set()).add(checksum)
479 # find all supported files in the base directory and all subfolders
480 # we work bottom up, as-in we derive all info from the tracks
481 cur_filenames: set[str] = set()
482 prev_filenames = set(file_checksums.keys())
483
484 items_to_process: list[tuple[FileSystemItem, str | None]] = []
485 unchanged_cue_items: list[FileSystemItem] = []
486 # absolute paths of every CUE sheet in this scan with the ".cue" stripped,
487 # used for O(1) companion-CUE lookups per audio file
488 cue_stems: set[str] = set()
489 # collects the errors raised while walking the tree; any error means the
490 # scan is incomplete, a fatal one means the provider is unreachable
491 scan_errors = ScanErrors()
492
493 self.sync_running = True
494 try:
495 await self._enumerate_files_for_sync(
496 file_checksums=file_checksums,
497 cue_file_checksums=cue_file_checksums,
498 cur_filenames=cur_filenames,
499 items_to_process=items_to_process,
500 unchanged_cue_items=unchanged_cue_items,
501 cue_stems=cue_stems,
502 scan_errors=scan_errors,
503 )
504 if scan_errors.fatal:
505 # the storage is gone, so reading the files collected before it went
506 # away would only add a timeout each
507 self.logger.error("Aborting sync for %s: %s", self.name, scan_errors.fatal)
508 report_current_task_failure("Sync aborted: filesystem unavailable during scan")
509 self._set_available(False)
510 return
511 # a CUE may name an audio file other than its own; hide that companion too
512 if self.media_content_type == "music":
513 for cue_item in (
514 *unchanged_cue_items,
515 *(item for item, _ in items_to_process if item.ext in CUE_EXTENSIONS),
516 ):
517 try:
518 cue_sheet = await self._cue.load_cue_sheet(cue_item)
519 except InvalidDataError:
520 continue
521 if companion_stem := cue_referenced_audio_stem(cue_item, cue_sheet):
522 cue_stems.add(companion_stem)
523 # drop CUE companion audio: absorbed into CUE tracks and not tracked in
524 # provider_mappings, so they would otherwise flag as changed every sync
525 items_to_process = [
526 (item, prev)
527 for item, prev in items_to_process
528 if not (
529 item.ext in TRACK_EXTENSIONS
530 and item.absolute_path.rsplit(".", 1)[0] in cue_stems
531 )
532 ]
533 # register synthetic track IDs for unchanged CUE files so the
534 # deletion pass does not treat them as removed
535 for cue_item in unchanged_cue_items:
536 try:
537 cue_sheet = await self._cue.load_cue_sheet(cue_item)
538 except InvalidDataError as err:
539 self.logger.warning(
540 "Unable to parse CUE sheet %s: %s", cue_item.relative_path, err
541 )
542 continue
543 for cue_track in cue_sheet.tracks:
544 cur_filenames.add(make_cue_track_id(cue_item.relative_path, cue_track.number))
545 total_items = len(items_to_process)
546 self.logger.info(
547 "Found %d changed/new items to process for %s",
548 total_items,
549 self.name,
550 )
551
552 # _SYNC_CONCURRENCY caps parallelism per provider (NFS/SMB/WebDAV friendly)
553 processed_count = 0
554
555 async def _process(item: FileSystemItem, prev_checksum: str | None) -> None:
556 nonlocal processed_count
557 if await self._process_item_async(
558 item, prev_checksum, cur_filenames, cue_stems, prev_filenames
559 ):
560 cur_filenames.add(item.relative_path)
561 processed_count += 1
562 if processed_count % 50 == 0 or processed_count == total_items:
563 update_current_task_progress_from_index(
564 processed_count,
565 total_items,
566 f"Processed {processed_count}/{total_items} files",
567 )
568
569 async with TaskManager(self.mass, self._SYNC_CONCURRENCY) as tm:
570 for item, prev_checksum in items_to_process:
571 await tm.create_task_with_limit(_process(item, prev_checksum))
572 finally:
573 self.sync_running = False
574
575 # do not run deletions on a clean but empty scan of a previously non-empty library
576 # (wrong share mounted, empty backup mount, ...)
577 if prev_filenames and not cur_filenames:
578 self.logger.error(
579 "Aborting sync for %s: scan found no files but %d were previously indexed",
580 self.name,
581 len(prev_filenames),
582 )
583 report_current_task_failure(
584 f"Sync aborted: scan found no files but {len(prev_filenames)} "
585 "were previously indexed"
586 )
587 return
588
589 # a scan that skipped folders or files is incomplete: what it missed is still
590 # there, so deleting it from the library would throw away valid content
591 if scan_errors.incomplete:
592 summary = scan_errors.describe()
593 self.logger.warning("Skipping deletions for %s: %s", self.name, summary)
594 report_current_task_failure(f"Deletions skipped: {summary}")
595 else:
596 deleted_files = prev_filenames - cur_filenames
597 await self._process_deletions(deleted_files)
598 await self._process_orphaned_albums_and_artists()
599
600 # flag provider as available again if an earlier sync had marked it down
601 self._set_available(True)
602
603 async def get_artist(self, prov_artist_id: str) -> Artist:
604 """Get full artist details by id."""
605 db_artist = await self.mass.music.artists.get_library_item_by_prov_id(
606 prov_artist_id, self.instance_id
607 )
608 if not db_artist:
609 # this may happen if the artist is not in the db yet
610 # e.g. when browsing the filesystem
611 if await self.exists(prov_artist_id):
612 return await self._parse_artist(prov_artist_id, artist_path=prov_artist_id)
613 return await self._parse_artist(prov_artist_id)
614
615 # prov_artist_id is either an actual (relative) path or a name (as fallback)
616 safe_artist_name = create_safe_string(prov_artist_id, lowercase=False, replace_space=False)
617 if await self.exists(prov_artist_id):
618 artist_path = prov_artist_id
619 elif await self.exists(safe_artist_name):
620 artist_path = safe_artist_name
621 else:
622 for prov_mapping in db_artist.provider_mappings:
623 if prov_mapping.provider_instance != self.instance_id:
624 continue
625 if prov_mapping.url:
626 artist_path = prov_mapping.url
627 break
628 else:
629 # this is an artist without an actual path on disk
630 # return the info we already have in the db
631 return db_artist
632 return await self._parse_artist(
633 db_artist.name,
634 sort_name=db_artist.sort_name,
635 mbid=db_artist.mbid,
636 artist_path=artist_path,
637 )
638
639 async def get_album(self, prov_album_id: str) -> Album:
640 """Get full album details by id."""
641 parsed_cue_paths: set[str] = set()
642 for track in await self.get_album_tracks(prov_album_id):
643 for prov_mapping in track.provider_mappings:
644 if prov_mapping.provider_instance != self.instance_id:
645 continue
646 if parsed := parse_cue_track_id(prov_mapping.item_id):
647 # every track from the same CUE shares the same album; only parse once
648 if parsed[0] in parsed_cue_paths:
649 continue
650 parsed_cue_paths.add(parsed[0])
651 cue_item = await self.resolve(parsed[0])
652 for cue_track in await self._cue.parse_tracks(cue_item):
653 if isinstance(cue_track.album, Album):
654 return cue_track.album
655 continue
656 file_item = await self.resolve(prov_mapping.item_id)
657 tags = await async_parse_tags(file_item.absolute_path, file_item.file_size)
658 full_track = await self._parse_track(file_item, tags)
659 assert isinstance(full_track.album, Album)
660 return full_track.album
661 msg = f"Album not found: {prov_album_id}"
662 raise MediaNotFoundError(msg)
663
664 async def get_track(self, prov_track_id: str) -> Track:
665 """Get full track details by id."""
666 # ruff: noqa: PLR0915
667 if parsed := parse_cue_track_id(prov_track_id):
668 cue_item = await self.resolve(parsed[0])
669 for cue_track in await self._cue.parse_tracks(cue_item):
670 if cue_track.item_id == prov_track_id:
671 return cue_track
672 msg = f"CUE track not found: {prov_track_id}"
673 raise MediaNotFoundError(msg)
674
675 if not await self.exists(prov_track_id):
676 msg = f"Track path does not exist: {prov_track_id}"
677 raise MediaNotFoundError(msg)
678
679 file_item = await self.resolve(prov_track_id)
680 tags = await async_parse_tags(file_item.absolute_path, file_item.file_size)
681 return await self._parse_track(file_item, tags=tags, full_album_metadata=True)
682
683 async def get_podcast_episode(self, prov_episode_id: str) -> PodcastEpisode:
684 """Get (full) podcast episode details by id."""
685 if not await self.exists(prov_episode_id):
686 msg = f"Episode path does not exist: {prov_episode_id}"
687 raise MediaNotFoundError(msg)
688 file_item = await self.resolve(prov_episode_id)
689 tags = await async_parse_tags(file_item.absolute_path, file_item.file_size)
690 return await self._parse_podcast_episode(file_item, tags=tags)
691
692 async def get_playlist(self, prov_playlist_id: str) -> Playlist:
693 """Get full playlist details by id."""
694 if not await self.exists(prov_playlist_id):
695 msg = f"Playlist path does not exist: {prov_playlist_id}"
696 raise MediaNotFoundError(msg)
697
698 file_item = await self.resolve(prov_playlist_id)
699 playlist = Playlist(
700 item_id=file_item.relative_path,
701 provider=self.instance_id,
702 name=file_item.name,
703 provider_mappings={
704 ProviderMapping(
705 item_id=file_item.relative_path,
706 provider_domain=self.domain,
707 provider_instance=self.instance_id,
708 details=file_item.checksum,
709 in_library=True,
710 )
711 },
712 )
713 playlist.is_editable = ProviderFeature.PLAYLIST_TRACKS_EDIT in self.supported_features
714 # only playlists in the root are editable - all other are read only
715 if "/" in prov_playlist_id or "\\" in prov_playlist_id:
716 playlist.is_editable = False
717 # we do not (yet) have support to edit/create pls playlists, only m3u files can be edited
718 if file_item.ext == "pls":
719 playlist.is_editable = False
720 playlist.owner = self.name
721 # Check for local image with the same basename
722 if local_image := await self._get_playlist_local_image(file_item):
723 playlist.metadata.images = UniqueList([local_image])
724 return playlist
725
726 async def get_audiobook(self, prov_audiobook_id: str) -> Audiobook:
727 """Get full audiobook details by id."""
728 # ruff: noqa: PLR0915
729 if not await self.exists(prov_audiobook_id):
730 msg = f"Audiobook path does not exist: {prov_audiobook_id}"
731 raise MediaNotFoundError(msg)
732
733 file_item = await self.resolve(prov_audiobook_id)
734 tags = await async_parse_tags(file_item.absolute_path, file_item.file_size)
735 return await self._parse_audiobook(file_item, tags=tags)
736
737 async def get_podcast(self, prov_podcast_id: str) -> Podcast:
738 """Get full podcast details by id."""
739 async for episode in self.get_podcast_episodes(prov_podcast_id):
740 assert isinstance(episode.podcast, Podcast)
741 return episode.podcast
742 msg = f"Podcast not found: {prov_podcast_id}"
743 raise MediaNotFoundError(msg)
744
745 async def get_sound_effect(self, prov_sound_effect_id: str) -> SoundEffect:
746 """Get full sound effect details by id."""
747 if not await self.exists(prov_sound_effect_id):
748 msg = f"Sound effect path does not exist: {prov_sound_effect_id}"
749 raise MediaNotFoundError(msg)
750 file_item = await self.resolve(prov_sound_effect_id)
751 return await self._get_or_parse_sound_effect(file_item)
752
753 async def get_sound_effects(self) -> AsyncGenerator[SoundEffect]:
754 """Get all sound effect items this provider offers."""
755
756 def _walk() -> list[FileSystemItem]:
757 return sorted(
758 recursive_iter(
759 self.base_path, self.base_path, SOUND_EFFECT_EXTENSIONS, self.logger
760 ),
761 key=lambda x: x.relative_path,
762 )
763
764 for file_item in await asyncio.to_thread(_walk):
765 yield await self._get_or_parse_sound_effect(file_item)
766
767 async def get_album_tracks(self, prov_album_id: str) -> list[Track]:
768 """Get album tracks for given album id."""
769 # filesystem items are always stored in db so we can query the database
770 db_album = await self.mass.music.albums.get_library_item_by_prov_id(
771 prov_album_id, self.instance_id
772 )
773 if db_album is None:
774 msg = f"Album not found: {prov_album_id}"
775 raise MediaNotFoundError(msg)
776 album_tracks = await self.mass.music.albums.get_library_album_tracks(db_album.item_id)
777 return [
778 track
779 for track in album_tracks
780 if any(x.provider_instance == self.instance_id for x in track.provider_mappings)
781 ]
782
783 async def get_playlist_tracks(self, prov_playlist_id: str, page: int = 0) -> list[Track]:
784 """Get playlist tracks."""
785 result: list[Track] = []
786 if page > 0:
787 # paging not (yet) supported
788 return result
789 if not await self.exists(prov_playlist_id):
790 msg = f"Playlist path does not exist: {prov_playlist_id}"
791 raise MediaNotFoundError(msg)
792
793 file_item = await self.resolve(prov_playlist_id)
794 # We are using the checksum of the playlist file here to invalidate the cache
795 # when a change has been made to the playlist file (ie track addition/deletion)
796 cache_checksum = file_item.checksum
797
798 cache_key = f"get_playlist_tracks.{prov_playlist_id}"
799 cached_data = await self.mass.cache.get(
800 cache_key,
801 provider=self.instance_id,
802 checksum=cache_checksum,
803 category=0,
804 base_class=Track,
805 )
806 if cached_data is not None:
807 return cached_data # type: ignore[no-any-return]
808
809 _, ext = prov_playlist_id.rsplit(".", 1)
810 try:
811 # get playlist file contents
812 playlist_data_raw = await self._read_file(prov_playlist_id)
813 encoding = await detect_charset(playlist_data_raw)
814 playlist_data = playlist_data_raw.decode(encoding, errors="replace")
815
816 if ext in ("m3u", "m3u8"):
817 playlist_lines = parse_m3u(playlist_data)
818 else:
819 playlist_lines = parse_pls(playlist_data)
820
821 for idx, playlist_line in enumerate(playlist_lines, 1):
822 if "#EXT" in playlist_line.path:
823 continue
824 if track := await self._parse_playlist_line(
825 playlist_line.path, os.path.dirname(prov_playlist_id)
826 ):
827 track.position = idx
828 result.append(track)
829
830 except Exception as err:
831 self.logger.warning(
832 "Error while parsing playlist %s: %s",
833 prov_playlist_id,
834 str(err),
835 exc_info=err if self.logger.isEnabledFor(10) else None,
836 )
837
838 await self.mass.cache.set(
839 key=cache_key,
840 data=[track.to_dict() for track in result],
841 expiration=3600 * 24 * 365, # File timestamp checksum handles invalidation
842 provider=self.instance_id,
843 checksum=cache_checksum,
844 category=0,
845 )
846
847 return result
848
849 async def get_podcast_episodes(self, prov_podcast_id: str) -> AsyncGenerator[PodcastEpisode]:
850 """Get podcast episodes for given podcast id."""
851 folder_items = [item for item in await self._scandir(prov_podcast_id) if not item.is_dir]
852 episode_files = [x for x in folder_items if x.ext in PODCAST_EPISODE_EXTENSIONS]
853 # artwork and metadata.json count towards the signature too, because the parse embeds
854 # them into every episode. Case-insensitive, matching _get_podcast_metadata's exists()
855 signature_files = [
856 x
857 for x in folder_items
858 if x.ext in PODCAST_EPISODE_EXTENSIONS
859 or x.ext in IMAGE_EXTENSIONS
860 or x.filename.lower() == "metadata.json"
861 ]
862 cache_key = f"podcast_episodes.{prov_podcast_id}"
863 cache_checksum = get_folder_signature(signature_files)
864 if (
865 cached_episodes := await self.mass.cache.get(
866 cache_key,
867 provider=self.instance_id,
868 category=CACHE_CATEGORY_PODCAST_EPISODES,
869 checksum=cache_checksum,
870 base_class=PodcastEpisode,
871 )
872 ) is not None:
873 for episode in cached_episodes:
874 yield episode
875 return
876
877 # these caches have no checksum of their own, so drop them before parsing or the new
878 # entry gets the values the signature just invalidated. Refill once, or every parse
879 # task below misses at the same time and repeats the same scandir and file read
880 for stale_category in (CACHE_CATEGORY_FOLDER_IMAGES, CACHE_CATEGORY_PODCAST_METADATA):
881 await self.mass.cache.delete(
882 prov_podcast_id, category=stale_category, provider=self.instance_id
883 )
884 await self._get_local_images(prov_podcast_id)
885 await self._get_podcast_metadata(prov_podcast_id)
886
887 # collected by index so the listing keeps scandir order, not parse completion order
888 parsed: list[PodcastEpisode | None] = [None] * len(episode_files)
889
890 async def _process_podcast_episode(index: int, item: FileSystemItem) -> None:
891 try:
892 tags = await async_parse_tags(item.absolute_path, item.file_size)
893 parsed[index] = await self._parse_podcast_episode(item, tags)
894 except MusicAssistantError as err:
895 self.logger.warning(
896 "Could not parse uri/file %s to podcast episode: %s",
897 item.relative_path,
898 str(err),
899 )
900
901 # reuse the per-sync worker limit: the slowest filesystems to parse are exactly the
902 # ones that lower it
903 async with TaskManager(self.mass, self._SYNC_CONCURRENCY) as tm:
904 for index, item in enumerate(episode_files):
905 await tm.create_task_with_limit(_process_podcast_episode(index, item))
906
907 episodes = [episode for episode in parsed if episode is not None]
908 # cache an incomplete listing briefly rather than not at all, so one unreadable file
909 # cannot make every request re-parse the whole folder
910 complete = len(episodes) == len(episode_files)
911 await self.mass.cache.set(
912 key=cache_key,
913 data=[episode.to_dict() for episode in episodes],
914 # a complete listing is invalidated by the folder signature instead
915 expiration=3600 * 24 * 365 if complete else PARTIAL_LISTING_CACHE_EXPIRATION,
916 provider=self.instance_id,
917 category=CACHE_CATEGORY_PODCAST_EPISODES,
918 checksum=cache_checksum,
919 )
920
921 for episode in episodes:
922 yield episode
923
924 async def add_playlist_tracks(self, prov_playlist_id: str, prov_track_ids: list[str]) -> None:
925 """Add track(s) to playlist."""
926 if not await self.exists(prov_playlist_id):
927 msg = f"Playlist path does not exist: {prov_playlist_id}"
928 raise MediaNotFoundError(msg)
929 playlist_filename = self.get_absolute_path(prov_playlist_id)
930 async with aiofiles.open(playlist_filename, encoding="utf-8") as _file:
931 playlist_data = await _file.read()
932 for file_path in prov_track_ids:
933 track = await self.get_track(file_path)
934 playlist_data += f"\n#EXTINF:{track.duration or 0},{track.name}\n{file_path}\n"
935
936 # write playlist file (always in utf-8)
937 async with aiofiles.open(playlist_filename, "w", encoding="utf-8") as _file:
938 await _file.write(playlist_data)
939
940 async def remove_playlist_tracks(
941 self, prov_playlist_id: str, positions_to_remove: tuple[int, ...]
942 ) -> None:
943 """Remove track(s) from playlist."""
944 if not await self.exists(prov_playlist_id):
945 msg = f"Playlist path does not exist: {prov_playlist_id}"
946 raise MediaNotFoundError(msg)
947 _, ext = prov_playlist_id.rsplit(".", 1)
948 # get playlist file contents
949 playlist_filename = self.get_absolute_path(prov_playlist_id)
950 async with aiofiles.open(playlist_filename, encoding="utf-8") as _file:
951 playlist_data = await _file.read()
952 # get current contents first
953 if ext in ("m3u", "m3u8"):
954 playlist_items = parse_m3u(playlist_data)
955 else:
956 playlist_items = parse_pls(playlist_data)
957 # remove items by index
958 for i in sorted(positions_to_remove, reverse=True):
959 # position = index + 1
960 del playlist_items[i - 1]
961 # build new playlist data
962 new_playlist_data = "#EXTM3U\n"
963 for item in playlist_items:
964 new_playlist_data += f"\n#EXTINF:{item.length or 0},{item.title}\n{item.path}\n"
965 async with aiofiles.open(playlist_filename, "w", encoding="utf-8") as _file:
966 await _file.write(new_playlist_data)
967
968 async def create_playlist(self, name: str, media_types: set[MediaType]) -> Playlist:
969 """Create a new playlist on provider with given name."""
970 # creating a new playlist on the filesystem is as easy
971 # as creating a new (empty) file with the m3u extension...
972 # filename = await self.resolve(f"{name}.m3u")
973 filename = f"{name}.m3u"
974 playlist_filename = self.get_absolute_path(filename)
975 async with aiofiles.open(playlist_filename, "w", encoding="utf-8") as _file:
976 await _file.write("#EXTM3U\n")
977 return await self.get_playlist(filename)
978
979 async def get_stream_details(self, item_id: str, media_type: MediaType) -> StreamDetails:
980 """Return the content details for the given track when it will be streamed."""
981 try:
982 if media_type == MediaType.AUDIOBOOK:
983 return await self._get_stream_details_for_audiobook(item_id)
984 if media_type == MediaType.PODCAST_EPISODE:
985 return await self._get_stream_details_for_podcast_episode(item_id)
986 if media_type == MediaType.SOUND_EFFECT:
987 return await self._get_stream_details_for_sound_effect(item_id)
988 return await self._get_stream_details_for_track(item_id)
989 except FileNotFoundError:
990 self.logger.warning(
991 "File not found for media item %s",
992 item_id,
993 )
994 msg = f"Media file not found: {item_id}"
995 raise MediaNotFoundError(msg)
996
997 async def get_audio_stream(
998 self, streamdetails: StreamDetails, seek_position: int = 0
999 ) -> AsyncGenerator[bytes]:
1000 """Return the custom audio stream for the provider item."""
1001 # only CUE-derived tracks use StreamType.CUSTOM in this provider
1002 async for chunk in self._cue.get_audio_stream(streamdetails, seek_position):
1003 yield chunk
1004
1005 async def resolve_image(self, path: str) -> str | bytes:
1006 """
1007 Resolve an image from an image path.
1008
1009 This either returns (a generator to get) raw bytes of the image or
1010 a string with an http(s) URL or local path that is accessible from the server.
1011 """
1012 # drop the cache-busting suffix appended by _versioned_image_path
1013 try:
1014 file_item = await self.resolve(path.split("?cs=", 1)[0])
1015 except FileNotFoundError as err:
1016 # the referenced image file was removed from disk; surface a typed
1017 # not-found so the image layer treats it as a missing image
1018 raise MediaNotFoundError(f"Image not found: {path}") from err
1019 if file_item.is_dir:
1020 # handing the path back would have the image layer run an ffmpeg
1021 # embedded-artwork extraction on the directory before giving up
1022 raise MediaNotFoundError(f"Image path is a directory: {path}")
1023 return file_item.absolute_path
1024
1025 async def check_write_access(self) -> None:
1026 """Perform check if we have write access."""
1027 # verify write access to determine we have playlist create/edit support
1028 # overwrite with provider specific implementation if needed
1029 temp_file_name = self.get_absolute_path(f"{shortuuid.random(8)}.txt")
1030 try:
1031 async with aiofiles.open(temp_file_name, "w") as _file:
1032 await _file.write("test")
1033 await asyncio.to_thread(os.remove, temp_file_name)
1034 self.write_access = True
1035 except Exception as err:
1036 self.logger.debug("Write access disabled: %s", str(err))
1037
1038 async def resolve(self, file_path: str) -> FileSystemItem:
1039 """Resolve (absolute or relative) path to FileSystemItem."""
1040 absolute_path = self.get_absolute_path(file_path)
1041
1042 def _create_item() -> FileSystemItem:
1043 if Path(absolute_path).is_dir():
1044 return FileSystemItem(
1045 filename=Path(file_path).name,
1046 relative_path=get_relative_path(self.base_path, file_path),
1047 absolute_path=absolute_path,
1048 is_dir=True,
1049 )
1050 stat_info = Path(absolute_path).stat(follow_symlinks=False)
1051 return FileSystemItem(
1052 filename=Path(file_path).name,
1053 relative_path=get_relative_path(self.base_path, file_path),
1054 absolute_path=absolute_path,
1055 is_dir=False,
1056 checksum=str(int(stat_info.st_mtime)),
1057 file_size=stat_info.st_size,
1058 )
1059
1060 return await asyncio.to_thread(_create_item)
1061
1062 async def exists(self, file_path: str) -> bool:
1063 """Return bool is this FileSystem musicprovider has given file/dir."""
1064 if not file_path:
1065 return False
1066 try:
1067 abs_path = self.get_absolute_path(file_path)
1068 except MediaNotFoundError:
1069 # a path that escapes the base directory simply does not exist here
1070 return False
1071 return bool(await exists(abs_path))
1072
1073 def get_absolute_path(self, file_path: str) -> str:
1074 """Return absolute path for given file path."""
1075 return get_absolute_path(self.base_path, file_path)
1076
1077 async def _enumerate_files_for_sync(
1078 self,
1079 *,
1080 file_checksums: dict[str, str],
1081 cue_file_checksums: dict[str, set[str]],
1082 cur_filenames: set[str],
1083 items_to_process: list[tuple[FileSystemItem, str | None]],
1084 unchanged_cue_items: list[FileSystemItem],
1085 cue_stems: set[str],
1086 scan_errors: ScanErrors,
1087 ) -> None:
1088 """
1089 Walk every supported file under the provider root and populate the sync buckets.
1090
1091 Override in subclasses that cannot use a local ``os.scandir`` walk.
1092 Implementations must route each discovered file through
1093 :meth:`_classify_scan_item`, report every unreadable directory to
1094 ``scan_errors`` and stop the walk once it reports ``aborted``.
1095
1096 :param file_checksums: Previously stored checksum per provider item id.
1097 :param cue_file_checksums: Previously stored track checksums keyed by CUE relative_path.
1098 :param cur_filenames: Receives the ids/paths present in this scan.
1099 :param items_to_process: Receives changed or new items to process.
1100 :param unchanged_cue_items: Receives CUE sheets whose checksum matches.
1101 :param cue_stems: Receives absolute paths (minus extension) of CUE sheets.
1102 :param scan_errors: Receives the errors raised while walking the tree.
1103 """
1104 ignore_album_playlists = self.media_content_type == "music" and bool(
1105 self.config.get_value(CONF_ENTRY_IGNORE_ALBUM_PLAYLISTS.key)
1106 )
1107
1108 def _walk() -> None:
1109 for scanned, item in enumerate(
1110 recursive_iter(
1111 self.base_path,
1112 self.base_path,
1113 SUPPORTED_EXTENSIONS,
1114 self.logger,
1115 scan_errors=scan_errors,
1116 ),
1117 start=1,
1118 ):
1119 if scanned % 500 == 0:
1120 update_current_task_progress_text(f"Scanning files: {scanned} found")
1121 self._classify_scan_item(
1122 item,
1123 file_checksums=file_checksums,
1124 cue_file_checksums=cue_file_checksums,
1125 cur_filenames=cur_filenames,
1126 items_to_process=items_to_process,
1127 unchanged_cue_items=unchanged_cue_items,
1128 cue_stems=cue_stems,
1129 ignore_album_playlists=ignore_album_playlists,
1130 )
1131
1132 await asyncio.to_thread(_walk)
1133
1134 def _classify_scan_item(
1135 self,
1136 item: FileSystemItem,
1137 *,
1138 file_checksums: dict[str, str],
1139 cue_file_checksums: dict[str, set[str]],
1140 cur_filenames: set[str],
1141 items_to_process: list[tuple[FileSystemItem, str | None]],
1142 unchanged_cue_items: list[FileSystemItem],
1143 cue_stems: set[str],
1144 ignore_album_playlists: bool,
1145 ) -> None:
1146 """
1147 Route a single scanned file into the correct sync bucket.
1148
1149 :param item: The file to classify.
1150 :param file_checksums: Previously stored checksum per provider item id.
1151 :param cue_file_checksums: Previously stored track checksums keyed by CUE relative_path.
1152 :param cur_filenames: Receives the ids/paths present in this scan.
1153 :param items_to_process: Receives changed or new items to process.
1154 :param unchanged_cue_items: Receives CUE sheets whose checksum matches.
1155 :param cue_stems: Receives absolute paths (minus extension) of CUE sheets.
1156 :param ignore_album_playlists: When True, skip playlists nested inside
1157 album directories.
1158 """
1159 # a file this provider never imports gets no mapping, so it would flag as
1160 # changed on every sync; it is still on disk, so record it as present
1161 if not self._is_imported_file(item):
1162 cur_filenames.add(item.relative_path)
1163 return
1164 # skip playlists in album directories if configured
1165 if (
1166 item.ext in PLAYLIST_EXTENSIONS
1167 and ignore_album_playlists
1168 and len(item.relative_path.split("/")) > 2
1169 ):
1170 return
1171 is_cue = item.ext in CUE_EXTENSIONS and self.media_content_type == "music"
1172 item_checksum = item.checksum
1173 if is_cue:
1174 cue_stems.add(item.absolute_path.rsplit(".", 1)[0])
1175 item_checksum = cue_metadata_checksum(item.checksum)
1176 prev_checksums = cue_file_checksums.get(item.relative_path, set())
1177 prev_checksum = min(prev_checksums, default=None)
1178 checksum_matches = prev_checksums == {item_checksum}
1179 else:
1180 prev_checksum = file_checksums.get(item.relative_path)
1181 checksum_matches = item_checksum == prev_checksum
1182 if checksum_matches:
1183 # unchanged, just record it as still present
1184 cur_filenames.add(item.relative_path)
1185 if is_cue:
1186 unchanged_cue_items.append(item)
1187 else:
1188 items_to_process.append((item, prev_checksum))
1189
1190 def _is_imported_file(self, item: FileSystemItem) -> bool:
1191 """Return True when this provider imports the given file into the library."""
1192 if self.media_content_type == "music":
1193 if item.ext in CUE_EXTENSIONS:
1194 return True
1195 if item.ext in TRACK_EXTENSIONS:
1196 return self._sync_tracks
1197 if item.ext in PLAYLIST_EXTENSIONS:
1198 return self._sync_playlists
1199 return False
1200 if self.media_content_type == "audiobooks":
1201 return item.ext in AUDIOBOOK_EXTENSIONS
1202 if self.media_content_type == "podcasts":
1203 return item.ext in PODCAST_EPISODE_EXTENSIONS
1204 return False
1205
1206 def _set_available(self, available: bool) -> None:
1207 """Update the provider availability and notify listeners on change."""
1208 if self.available == available:
1209 return
1210 self.available = available
1211 if available:
1212 self._cancel_availability_probe()
1213 else:
1214 self._schedule_availability_probe()
1215 self.mass.signal_event(EventType.PROVIDERS_UPDATED, data=self.mass.get_providers())
1216
1217 async def _is_reachable(self) -> bool:
1218 """Return whether the storage backing this provider can be read."""
1219 return bool(await isdir(self.base_path))
1220
1221 @property
1222 def _availability_probe_id(self) -> str:
1223 """Return the timer id of this provider's reachability checks."""
1224 return f"filesystem_availability_probe_{self.instance_id}"
1225
1226 def _schedule_availability_probe(self) -> None:
1227 """Arm the next reachability check."""
1228 self.mass.call_later(
1229 AVAILABILITY_PROBE_INTERVAL,
1230 self._probe_availability,
1231 task_id=self._availability_probe_id,
1232 )
1233
1234 def _cancel_availability_probe(self) -> None:
1235 """Stop checking for the storage coming back."""
1236 self.mass.cancel_timer(self._availability_probe_id)
1237
1238 async def _probe_availability(self) -> None:
1239 """Mark the provider available again once its storage can be read."""
1240 try:
1241 reachable = await self._is_reachable()
1242 except MusicAssistantError as err:
1243 # storage that is simply still gone, which is what this loop waits for
1244 self.logger.debug("%s is still unreachable: %s", self.name, err)
1245 reachable = False
1246 except Exception:
1247 # an unexpected failure must not end the loop, since it is what brings the
1248 # provider back, but it is a defect rather than an outage so it is logged loudly
1249 self.logger.exception("Reachability check for %s failed", self.name)
1250 reachable = False
1251 if self.unloading:
1252 # the provider was torn down while this check was running; re-arming here
1253 # would leave a timer firing against an instance nothing owns anymore
1254 return
1255 if reachable:
1256 self.logger.info("%s is reachable again", self.name)
1257 self._set_available(True)
1258 return
1259 self._schedule_availability_probe()
1260
1261 async def _process_item_async(
1262 self,
1263 item: FileSystemItem,
1264 prev_checksum: str | None,
1265 cur_filenames: set[str] | None = None,
1266 cue_stems: set[str] | None = None,
1267 prev_filenames: set[str] | None = None,
1268 ) -> bool:
1269 """
1270 Process a single item asynchronously.
1271
1272 :param item: The filesystem item to process.
1273 :param prev_checksum: Previous checksum from the database, or None for new items.
1274 :param cur_filenames: Set of current filenames being tracked (for CUE track IDs).
1275 :param cue_stems: Absolute paths (without extension) of CUE sheets in this scan,
1276 used to detect companion-CUE audio files without a filesystem stat.
1277 :param prev_filenames: The ids/paths the previous scan found, used to keep the
1278 ids of a CUE sheet that fails to parse.
1279 """
1280 try:
1281 self.logger.log(VERBOSE_LOG_LEVEL, "Processing: %s", item.relative_path)
1282
1283 if prev_checksum is not None:
1284 # the file changed on disk: drop cached artwork derived from it
1285 # (thumbnails, source bytes, palette) so re-read embedded art is
1286 # served fresh, for both reference forms of the image path
1287 await self.mass.metadata.invalidate_image_cache(
1288 self.instance_id, item.relative_path
1289 )
1290 await self.mass.metadata.invalidate_image_cache(
1291 self.instance_id, self._versioned_image_path(item.relative_path, prev_checksum)
1292 )
1293
1294 if item.ext in CUE_EXTENSIONS and self.media_content_type == "music":
1295 tracks = await self._cue.parse_tracks(item)
1296 for track in tracks:
1297 track.favorite = False
1298 await self.mass.music.tracks.add_item_to_library(
1299 track, overwrite_existing=prev_checksum is not None
1300 )
1301 if cur_filenames is not None:
1302 cur_filenames.add(track.item_id)
1303 return True
1304
1305 if item.ext in TRACK_EXTENSIONS and self.media_content_type == "music":
1306 if not self._sync_tracks:
1307 return False
1308 # skip audio files that have a companion CUE sheet
1309 if cue_stems is not None and item.absolute_path.rsplit(".", 1)[0] in cue_stems:
1310 return False
1311 tags = await async_parse_tags(item.absolute_path, item.file_size)
1312 track = await self._parse_track(item, tags)
1313 track.favorite = False # TODO: implement favorite status based on rating ?
1314 await self.mass.music.tracks.add_item_to_library(
1315 track, overwrite_existing=prev_checksum is not None
1316 )
1317 return True
1318
1319 if item.ext in AUDIOBOOK_EXTENSIONS and self.media_content_type == "audiobooks":
1320 tags = await async_parse_tags(item.absolute_path, item.file_size)
1321 try:
1322 audiobook = await self._parse_audiobook(item, tags)
1323 except IsChapterFile:
1324 return True
1325 await self.mass.music.audiobooks.add_item_to_library(
1326 audiobook, overwrite_existing=prev_checksum is not None
1327 )
1328 return True
1329
1330 if item.ext in PODCAST_EPISODE_EXTENSIONS and self.media_content_type == "podcasts":
1331 tags = await async_parse_tags(item.absolute_path, item.file_size)
1332 episode = await self._parse_podcast_episode(item, tags)
1333 assert isinstance(episode.podcast, Podcast)
1334 await self.mass.music.podcasts.add_item_to_library(
1335 episode.podcast, overwrite_existing=prev_checksum is not None
1336 )
1337 return True
1338
1339 if item.ext in PLAYLIST_EXTENSIONS and self.media_content_type == "music":
1340 if not self._sync_playlists:
1341 return False
1342 playlist = await self.get_playlist(item.relative_path)
1343 await self.mass.music.playlists.add_item_to_library(
1344 playlist, overwrite_existing=prev_checksum is not None
1345 )
1346 return True
1347
1348 except Exception as err:
1349 # we don't want the whole sync to crash on one file so we catch all exceptions here
1350 self.logger.error(
1351 "Error processing %s - %s",
1352 item.relative_path,
1353 str(err),
1354 exc_info=err if self.logger.isEnabledFor(logging.DEBUG) else None,
1355 )
1356 report_current_task_failure(f"Failed to process {item.relative_path}: {err}")
1357 # the file is still on the storage, so keep it in the scan result:
1358 # leaving it out makes the deletion step treat it as removed
1359 self._keep_failed_item(item, cur_filenames, prev_filenames)
1360 return False
1361
1362 def _keep_failed_item(
1363 self,
1364 item: FileSystemItem,
1365 cur_filenames: set[str] | None,
1366 prev_filenames: set[str] | None,
1367 ) -> None:
1368 """
1369 Keep an item that could not be processed in the scan result.
1370
1371 :param item: The item that failed to process.
1372 :param cur_filenames: Receives the ids/paths present in this scan.
1373 :param prev_filenames: The ids/paths the previous scan found.
1374 """
1375 if cur_filenames is None:
1376 return
1377 cur_filenames.add(item.relative_path)
1378 if (
1379 item.ext not in CUE_EXTENSIONS
1380 or self.media_content_type != "music"
1381 or not prev_filenames
1382 ):
1383 return
1384 # a CUE sheet stands in for one id per track it describes and those cannot be
1385 # rebuilt without parsing it, so carry over the ids of the previous scan
1386 cur_filenames.update(
1387 item_id
1388 for item_id in prev_filenames
1389 if (parsed := parse_cue_track_id(item_id)) and parsed[0] == item.relative_path
1390 )
1391
1392 async def _process_orphaned_albums_and_artists(self) -> None:
1393 """Process deletion of orphaned albums and artists."""
1394 assert self.mass.music.database
1395 # Remove albums without any tracks
1396 query = (
1397 f"SELECT item_id FROM {DB_TABLE_ALBUMS} "
1398 f"WHERE item_id not in ( SELECT album_id from {DB_TABLE_ALBUM_TRACKS}) "
1399 f"AND item_id in ( SELECT item_id from {DB_TABLE_PROVIDER_MAPPINGS} "
1400 f"WHERE provider_instance = '{self.instance_id}' and media_type = 'album' )"
1401 )
1402 for db_row in await self.mass.music.database.get_rows_from_query(
1403 query,
1404 limit=100000,
1405 ):
1406 await self.mass.music.albums.remove_item_from_library(db_row["item_id"])
1407
1408 # Remove artists without any tracks or albums
1409 query = (
1410 f"SELECT item_id FROM {DB_TABLE_ARTISTS} "
1411 f"WHERE item_id not in "
1412 f"( select artist_id from {DB_TABLE_TRACK_ARTISTS} "
1413 f"UNION SELECT artist_id from {DB_TABLE_ALBUM_ARTISTS} ) "
1414 f"AND item_id in ( SELECT item_id from {DB_TABLE_PROVIDER_MAPPINGS} "
1415 f"WHERE provider_instance = '{self.instance_id}' and media_type = 'artist' )"
1416 )
1417 for db_row in await self.mass.music.database.get_rows_from_query(
1418 query,
1419 limit=100000,
1420 ):
1421 await self.mass.music.artists.remove_item_from_library(db_row["item_id"])
1422
1423 async def _process_deletions(self, deleted_files: set[str]) -> None:
1424 """Process all deletions."""
1425 # process deleted tracks/playlists
1426 album_ids = set()
1427 artist_ids = set()
1428 for file_path in deleted_files:
1429 if parse_cue_track_id(file_path) is not None and self.media_content_type == "music":
1430 controller = self.mass.music.get_controller(MediaType.TRACK)
1431 elif "." not in file_path:
1432 continue
1433 else:
1434 _, ext = file_path.rsplit(".", 1)
1435 if ext in PODCAST_EPISODE_EXTENSIONS and self.media_content_type == "podcasts":
1436 controller = self.mass.music.get_controller(MediaType.PODCAST_EPISODE)
1437 elif ext in AUDIOBOOK_EXTENSIONS and self.media_content_type == "audiobooks":
1438 controller = self.mass.music.get_controller(MediaType.AUDIOBOOK)
1439 elif ext in PLAYLIST_EXTENSIONS and self.media_content_type == "music":
1440 controller = self.mass.music.get_controller(MediaType.PLAYLIST)
1441 elif ext in TRACK_EXTENSIONS and self.media_content_type == "music":
1442 controller = self.mass.music.get_controller(MediaType.TRACK)
1443 else:
1444 # unsupported file extension?
1445 continue
1446
1447 if library_item := await controller.get_library_item_by_prov_id(
1448 file_path, self.instance_id
1449 ):
1450 if is_track(library_item):
1451 if library_item.album:
1452 album_ids.add(library_item.album.item_id)
1453 # need to fetch the library album to resolve the itemmapping
1454 db_album = await self.mass.music.albums.get_library_item(
1455 library_item.album.item_id
1456 )
1457 for artist in db_album.artists:
1458 artist_ids.add(artist.item_id)
1459 for artist in library_item.artists:
1460 artist_ids.add(artist.item_id)
1461 await controller.remove_item_from_library(library_item.item_id)
1462 # check if any albums need to be cleaned up
1463 for album_id in album_ids:
1464 if not await self.mass.music.albums.tracks(album_id, "library"):
1465 await self.mass.music.albums.remove_item_from_library(album_id)
1466 # check if any artists need to be cleaned up
1467 for artist_id in artist_ids:
1468 artist_albums = await self.mass.music.artists.albums(artist_id, "library")
1469 artist_tracks = await self.mass.music.artists.tracks(artist_id, "library")
1470 if not (artist_albums or artist_tracks):
1471 await self.mass.music.artists.remove_item_from_library(artist_id)
1472
1473 async def _get_playlist_local_image(self, file_item: FileSystemItem) -> MediaItemImage | None:
1474 """Return a local image alongside the playlist file (matching basename) if any."""
1475 cache_key = f"playlist_image.{file_item.relative_path}"
1476 cached = await self.cache.get(
1477 key=cache_key,
1478 provider=self.instance_id,
1479 category=CACHE_CATEGORY_FOLDER_IMAGES,
1480 base_class=MediaItemImage,
1481 )
1482 if cached is not None:
1483 return cached[0] if cached else None
1484 try:
1485 folder_files = await self._scandir(file_item.relative_parent_path)
1486 except OSError, MusicAssistantError:
1487 return None
1488 target = file_item.name.lower()
1489 result: MediaItemImage | None = None
1490 for item in folder_files:
1491 if item.is_dir or not item.ext:
1492 continue
1493 if item.ext.lower() not in IMAGE_EXTENSIONS:
1494 continue
1495 if item.name.lower() != target:
1496 continue
1497 result = MediaItemImage(
1498 type=ImageType.THUMB,
1499 path=item.relative_path,
1500 provider=self.instance_id,
1501 remotely_accessible=False,
1502 )
1503 break
1504 await self.cache.set(
1505 key=cache_key,
1506 data=[result.to_dict()] if result is not None else [],
1507 provider=self.instance_id,
1508 category=CACHE_CATEGORY_FOLDER_IMAGES,
1509 expiration=120,
1510 )
1511 return result
1512
1513 async def _parse_playlist_line(self, line: str, playlist_path: str) -> Track | None:
1514 """Try to parse a track from a playlist line."""
1515 try:
1516 line = line.replace("file://", "").strip()
1517 # try to resolve the filename (both normal and url decoded):
1518 # - relative to the playlist folder (normpath resolves parent .. references)
1519 # - as-is: an absolute path, or relative to our base path
1520 # candidates stay relative so subclasses with virtual paths (cloud,
1521 # webdav) resolve them too, instead of leaking the server CWD
1522 for _line in (line, urllib.parse.unquote(line)):
1523 if playlist_path:
1524 normalized = posixpath.normpath(f"{playlist_path}/{_line}")
1525 with contextlib.suppress(FileNotFoundError, MediaNotFoundError):
1526 file_item = await self.resolve(normalized)
1527 tags = await async_parse_tags(file_item.absolute_path, file_item.file_size)
1528 return await self._parse_track(file_item, tags)
1529 with contextlib.suppress(FileNotFoundError, MediaNotFoundError):
1530 file_item = await self.resolve(_line)
1531 tags = await async_parse_tags(file_item.absolute_path, file_item.file_size)
1532 return await self._parse_track(file_item, tags)
1533 # all attempts failed
1534 raise MediaNotFoundError("Invalid path/uri")
1535
1536 except MusicAssistantError as err:
1537 self.logger.warning("Could not parse %s to track: %s", line, str(err))
1538
1539 return None
1540
1541 @staticmethod
1542 def _versioned_image_path(relative_path: str, checksum: str | None) -> str:
1543 """Append the file checksum so the image cache busts when the file is replaced."""
1544 if checksum:
1545 return f"{relative_path}?cs={checksum}"
1546 return relative_path
1547
1548 @staticmethod
1549 def _codec_type_from_tags(tags: AudioTags) -> ContentType:
1550 """Return the audio codec detected by ffprobe, if any."""
1551 if tags.raw and (streams := tags.raw.get("streams")):
1552 if codec_name := streams[0].get("codec_name"):
1553 return ContentType.try_parse(codec_name)
1554 return ContentType.UNKNOWN
1555
1556 async def _parse_track(
1557 self, file_item: FileSystemItem, tags: AudioTags, full_album_metadata: bool = False
1558 ) -> Track:
1559 """Parse full track details from file tags."""
1560 # ruff: noqa: PLR0915
1561 name, version = parse_title_and_version(tags.title, tags.version)
1562 track = Track(
1563 item_id=file_item.relative_path,
1564 provider=self.instance_id,
1565 name=name,
1566 sort_name=tags.title_sort,
1567 version=version,
1568 provider_mappings={
1569 ProviderMapping(
1570 item_id=file_item.relative_path,
1571 provider_domain=self.domain,
1572 provider_instance=self.instance_id,
1573 audio_format=AudioFormat(
1574 content_type=ContentType.try_parse(file_item.ext or tags.format),
1575 codec_type=self._codec_type_from_tags(tags),
1576 sample_rate=tags.sample_rate,
1577 bit_depth=tags.bits_per_sample,
1578 channels=tags.channels,
1579 bit_rate=tags.bit_rate,
1580 ),
1581 details=file_item.checksum,
1582 in_library=True,
1583 )
1584 },
1585 disc_number=tags.disc or 0,
1586 track_number=tags.track or 0,
1587 date_added=(
1588 datetime.fromtimestamp(file_item.created_at, tz=UTC)
1589 if file_item.created_at
1590 else None
1591 ),
1592 )
1593
1594 if isrc_tags := tags.isrc:
1595 for isrsc in isrc_tags:
1596 track.external_ids.add((ExternalID.ISRC, isrsc))
1597
1598 if acoustid := tags.get("acoustid"):
1599 track.external_ids.add((ExternalID.ACOUSTID, acoustid))
1600
1601 # album
1602 album = track.album = (
1603 await self._parse_album(
1604 track_path=file_item.relative_path,
1605 track_tags=tags,
1606 track_created_at=file_item.created_at,
1607 )
1608 if tags.album
1609 else None
1610 )
1611
1612 # track artist(s)
1613 resolved_track_artists = await self._resolve_artists_with_mbids(
1614 tags.artists,
1615 tags.musicbrainz_artistids,
1616 tags.artist_sort_names,
1617 log_label="ARTISTS tag",
1618 )
1619 for name, mbid, sort_name in resolved_track_artists:
1620 # prefer the existing album artist object when it's the same artist
1621 if album_artist_match := self._match_album_artist(album, name, mbid):
1622 track.artists.append(album_artist_match)
1623 continue
1624 artist = await self._parse_artist(name, sort_name=sort_name, mbid=mbid)
1625 track.artists.append(artist)
1626
1627 # handle embedded cover image
1628 if tags.has_cover_image:
1629 # we do not actually embed the image in the metadata because that would consume too
1630 # much space and bandwidth. Instead we set the filename as value so the image can
1631 # be retrieved later in realtime.
1632 track.metadata.images = UniqueList(
1633 [
1634 MediaItemImage(
1635 type=ImageType.THUMB,
1636 path=file_item.relative_path,
1637 provider=self.instance_id,
1638 remotely_accessible=False,
1639 )
1640 ]
1641 )
1642
1643 # copy (embedded) album image from track (if the album itself doesn't have an image)
1644 if album and not album.image and track.image:
1645 album.metadata.images = UniqueList([track.image])
1646
1647 # parse other info
1648 track.duration = int(tags.duration or 0)
1649 track.metadata.genres = set(tags.genres)
1650 if tags.disc:
1651 track.disc_number = tags.disc
1652 if tags.track:
1653 track.track_number = tags.track
1654 track.metadata.copyright = tags.get("copyright")
1655 track.metadata.lyrics = tags.lyrics
1656 track.metadata.grouping = tags.get("grouping")
1657 track.metadata.description = tags.get("comment")
1658 explicit_tag = tags.get("itunesadvisory")
1659 if explicit_tag is not None:
1660 track.metadata.explicit = explicit_tag == "1"
1661 if recording_mbid := clean_mbid(tags.musicbrainz_recordingid, tags.filename):
1662 track.mbid = recording_mbid
1663
1664 # handle (optional) loudness measurement tag(s)
1665 if tags.track_loudness is not None:
1666 self.mass.create_task(
1667 self.mass.streams.audio_analysis.set_track_loudness(
1668 track.item_id,
1669 self.instance_id,
1670 tags.track_loudness,
1671 tags.track_album_loudness,
1672 )
1673 )
1674
1675 # possible lrclib metadata
1676 # synced lyrics are saved as "filename.lrc" by lrcget alongside
1677 # the actual file location - just change the file extension
1678 assert file_item.ext is not None # for type checking
1679 lrc_path = f"{file_item.relative_path.removesuffix(file_item.ext)}lrc"
1680 if await self.exists(lrc_path):
1681 try:
1682 raw = await self._read_file(lrc_path)
1683 track.metadata.lrc_lyrics = raw.decode("utf-8")
1684 except Exception as err:
1685 self.logger.warning(
1686 "Failed to read lyrics file %s: %s",
1687 lrc_path,
1688 str(err),
1689 )
1690 elif syn_lyrics := tags.synchronized_lyrics:
1691 track.metadata.lrc_lyrics = lyrics.convert_to_lrc_lyrics(syn_lyrics)
1692
1693 return track
1694
1695 async def _resolve_artists_with_mbids(
1696 self,
1697 parsed_names: tuple[str, ...],
1698 mbids: tuple[str, ...],
1699 sort_names: tuple[str, ...],
1700 log_label: str,
1701 ) -> list[tuple[str, str | None, str | None]]:
1702 """
1703 Return ``(name, mbid, sort_name)`` triples for a track's or album's artists.
1704
1705 When the parsed name count and the MBID count disagree, canonical names
1706 are looked up from MusicBrainz; otherwise the tag-parsed names are used.
1707
1708 :param parsed_names: Tag-parsed artist names.
1709 :param mbids: MusicBrainz artist IDs from the tag.
1710 :param sort_names: Sort names from the corresponding *sort tag.
1711 :param log_label: Tag name used in warning messages (e.g. "ARTISTS tag").
1712 """
1713
1714 def _sort_name(index: int) -> str | None:
1715 return sort_names[index] if index < len(sort_names) else None
1716
1717 def _from_tags() -> list[tuple[str, str | None, str | None]]:
1718 return [
1719 (
1720 name,
1721 mbids[i] if i < len(mbids) else None,
1722 _sort_name(i),
1723 )
1724 for i, name in enumerate(parsed_names)
1725 ]
1726
1727 if not mbids or len(parsed_names) == len(mbids):
1728 return _from_tags()
1729
1730 mb_provider = cast("MusicbrainzProvider | None", self.mass.get_provider("musicbrainz"))
1731 if mb_provider is None:
1732 self.logger.warning(
1733 "%s count (%d) doesn't match MBID count (%d) and MusicBrainz "
1734 "provider is not loaded; using tag-parsed names: %s",
1735 log_label,
1736 len(parsed_names),
1737 len(mbids),
1738 parsed_names,
1739 )
1740 return _from_tags()
1741
1742 mb_results = await mb_provider.resolve_artists_from_mbids(mbids)
1743 # counts disagree, so positional fallback to a tag name is unreliable;
1744 # drop any MBID whose lookup failed (already logged per-MBID)
1745 resolved: list[tuple[str, str | None, str | None]] = [
1746 mb_result for mb_result in mb_results if mb_result is not None
1747 ]
1748 if not resolved:
1749 self.logger.warning(
1750 "%s count (%d) didn't match MBID count (%d) and every MusicBrainz "
1751 "lookup failed; falling back to tag-parsed names: %s",
1752 log_label,
1753 len(parsed_names),
1754 len(mbids),
1755 parsed_names,
1756 )
1757 return _from_tags()
1758 self.logger.info(
1759 "%s count (%d) didn't match MBID count (%d); resolved canonical names "
1760 "via MusicBrainz: %s",
1761 log_label,
1762 len(parsed_names),
1763 len(mbids),
1764 [r[0] for r in resolved],
1765 )
1766 return resolved
1767
1768 def _match_album_artist(
1769 self, album: Album | None, name: str, mbid: str | None
1770 ) -> Artist | ItemMapping | None:
1771 """
1772 Return an existing album artist representing the same artist, if any.
1773
1774 Matches on MusicBrainz ID when available (names may differ when only one
1775 side was resolved against MusicBrainz), otherwise on exact name.
1776
1777 :param album: The track's album, if known.
1778 :param name: Resolved track-artist name.
1779 :param mbid: Resolved track-artist MusicBrainz ID, if any.
1780 """
1781 if not album:
1782 return None
1783 return next(
1784 (x for x in album.artists if (mbid and x.mbid == mbid) or x.name == name),
1785 None,
1786 )
1787
1788 async def _parse_artist(
1789 self,
1790 name: str,
1791 album_dir: str | None = None,
1792 sort_name: str | None = None,
1793 mbid: str | None = None,
1794 artist_path: str | None = None,
1795 ) -> Artist:
1796 """Parse full (album) Artist."""
1797 if not artist_path:
1798 # we need to hunt for the artist (metadata) path on disk
1799 # this can either be relative to the album path or at root level
1800 # check if we have an artist folder for this artist at root level
1801 safe_artist_name = create_safe_string(name, lowercase=False, replace_space=False)
1802 if await self.exists(name):
1803 artist_path = name
1804 elif await self.exists(safe_artist_name):
1805 artist_path = safe_artist_name
1806 elif album_dir and (foldermatch := get_artist_dir(name, album_dir=album_dir)):
1807 # try to find (album)artist folder based on album path
1808 artist_path = foldermatch
1809 else:
1810 # check if we have an existing item to retrieve the artist path
1811 async for item in self.mass.music.artists.iter_library_items(
1812 search=name, provider=self.instance_id
1813 ):
1814 if not compare_strings(name, item.name):
1815 continue
1816 for prov_mapping in item.provider_mappings:
1817 if prov_mapping.provider_instance != self.instance_id:
1818 continue
1819 if prov_mapping.url:
1820 artist_path = prov_mapping.url
1821 break
1822 if artist_path:
1823 break
1824
1825 # prefer (short lived) cache for a bit more speed
1826 if artist_path and (
1827 cache := await self.cache.get(
1828 key=artist_path,
1829 provider=self.instance_id,
1830 category=CACHE_CATEGORY_ARTIST_INFO,
1831 base_class=Artist,
1832 )
1833 ):
1834 return cache # type: ignore[no-any-return]
1835
1836 prov_artist_id = artist_path or name
1837 artist = Artist(
1838 item_id=prov_artist_id,
1839 provider=self.instance_id,
1840 name=name,
1841 sort_name=sort_name,
1842 provider_mappings={
1843 ProviderMapping(
1844 item_id=prov_artist_id,
1845 provider_domain=self.domain,
1846 provider_instance=self.instance_id,
1847 url=artist_path,
1848 in_library=True,
1849 )
1850 },
1851 )
1852 if mbid := clean_mbid(mbid, f"tags of artist {name}"):
1853 artist.mbid = mbid
1854 if not artist_path or not await self.exists(artist_path):
1855 return artist
1856
1857 # grab additional metadata within the Artist's folder
1858 nfo_file = os.path.join(artist_path, "artist.nfo")
1859 if await self.exists(nfo_file):
1860 # found NFO file with metadata
1861 # https://kodi.wiki/view/NFO_files/Artists
1862 try:
1863 data = (await self._read_file(nfo_file)).decode("utf-8")
1864 info = await asyncio.to_thread(xmltodict.parse, data)
1865 info = info["artist"]
1866 artist.name = info.get("title", info.get("name", name))
1867 if sort_name := info.get("sortname"):
1868 artist.sort_name = sort_name
1869 if mbid := clean_mbid(info.get("musicbrainzartistid"), nfo_file):
1870 artist.mbid = mbid
1871 if description := info.get("biography"):
1872 artist.metadata.description = description
1873 if genre := info.get("genre"):
1874 artist.metadata.genres = set(split_items(genre))
1875 except (ExpatError, KeyError) as err:
1876 self.logger.warning(
1877 "Failed to parse artist NFO file %s: %s",
1878 nfo_file,
1879 str(err),
1880 )
1881 # find local images
1882 if images := await self._get_local_images(artist_path, extra_thumb_names=("artist",)):
1883 artist.metadata.images = UniqueList(images)
1884
1885 await self.cache.set(
1886 key=artist_path,
1887 data=artist.to_dict(),
1888 provider=self.instance_id,
1889 category=CACHE_CATEGORY_ARTIST_INFO,
1890 expiration=120,
1891 )
1892
1893 return artist
1894
1895 async def _parse_audiobook(self, file_item: FileSystemItem, tags: AudioTags) -> Audiobook:
1896 """
1897 Parse Audiobook details from file tags.
1898
1899 Audiobooks can be single files with embedded chapters or multiple files per folder.
1900 Only the first file (by track number or alphabetically) is processed as the audiobook.
1901 """
1902 # Skip files that aren't the first chapter.
1903 # A file carrying its own embedded chapter markers is a standalone audiobook,
1904 # so it should never be treated as a chapter file of another book.
1905 track_tag = tags.tags.get("track")
1906 if track_tag:
1907 track_num = try_parse_int(str(track_tag).split("/")[0], None)
1908 if track_num and track_num > 1 and not tags.chapters:
1909 raise IsChapterFile
1910 elif not tags.chapters:
1911 # No track tag and no embedded chapters -
1912 # assume part of a multi-file audiobook, only process the first file alphabetically
1913 items = await self._scandir(file_item.relative_parent_path)
1914 # Sort by filename for alphabetical ordering
1915 items.sort(key=lambda x: x.filename.lower())
1916 for item in items:
1917 if item.is_dir or item.ext not in AUDIOBOOK_EXTENSIONS:
1918 continue
1919 if item.absolute_path != file_item.absolute_path:
1920 raise IsChapterFile
1921 break
1922
1923 # For multi-file audiobooks, album tag is the book name, title is the chapter name
1924 if tags.album:
1925 book_name = tags.album
1926 sort_name = tags.album_sort
1927 elif (title := tags.tags.get("title")) and tags.track is None:
1928 book_name = title
1929 sort_name = tags.title_sort
1930 else:
1931 # file(s) without tags, use foldername
1932 book_name = file_item.parent_name
1933 sort_name = None
1934
1935 # collect all chapters
1936 total_duration, chapters = await self._get_chapters_for_audiobook(file_item, tags)
1937
1938 audio_book = Audiobook(
1939 item_id=file_item.relative_path,
1940 provider=self.instance_id,
1941 name=book_name,
1942 sort_name=sort_name,
1943 version=tags.version,
1944 duration=total_duration or int(tags.duration or 0),
1945 provider_mappings={
1946 ProviderMapping(
1947 item_id=file_item.relative_path,
1948 provider_domain=self.domain,
1949 provider_instance=self.instance_id,
1950 audio_format=AudioFormat(
1951 content_type=ContentType.try_parse(file_item.ext or tags.format),
1952 codec_type=self._codec_type_from_tags(tags),
1953 sample_rate=tags.sample_rate,
1954 bit_depth=tags.bits_per_sample,
1955 channels=tags.channels,
1956 bit_rate=tags.bit_rate,
1957 ),
1958 details=file_item.checksum,
1959 in_library=True,
1960 )
1961 },
1962 )
1963 audio_book.metadata.chapters = chapters
1964
1965 # handle embedded cover image
1966 if tags.has_cover_image:
1967 # we do not actually embed the image in the metadata because that would consume too
1968 # much space and bandwidth. Instead we set the filename as value so the image can
1969 # be retrieved later in realtime.
1970 audio_book.metadata.add_image(
1971 MediaItemImage(
1972 type=ImageType.THUMB,
1973 path=self._versioned_image_path(file_item.relative_path, file_item.checksum),
1974 provider=self.instance_id,
1975 remotely_accessible=False,
1976 )
1977 )
1978
1979 # parse other info
1980 audio_book.authors.set(tags.writers or tags.album_artists or tags.artists)
1981 audio_book.metadata.genres = (
1982 set(tags.genres) if tags.genres else {DEFAULT_AUDIOBOOK_PODCAST_GENRE}
1983 )
1984 audio_book.metadata.copyright = tags.get("copyright")
1985 audio_book.metadata.lyrics = tags.lyrics
1986 audio_book.metadata.description = tags.get("comment")
1987 explicit_tag = tags.get("itunesadvisory")
1988 if explicit_tag is not None:
1989 audio_book.metadata.explicit = explicit_tag == "1"
1990 if recording_mbid := clean_mbid(tags.musicbrainz_recordingid, tags.filename):
1991 audio_book.mbid = recording_mbid
1992
1993 # try to fetch additional metadata from the folder
1994 if not audio_book.image or not audio_book.metadata.description:
1995 # try to get an image by traversing files in the same folder
1996 for _item in await self._scandir(file_item.relative_parent_path):
1997 if "." not in _item.relative_path or _item.is_dir:
1998 continue
1999 if _item.ext in IMAGE_EXTENSIONS and not audio_book.image:
2000 audio_book.metadata.add_image(
2001 MediaItemImage(
2002 type=ImageType.THUMB,
2003 path=self._versioned_image_path(_item.relative_path, _item.checksum),
2004 provider=self.instance_id,
2005 remotely_accessible=False,
2006 )
2007 )
2008 if _item.ext == "txt" and not audio_book.metadata.description:
2009 # try to parse a description from a text file
2010 try:
2011 raw = await self._read_file(_item.relative_path)
2012 audio_book.metadata.description = raw.decode("utf-8")
2013 except Exception as err:
2014 self.logger.warning(
2015 "Could not read description from file %s: %s",
2016 _item.relative_path,
2017 str(err),
2018 )
2019
2020 # handle (optional) loudness measurement tag(s)
2021 if tags.track_loudness is not None:
2022 self.mass.create_task(
2023 self.mass.streams.audio_analysis.set_track_loudness(
2024 audio_book.item_id,
2025 self.instance_id,
2026 tags.track_loudness,
2027 tags.track_album_loudness,
2028 media_type=MediaType.AUDIOBOOK,
2029 )
2030 )
2031 return audio_book
2032
2033 async def _parse_podcast_episode(
2034 self, file_item: FileSystemItem, tags: AudioTags
2035 ) -> PodcastEpisode:
2036 """Parse full PodcastEpisode details from file tags."""
2037 # ruff: noqa: PLR0915
2038 podcast_name = tags.album or file_item.parent_name
2039 podcast_path = file_item.relative_parent_path
2040 episode = PodcastEpisode(
2041 item_id=file_item.relative_path,
2042 provider=self.instance_id,
2043 name=tags.title,
2044 sort_name=tags.title_sort,
2045 provider_mappings={
2046 ProviderMapping(
2047 item_id=file_item.relative_path,
2048 provider_domain=self.domain,
2049 provider_instance=self.instance_id,
2050 audio_format=AudioFormat(
2051 content_type=ContentType.try_parse(file_item.ext or tags.format),
2052 codec_type=self._codec_type_from_tags(tags),
2053 sample_rate=tags.sample_rate,
2054 bit_depth=tags.bits_per_sample,
2055 channels=tags.channels,
2056 bit_rate=tags.bit_rate,
2057 ),
2058 details=file_item.checksum,
2059 in_library=True,
2060 )
2061 },
2062 position=tags.track or 0,
2063 duration=try_parse_int(tags.duration) or 0,
2064 podcast=Podcast(
2065 item_id=podcast_path,
2066 provider=self.instance_id,
2067 name=podcast_name,
2068 sort_name=tags.album_sort,
2069 publisher=tags.tags.get("publisher"),
2070 provider_mappings={
2071 ProviderMapping(
2072 item_id=podcast_path,
2073 provider_domain=self.domain,
2074 provider_instance=self.instance_id,
2075 in_library=True,
2076 )
2077 },
2078 ),
2079 )
2080 # handle embedded cover image
2081 if tags.has_cover_image:
2082 # we do not actually embed the image in the metadata because that would consume too
2083 # much space and bandwidth. Instead we set the filename as value so the image can
2084 # be retrieved later in realtime.
2085 episode.metadata.add_image(
2086 MediaItemImage(
2087 type=ImageType.THUMB,
2088 path=file_item.relative_path,
2089 provider=self.instance_id,
2090 remotely_accessible=False,
2091 )
2092 )
2093 # parse other info
2094 episode.metadata.genres = (
2095 set(tags.genres) if tags.genres else {DEFAULT_AUDIOBOOK_PODCAST_GENRE}
2096 )
2097 episode.metadata.copyright = tags.get("copyright")
2098 episode.metadata.lyrics = tags.lyrics
2099 episode.metadata.description = tags.get("comment")
2100 explicit_tag = tags.get("itunesadvisory")
2101 if explicit_tag is not None:
2102 episode.metadata.explicit = explicit_tag == "1"
2103
2104 # handle (optional) chapters
2105 if tags.chapters:
2106 episode.metadata.chapters = [
2107 MediaItemChapter(
2108 position=chapter.chapter_id,
2109 name=chapter.title or f"Chapter {chapter.chapter_id}",
2110 start=chapter.position_start,
2111 end=chapter.position_end,
2112 )
2113 for chapter in tags.chapters
2114 ]
2115
2116 # try to fetch additional Podcast metadata from the folder
2117 assert isinstance(episode.podcast, Podcast)
2118 if images := await self._get_local_images(file_item.relative_parent_path):
2119 episode.podcast.metadata.images = images
2120 if metadata := await self._get_podcast_metadata(file_item.relative_parent_path):
2121 if title := metadata.get("title"):
2122 episode.podcast.name = title
2123 if sort_name := metadata.get("sorttitle"):
2124 episode.podcast.sort_name = sort_name
2125 if description := metadata.get("description"):
2126 episode.podcast.metadata.description = description
2127 if genres := metadata.get("genres"):
2128 episode.podcast.metadata.genres = set(genres)
2129 if publisher := metadata.get("publisher"):
2130 episode.podcast.publisher = publisher
2131 if image := metadata.get("imageURL"):
2132 episode.podcast.metadata.add_image(
2133 MediaItemImage(
2134 type=ImageType.THUMB,
2135 path=image,
2136 provider=self.instance_id,
2137 remotely_accessible=True,
2138 )
2139 )
2140 # copy (embedded) image from episode (or vice versa)
2141 if not episode.podcast.image and episode.image:
2142 episode.podcast.metadata.add_image(episode.image)
2143 elif not episode.image and episode.podcast.image:
2144 episode.metadata.add_image(episode.podcast.image)
2145 # ensure podcast has a default genre if none set
2146 if not episode.podcast.metadata.genres:
2147 episode.podcast.metadata.genres = {DEFAULT_AUDIOBOOK_PODCAST_GENRE}
2148
2149 # handle (optional) loudness measurement tag(s)
2150 if tags.track_loudness is not None:
2151 self.mass.create_task(
2152 self.mass.streams.audio_analysis.set_track_loudness(
2153 episode.item_id,
2154 self.instance_id,
2155 tags.track_loudness,
2156 tags.track_album_loudness,
2157 media_type=MediaType.PODCAST_EPISODE,
2158 )
2159 )
2160 return episode
2161
2162 async def _parse_sound_effect(self, file_item: FileSystemItem, tags: AudioTags) -> SoundEffect:
2163 """Parse full sound effect details from file tags."""
2164 sound_effect = SoundEffect(
2165 item_id=file_item.relative_path,
2166 provider=self.instance_id,
2167 name=tags.title,
2168 sort_name=tags.title_sort,
2169 duration=int(tags.duration or 0),
2170 provider_mappings={
2171 ProviderMapping(
2172 item_id=file_item.relative_path,
2173 provider_domain=self.domain,
2174 provider_instance=self.instance_id,
2175 audio_format=AudioFormat(
2176 content_type=ContentType.try_parse(file_item.ext or tags.format),
2177 codec_type=self._codec_type_from_tags(tags),
2178 sample_rate=tags.sample_rate,
2179 bit_depth=tags.bits_per_sample,
2180 channels=tags.channels,
2181 bit_rate=tags.bit_rate,
2182 ),
2183 details=file_item.checksum,
2184 in_library=True,
2185 )
2186 },
2187 )
2188 sound_effect.metadata.description = tags.get("comment")
2189 # handle embedded cover image
2190 if tags.has_cover_image:
2191 # we do not actually embed the image in the metadata because that would consume too
2192 # much space and bandwidth. Instead we set the filename as value so the image can
2193 # be retrieved later in realtime.
2194 sound_effect.metadata.add_image(
2195 MediaItemImage(
2196 type=ImageType.THUMB,
2197 path=file_item.relative_path,
2198 provider=self.instance_id,
2199 remotely_accessible=False,
2200 )
2201 )
2202 return sound_effect
2203
2204 async def _get_or_parse_sound_effect(self, file_item: FileSystemItem) -> SoundEffect:
2205 """Return the (cached) SoundEffect for the given file, parsing tags when needed."""
2206 cache_key = f"sound_effect.{file_item.relative_path}"
2207 cached_data: SoundEffect | None = await self.cache.get(
2208 cache_key,
2209 provider=self.instance_id,
2210 checksum=file_item.checksum,
2211 category=CACHE_CATEGORY_SOUND_EFFECTS,
2212 base_class=SoundEffect,
2213 )
2214 if cached_data is not None:
2215 return cached_data
2216 tags = await async_parse_tags(file_item.absolute_path, file_item.file_size)
2217 sound_effect = await self._parse_sound_effect(file_item, tags)
2218 await self.cache.set(
2219 cache_key,
2220 sound_effect.to_dict(),
2221 expiration=3600 * 24 * 365, # File timestamp checksum handles invalidation
2222 provider=self.instance_id,
2223 checksum=file_item.checksum,
2224 category=CACHE_CATEGORY_SOUND_EFFECTS,
2225 )
2226 return sound_effect
2227
2228 async def _parse_album(
2229 self, track_path: str, track_tags: AudioTags, track_created_at: int | None = None
2230 ) -> Album:
2231 """
2232 Parse Album metadata from Track tags.
2233
2234 :param track_path: Path to the track file.
2235 :param track_tags: Audio tags from the track.
2236 :param track_created_at: Creation timestamp of the track file (Unix epoch).
2237 """
2238 assert track_tags.album
2239 # work out if we have an album and/or disc folder
2240 # track_dir is the folder level where the tracks are located
2241 # this may be a separate disc folder (Disc 1, Disc 2 etc) underneath the album folder
2242 # or this is an album folder with the disc attached
2243 track_dir = os.path.dirname(track_path)
2244 album_dir = get_album_dir(track_dir, track_tags.album)
2245
2246 if album_dir and (
2247 cache := await self.cache.get(
2248 key=album_dir,
2249 provider=self.instance_id,
2250 category=CACHE_CATEGORY_ALBUM_INFO,
2251 base_class=Album,
2252 )
2253 ):
2254 return cache # type: ignore[no-any-return]
2255
2256 # album artist(s)
2257 album_artists: UniqueList[Artist | ItemMapping] = UniqueList()
2258 if track_tags.album_artists:
2259 resolved_album_artists = await self._resolve_artists_with_mbids(
2260 track_tags.album_artists,
2261 track_tags.musicbrainz_albumartistids,
2262 track_tags.album_artist_sort_names,
2263 log_label="ALBUMARTIST tag",
2264 )
2265 for name, mbid, sort_name in resolved_album_artists:
2266 artist = await self._parse_artist(
2267 name, album_dir=album_dir, sort_name=sort_name, mbid=mbid
2268 )
2269 album_artists.append(artist)
2270 else:
2271 # album artist tag is missing, determine fallback
2272 fallback_action = self.config.get_value(CONF_ENTRY_MISSING_ALBUM_ARTIST.key)
2273 if fallback_action == "folder_name" and album_dir:
2274 possible_artist_folder = os.path.dirname(album_dir)
2275 self.logger.warning(
2276 "%s is missing ID3 tag [albumartist], using foldername %s as fallback",
2277 track_path,
2278 possible_artist_folder,
2279 )
2280 album_artist_str = Path(possible_artist_folder).name
2281 album_artists = UniqueList(
2282 [await self._parse_artist(name=album_artist_str, album_dir=album_dir)]
2283 )
2284 # fallback to track artists (if defined by user)
2285 elif fallback_action == "track_artist":
2286 self.logger.warning(
2287 "%s is missing ID3 tag [albumartist], using track artist(s) as fallback",
2288 track_path,
2289 )
2290 album_artists = UniqueList(
2291 [
2292 await self._parse_artist(name=track_artist_str, album_dir=album_dir)
2293 for track_artist_str in track_tags.artists
2294 ]
2295 )
2296 # all other: fallback to various artists
2297 else:
2298 self.logger.warning(
2299 "%s is missing ID3 tag [albumartist], using %s as fallback",
2300 track_path,
2301 VARIOUS_ARTISTS_NAME,
2302 )
2303 album_artists = UniqueList(
2304 [await self._parse_artist(name=VARIOUS_ARTISTS_NAME, mbid=VARIOUS_ARTISTS_MBID)]
2305 )
2306
2307 if album_dir: # noqa: SIM108
2308 # prefer the path as id
2309 item_id = album_dir
2310 else:
2311 # create fake item_id based on artist + album
2312 item_id = album_artists[0].name + os.sep + track_tags.album
2313
2314 name, version = parse_title_and_version(track_tags.album)
2315 album = Album(
2316 item_id=item_id,
2317 provider=self.instance_id,
2318 name=name,
2319 version=version,
2320 sort_name=track_tags.album_sort,
2321 artists=album_artists,
2322 provider_mappings={
2323 ProviderMapping(
2324 item_id=item_id,
2325 provider_domain=self.domain,
2326 provider_instance=self.instance_id,
2327 url=album_dir,
2328 in_library=True,
2329 )
2330 },
2331 date_added=(
2332 datetime.fromtimestamp(track_created_at, tz=UTC) if track_created_at else None
2333 ),
2334 )
2335 if track_tags.barcode:
2336 album.external_ids.add((ExternalID.BARCODE, track_tags.barcode))
2337
2338 if album_mbid := clean_mbid(track_tags.musicbrainz_albumid, track_tags.filename):
2339 album.mbid = album_mbid
2340 if releasegroup_mbid := clean_mbid(
2341 track_tags.musicbrainz_releasegroupid, track_tags.filename
2342 ):
2343 album.add_external_id(ExternalID.MB_RELEASEGROUP, releasegroup_mbid)
2344 if track_tags.year:
2345 album.year = track_tags.year
2346 album.album_type = track_tags.album_type
2347
2348 # hunt for additional metadata and images in the folder structure
2349 if not album_dir:
2350 return album
2351
2352 for folder_path in (track_dir, album_dir):
2353 if not folder_path or not await self.exists(folder_path):
2354 continue
2355 nfo_file = os.path.join(folder_path, "album.nfo")
2356 if await self.exists(nfo_file):
2357 # found NFO file with metadata
2358 # https://kodi.wiki/view/NFO_files/Artists
2359 try:
2360 data = (await self._read_file(nfo_file)).decode("utf-8")
2361 info = await asyncio.to_thread(xmltodict.parse, data)
2362 parse_album_nfo(album, info["album"], nfo_file)
2363 except (ExpatError, KeyError) as err:
2364 self.logger.warning(
2365 "Failed to parse album NFO file %s: %s",
2366 nfo_file,
2367 str(err),
2368 )
2369
2370 # find local images
2371 if images := await self._get_local_images(folder_path, extra_thumb_names=("album",)):
2372 if album.metadata.images is None:
2373 album.metadata.images = UniqueList(images)
2374 else:
2375 album.metadata.images += images
2376
2377 await self.cache.set(
2378 key=album_dir,
2379 data=album.to_dict(),
2380 provider=self.instance_id,
2381 category=CACHE_CATEGORY_ALBUM_INFO,
2382 expiration=120,
2383 )
2384 return album
2385
2386 async def _get_local_images(
2387 self, folder: str, extra_thumb_names: tuple[str, ...] | None = None
2388 ) -> UniqueList[MediaItemImage]:
2389 """Return local images found in a given folderpath."""
2390 if (
2391 cache := await self.cache.get(
2392 key=folder,
2393 provider=self.instance_id,
2394 category=CACHE_CATEGORY_FOLDER_IMAGES,
2395 base_class=MediaItemImage,
2396 )
2397 ) is not None:
2398 return UniqueList(cache)
2399 if extra_thumb_names is None:
2400 extra_thumb_names = ()
2401 images: UniqueList[MediaItemImage] = UniqueList()
2402 folder_files = await self._scandir(folder)
2403 for item in folder_files:
2404 if "." not in item.relative_path or item.is_dir or not item.ext:
2405 continue
2406 if item.ext.lower() not in IMAGE_EXTENSIONS:
2407 continue
2408 # try match on filename = one of our imagetypes
2409 if item.name.lower() in ImageType:
2410 images.append(
2411 MediaItemImage(
2412 type=ImageType(item.name),
2413 path=item.relative_path,
2414 provider=self.instance_id,
2415 remotely_accessible=False,
2416 )
2417 )
2418
2419 # try alternative names for thumbs
2420 extra_thumb_names = ("folder", "cover", *extra_thumb_names)
2421 for item in folder_files:
2422 if "." not in item.relative_path or item.is_dir or not item.ext:
2423 continue
2424 if item.ext.lower() not in IMAGE_EXTENSIONS:
2425 continue
2426 if item.name.lower() not in extra_thumb_names:
2427 continue
2428 images.append(
2429 MediaItemImage(
2430 type=ImageType.THUMB,
2431 path=item.relative_path,
2432 provider=self.instance_id,
2433 remotely_accessible=False,
2434 )
2435 )
2436
2437 await self.cache.set(
2438 key=folder,
2439 data=[img.to_dict() for img in images],
2440 provider=self.instance_id,
2441 category=CACHE_CATEGORY_FOLDER_IMAGES,
2442 expiration=120,
2443 )
2444 return images
2445
2446 async def _get_stream_details_for_track(self, item_id: str) -> StreamDetails:
2447 """Return the streamdetails for a track/song."""
2448 if parse_cue_track_id(item_id) is not None:
2449 return await self._cue.get_stream_details(item_id)
2450
2451 library_item = await self.mass.music.tracks.get_library_item_by_prov_id(
2452 item_id, self.instance_id
2453 )
2454 if library_item is None:
2455 # this could be a file that has just been added, try parsing it
2456 file_item = await self.resolve(item_id)
2457 tags = await async_parse_tags(file_item.absolute_path, file_item.file_size)
2458 if not (library_item := await self._parse_track(file_item, tags)):
2459 msg = f"Item not found: {item_id}"
2460 raise MediaNotFoundError(msg)
2461
2462 prov_mapping = next(x for x in library_item.provider_mappings if x.item_id == item_id)
2463 file_item = await self.resolve(item_id)
2464
2465 return StreamDetails(
2466 provider=self.instance_id,
2467 item_id=item_id,
2468 audio_format=prov_mapping.audio_format,
2469 media_type=MediaType.TRACK,
2470 stream_type=StreamType.LOCAL_FILE,
2471 duration=library_item.duration,
2472 size=file_item.file_size,
2473 data=file_item,
2474 path=file_item.absolute_path,
2475 can_seek=True,
2476 allow_seek=True,
2477 )
2478
2479 async def _get_stream_details_for_podcast_episode(self, item_id: str) -> StreamDetails:
2480 """Return the streamdetails for a podcast episode."""
2481 # podcasts episodes are never stored in the library so we need to parse the file
2482 file_item = await self.resolve(item_id)
2483 tags = await async_parse_tags(file_item.absolute_path, file_item.file_size)
2484 return StreamDetails(
2485 provider=self.instance_id,
2486 item_id=item_id,
2487 audio_format=AudioFormat(
2488 content_type=ContentType.try_parse(file_item.ext or tags.format),
2489 codec_type=self._codec_type_from_tags(tags),
2490 sample_rate=tags.sample_rate,
2491 bit_depth=tags.bits_per_sample,
2492 channels=tags.channels,
2493 bit_rate=tags.bit_rate,
2494 ),
2495 media_type=MediaType.PODCAST_EPISODE,
2496 stream_type=StreamType.LOCAL_FILE,
2497 duration=try_parse_int(tags.duration or 0),
2498 size=file_item.file_size,
2499 data=file_item,
2500 path=file_item.absolute_path,
2501 allow_seek=True,
2502 can_seek=True,
2503 )
2504
2505 async def _get_stream_details_for_sound_effect(self, item_id: str) -> StreamDetails:
2506 """Return the streamdetails for a sound effect."""
2507 # sound effects are never stored in the library so we parse the file,
2508 # served from cache unless the file changed on disk
2509 file_item = await self.resolve(item_id)
2510 sound_effect = await self._get_or_parse_sound_effect(file_item)
2511 prov_mapping = next(x for x in sound_effect.provider_mappings if x.item_id == item_id)
2512 return StreamDetails(
2513 provider=self.instance_id,
2514 item_id=item_id,
2515 audio_format=prov_mapping.audio_format,
2516 media_type=MediaType.SOUND_EFFECT,
2517 stream_type=StreamType.LOCAL_FILE,
2518 duration=sound_effect.duration,
2519 size=file_item.file_size,
2520 data=file_item,
2521 path=file_item.absolute_path,
2522 allow_seek=True,
2523 can_seek=True,
2524 )
2525
2526 async def _get_stream_details_for_audiobook(self, item_id: str) -> StreamDetails:
2527 """Return the streamdetails for an audiobook."""
2528 library_item = await self.mass.music.audiobooks.get_library_item_by_prov_id(
2529 item_id, self.instance_id
2530 )
2531 if library_item is None:
2532 # this could be a file that has just been added, try parsing it
2533 file_item = await self.resolve(item_id)
2534 tags = await async_parse_tags(file_item.absolute_path, file_item.file_size)
2535 if not (library_item := await self._parse_audiobook(file_item, tags)):
2536 msg = f"Item not found: {item_id}"
2537 raise MediaNotFoundError(msg)
2538
2539 prov_mapping = next(x for x in library_item.provider_mappings if x.item_id == item_id)
2540 file_item = await self.resolve(item_id)
2541 duration = library_item.duration
2542 file_based_chapters: list[tuple[str, float]] | None = await self.cache.get(
2543 key=file_item.relative_path,
2544 provider=self.instance_id,
2545 category=CACHE_CATEGORY_AUDIOBOOK_CHAPTERS,
2546 )
2547 if file_based_chapters is None:
2548 # no cache available for this audiobook, we need to parse the chapters
2549 tags = await async_parse_tags(file_item.absolute_path, file_item.file_size)
2550 await self._parse_audiobook(file_item, tags)
2551 file_based_chapters = await self.cache.get(
2552 key=file_item.relative_path,
2553 provider=self.instance_id,
2554 category=CACHE_CATEGORY_AUDIOBOOK_CHAPTERS,
2555 )
2556
2557 if file_based_chapters:
2558 # this is a multi-file audiobook
2559 return StreamDetails(
2560 provider=self.instance_id,
2561 item_id=item_id,
2562 audio_format=prov_mapping.audio_format,
2563 media_type=MediaType.AUDIOBOOK,
2564 stream_type=StreamType.LOCAL_FILE,
2565 duration=duration,
2566 path=[
2567 MultiPartPath(
2568 path=self._get_chapter_path(chapter_path),
2569 duration=chapter_duration,
2570 )
2571 for chapter_path, chapter_duration in file_based_chapters
2572 ],
2573 allow_seek=True,
2574 )
2575
2576 # regular single-file streaming, simply let ffmpeg deal with the file directly
2577 return StreamDetails(
2578 provider=self.instance_id,
2579 item_id=item_id,
2580 audio_format=prov_mapping.audio_format,
2581 media_type=MediaType.AUDIOBOOK,
2582 stream_type=StreamType.LOCAL_FILE,
2583 duration=library_item.duration,
2584 size=file_item.file_size,
2585 data=file_item,
2586 path=file_item.absolute_path,
2587 allow_seek=True,
2588 can_seek=True,
2589 )
2590
2591 def _get_chapter_path(self, relative_path: str) -> str:
2592 """Return absolute path for a chapter file. Override for network storage."""
2593 return self.get_absolute_path(relative_path)
2594
2595 async def _get_chapters_for_audiobook(
2596 self, audiobook_file_item: FileSystemItem, tags: AudioTags
2597 ) -> tuple[int, list[MediaItemChapter]]:
2598 """
2599 Return chapters for an audiobook.
2600
2601 Chapter sources in order of preference:
2602 1. Multiple files with track tags - sorted by track number
2603 2. Single file with embedded chapters - use embedded chapter markers
2604 3. Multiple files without track tags - sorted alphabetically (fallback)
2605 """
2606 chapters: list[MediaItemChapter] = []
2607 all_chapter_files: list[tuple[str, float]] = []
2608 total_duration = 0.0
2609
2610 # Scan folder for chapter files, separating tagged from untagged
2611 chapter_file_items: list[tuple[FileSystemItem, AudioTags]] = []
2612 untagged_file_items: list[tuple[FileSystemItem, AudioTags]] = []
2613
2614 items = await self._scandir(audiobook_file_item.relative_parent_path)
2615 # Sort by filename for consistent alphabetical ordering
2616 items.sort(key=lambda x: x.filename.lower())
2617
2618 for item in items:
2619 if "." not in item.relative_path or item.is_dir:
2620 continue
2621 if item.ext not in AUDIOBOOK_EXTENSIONS:
2622 continue
2623 item_tags = await async_parse_tags(item.absolute_path, item.file_size)
2624 if not (tags.album == item_tags.album or (item_tags.tags.get("title") is None)):
2625 continue
2626 if item_tags.tags.get("track") is None:
2627 untagged_file_items.append((item, item_tags))
2628 else:
2629 chapter_file_items.append((item, item_tags))
2630
2631 # Determine chapter source
2632 use_embedded = False
2633 use_alphabetical = False
2634
2635 if len(chapter_file_items) > 1:
2636 chapter_file_items.sort(key=lambda x: (x[1].disc or 0, x[1].track or 0))
2637 elif len(chapter_file_items) <= 1 and tags.chapters:
2638 use_embedded = True
2639 elif len(untagged_file_items) > 1:
2640 use_alphabetical = True
2641 chapter_file_items = untagged_file_items
2642 self.logger.info(
2643 "Audiobook files have no track tags, using alphabetical order: %s",
2644 tags.album,
2645 )
2646
2647 if use_embedded:
2648 chapters = [
2649 MediaItemChapter(
2650 position=chapter.chapter_id,
2651 name=chapter.title or f"Chapter {chapter.chapter_id}",
2652 start=chapter.position_start,
2653 end=chapter.position_end,
2654 )
2655 for chapter in tags.chapters
2656 ]
2657 total_duration = try_parse_int(tags.duration) or 0
2658 self.logger.log(
2659 VERBOSE_LOG_LEVEL,
2660 "Audiobook '%s': %d embedded chapters, duration=%d",
2661 tags.album,
2662 len(chapters),
2663 int(total_duration),
2664 )
2665 else:
2666 for position, (chapter_item, chapter_tags) in enumerate(chapter_file_items, start=1):
2667 if chapter_tags.duration is None:
2668 self.logger.warning(
2669 "Chapter file has no duration, skipping: %s",
2670 chapter_item.relative_path,
2671 )
2672 continue
2673 self.logger.debug("Chapter filename: %s", chapter_item.relative_path)
2674 chapters.append(
2675 MediaItemChapter(
2676 position=position,
2677 name=chapter_tags.title,
2678 start=total_duration,
2679 end=total_duration + chapter_tags.duration,
2680 )
2681 )
2682 all_chapter_files.append(
2683 (
2684 chapter_item.relative_path,
2685 chapter_tags.duration,
2686 )
2687 )
2688 total_duration += chapter_tags.duration
2689 sort_method = "alphabetical" if use_alphabetical else "track"
2690 self.logger.log(
2691 VERBOSE_LOG_LEVEL,
2692 "Audiobook '%s': %d files (%s order), duration=%d",
2693 tags.album,
2694 len(chapters),
2695 sort_method,
2696 int(total_duration),
2697 )
2698 # Cache chapter files for streaming
2699 await self.cache.set(
2700 key=audiobook_file_item.relative_path,
2701 data=all_chapter_files,
2702 provider=self.instance_id,
2703 category=CACHE_CATEGORY_AUDIOBOOK_CHAPTERS,
2704 )
2705 return int(total_duration), chapters
2706
2707 async def _get_podcast_metadata(self, podcast_folder: str) -> dict[str, Any]:
2708 """Return metadata for a podcast."""
2709 if (
2710 cache := await self.cache.get(
2711 key=podcast_folder,
2712 provider=self.instance_id,
2713 category=CACHE_CATEGORY_PODCAST_METADATA,
2714 )
2715 ) is not None:
2716 return cast("dict[str, Any]", cache)
2717 data: dict[str, Any] = {}
2718 metadata_file = os.path.join(podcast_folder, "metadata.json")
2719 if await self.exists(metadata_file):
2720 # found json file with metadata
2721 raw = await self._read_file(metadata_file)
2722 data.update(json_loads(raw.decode("utf-8")))
2723 await self.cache.set(
2724 key=podcast_folder,
2725 data=data,
2726 provider=self.instance_id,
2727 category=CACHE_CATEGORY_PODCAST_METADATA,
2728 )
2729 return data
2730
2731 async def _scandir(self, path: str) -> list[FileSystemItem]:
2732 """List directory contents in natural sort order."""
2733 # raw scandir order depends on the underlying filesystem (e.g. hash order
2734 # on ext4) so sort to make browse and folder playback order deterministic
2735 abs_path = self.get_absolute_path(path)
2736 return await asyncio.to_thread(sorted_scandir, self.base_path, abs_path, sort=True)
2737
2738 async def _read_file(self, path: str) -> bytes:
2739 """Read file contents. Override for network storage."""
2740 async with aiofiles.open(self.get_absolute_path(path), mode="rb") as f:
2741 return cast("bytes", await f.read())
2742