/
/
1"""Some helpers for Filesystem based Musicproviders."""
2
3from __future__ import annotations
4
5import errno
6import hashlib
7import logging
8import os
9import re
10from collections.abc import Iterator
11from dataclasses import dataclass, field
12from pathlib import Path
13
14from music_assistant_models.errors import MediaNotFoundError
15
16from music_assistant.helpers.compare import compare_strings
17from music_assistant.helpers.json import make_utf8_safe
18from music_assistant.helpers.security import is_safe_path
19
20logger = logging.getLogger(__name__)
21
22# number of consecutive unreadable directories that marks the storage itself as gone
23MAX_CONSECUTIVE_SCAN_ERRORS = 10
24
25# number of example paths kept for the scan summary the user gets to see
26MAX_REPORTED_FAILED_PATHS = 5
27
28IGNORE_DIRS = (
29 "recycle",
30 "Recently-Snaphot",
31 "Recently-Snapshot",
32 "#recycle",
33 "System Volume Information",
34 "lost+found",
35 "@eaDir",
36)
37
38
39@dataclass
40class ScanErrors:
41 """
42 Error state of a single (recursive) scan of a filesystem provider.
43
44 Shared by all directory levels of one scan, so a storage that goes away
45 halfway through is detected after a handful of failures instead of failing
46 once per remaining directory.
47
48 - fatal: The error that ended the scan: the provider root itself is unreadable
49 or too many directories failed in a row. Callers abort the sync and mark
50 the provider unavailable.
51 - failed_dirs: Number of directories that could not be read. A scan with
52 failed directories is incomplete, so callers must not run deletions.
53 - failed_entries: Number of files that could not be read or processed. Those
54 files are missing from the scan result too, so they block deletions as well.
55 - failed_paths: The first few paths that could not be read, named in the summary
56 so the user can find them without turning on debug logging.
57 - consecutive_failures: Directories that failed since the last one read
58 successfully, excluding failures that do not point at unreachable storage.
59 """
60
61 fatal: Exception | None = None
62 failed_dirs: int = 0
63 failed_entries: int = 0
64 consecutive_failures: int = 0
65 failed_paths: list[str] = field(default_factory=list)
66
67 @property
68 def aborted(self) -> bool:
69 """Return True if the scan must be stopped."""
70 return self.fatal is not None
71
72 @property
73 def incomplete(self) -> bool:
74 """Return True if the scan missed content that is still on the storage."""
75 return bool(self.failed_dirs or self.failed_entries)
76
77 def describe(self) -> str:
78 """Return a summary of what this scan could not read. Only meaningful when incomplete."""
79 parts = []
80 if self.failed_dirs:
81 parts.append(f"{self.failed_dirs} folder(s)")
82 if self.failed_entries:
83 parts.append(f"{self.failed_entries} file(s)")
84 summary = f"{' and '.join(parts)} could not be read"
85 if self.failed_paths:
86 summary += f" (e.g. {', '.join(self.failed_paths)})"
87 return summary
88
89 def record_dir_read(self) -> None:
90 """Register a directory that was read successfully."""
91 self.consecutive_failures = 0
92
93 def record_dir_error(
94 self,
95 err: Exception,
96 *,
97 is_root: bool,
98 counts_toward_abort: bool = True,
99 path: str | None = None,
100 ) -> None:
101 """
102 Register a directory that could not be read.
103
104 :param err: The error raised while reading the directory.
105 :param is_root: True if the directory is the provider's root path.
106 :param counts_toward_abort: False for an error that leaves the scan incomplete
107 but says nothing about the storage being reachable, such as a folder that
108 is only permission-denied.
109 :param path: Path of the directory, named in the summary shown to the user.
110 """
111 if is_root:
112 self.fatal = err
113 return
114 self.failed_dirs += 1
115 self._remember_path(path)
116 if not counts_toward_abort:
117 return
118 self.consecutive_failures += 1
119 if self.consecutive_failures >= MAX_CONSECUTIVE_SCAN_ERRORS:
120 self.fatal = err
121
122 def record_entry_error(self, err: Exception, path: str | None = None) -> None:
123 """
124 Register a file that could not be read or processed.
125
126 :param err: The error raised while reading or processing the file.
127 :param path: Path of the file, named in the summary shown to the user.
128 """
129 # a file that disappeared between the listing and the read is a normal race
130 # during a long scan, and it really is gone, so deletions may handle it
131 if getattr(err, "errno", None) == errno.ENOENT:
132 return
133 self.failed_entries += 1
134 self._remember_path(path)
135
136 def _remember_path(self, path: str | None) -> None:
137 """Keep the first few failed paths as examples for the user."""
138 if path and len(self.failed_paths) < MAX_REPORTED_FAILED_PATHS:
139 self.failed_paths.append(path)
140
141
142@dataclass
143class FileSystemItem:
144 """
145 Representation of an item (file or directory) on the filesystem.
146
147 - filename: Name (not path) of the file (or directory).
148 - relative_path: Relative path to the item on this filesystem provider.
149 - absolute_path: Absolute path to this item.
150 - is_dir: Boolean if item is directory (not file).
151 - checksum: Checksum for this path (usually last modified time) None for dir.
152 - file_size : File size in number of bytes or None if unknown (or not a file).
153 - created_at: File creation timestamp (Unix epoch) or None for directories.
154 """
155
156 filename: str
157 relative_path: str
158 absolute_path: str
159 is_dir: bool
160 checksum: str | None = None
161 file_size: int | None = None
162 created_at: int | None = None # file creation timestamp (Unix epoch)
163
164 @property
165 def ext(self) -> str | None:
166 """Return file extension."""
167 try:
168 # convert to lowercase to make it case insensitive when comparing
169 return self.filename.rsplit(".", 1)[1].lower()
170 except IndexError:
171 return None
172
173 @property
174 def name(self) -> str:
175 """Return file name (without extension)."""
176 return self.filename.rsplit(".", 1)[0]
177
178 @property
179 def parent_name(self) -> str:
180 """Return the name of this item's parent directory."""
181 # derived from the relative path: the absolute path may be a URL on
182 # network/cloud providers (webdav, cloud filesystems)
183 return Path(self.relative_parent_path).name
184
185 @property
186 def relative_parent_path(self) -> str:
187 """Return relative parent path of this item."""
188 return os.path.dirname(self.relative_path)
189
190 @classmethod
191 def from_dir_entry(cls, entry: os.DirEntry[str], base_path: str) -> FileSystemItem:
192 """
193 Create FileSystemItem from os.DirEntry. NOT Async friendly.
194
195 :raises OSError: If the file cannot be stat'd (e.g., invalid filename encoding).
196 """
197 if entry.is_dir(follow_symlinks=False):
198 return cls(
199 filename=entry.name,
200 relative_path=get_relative_path(base_path, entry.path),
201 absolute_path=entry.path,
202 is_dir=True,
203 checksum=None,
204 file_size=None,
205 )
206 # This can raise OSError for files with invalid encoding (e.g., emojis on SMB mounts)
207 # Let the caller handle the exception
208 stat = entry.stat(follow_symlinks=False)
209 # st_birthtime is available on macOS/Windows, st_ctime on Linux
210 # (on Linux st_ctime is metadata change time, not creation time)
211 created_at = int(getattr(stat, "st_birthtime", stat.st_ctime))
212 return cls(
213 filename=entry.name,
214 relative_path=get_relative_path(base_path, entry.path),
215 absolute_path=entry.path,
216 is_dir=False,
217 checksum=str(int(stat.st_mtime)),
218 file_size=stat.st_size,
219 created_at=created_at,
220 )
221
222
223def get_folder_signature(items: list[FileSystemItem]) -> str:
224 """
225 Return an order-independent digest of the given files' paths, mtimes and sizes.
226
227 Intended as a cache checksum: any file added, removed, replaced or retagged changes it.
228
229 :param items: The files to include in the digest.
230 """
231 parts = sorted(f"{x.relative_path}\0{x.checksum}\0{x.file_size}" for x in items)
232 return hashlib.sha256("\0\0".join(parts).encode()).hexdigest()
233
234
235def get_artist_dir(
236 artist_name: str,
237 album_dir: str | None,
238) -> str | None:
239 """Look for (Album)Artist directory in path of a track (or album)."""
240 if not album_dir:
241 return None
242 parentdir = os.path.dirname(album_dir)
243 # account for disc or album sublevel by ignoring (max) 2 levels if needed
244 matched_dir: str | None = None
245 for _ in range(3):
246 dirname = Path(parentdir).name
247 if compare_strings(artist_name, dirname, False):
248 # literal match
249 # we keep hunting further down to account for the
250 # edge case where the album name has the same name as the artist
251 matched_dir = parentdir
252 parentdir = os.path.dirname(parentdir)
253 return matched_dir
254
255
256def tokenize(input_str: str, delimiters: str) -> list[str]:
257 """Tokenizes the album names or paths."""
258 normalised = re.sub(delimiters, "^^^", input_str)
259 return [x for x in normalised.split("^^^") if x != ""]
260
261
262def _dir_contains_album_name(id3_album_name: str, directory_name: str) -> bool:
263 """
264 Check if a directory name contains an album name.
265
266 This function tokenizes both input strings using different delimiters and
267 checks if the album name is a substring of the directory name.
268
269 First iteration considers the literal dash as one of the separators. The
270 second pass is to catch edge cases where the literal dash is part of the
271 album's name, not an actual separator. For example, an album like 'Aphex
272 Twin - Selected Ambient Works 85-92' would be correctly handled.
273
274 Args:
275 id3_album_name (str): The album name to search for.
276 directory_name (str): The directory name to search in.
277
278 Returns:
279 bool: True if the directory name contains the album name, False otherwise.
280 """
281 for delims in ["[-_ ]", "[_ ]"]:
282 tokenized_album_name = tokenize(id3_album_name, delims)
283 tokenized_dirname = tokenize(directory_name, delims)
284
285 # Exact match, potentially just on the album name
286 # in case artist's name is not included in id3_album_name
287 if all(token in tokenized_dirname for token in tokenized_album_name):
288 return True
289
290 if len(tokenized_album_name) <= len(tokenized_dirname) and compare_strings(
291 "".join(tokenized_album_name),
292 "".join(tokenized_dirname[0 : len(tokenized_album_name)]),
293 False,
294 ):
295 return True
296 return False
297
298
299def get_album_dir(track_dir: str, album_name: str) -> str | None:
300 """Return album/parent directory of a track."""
301 parentdir = track_dir
302 # account for disc sublevel by ignoring 1 level if needed
303 for _ in range(2):
304 dirname = Path(parentdir).name
305 if compare_strings(album_name, dirname, False):
306 # literal match
307 return parentdir
308 if compare_strings(album_name, dirname.split(" - ")[-1], False):
309 # account for ArtistName - AlbumName format in the directory name
310 return parentdir
311 if compare_strings(album_name, dirname.split(" - ")[-1].split("(")[0], False):
312 # account for ArtistName - AlbumName (Version) format in the directory name
313 return parentdir
314
315 if any(sep in dirname for sep in ["-", " ", "_"]) and album_name:
316 album_chunks = album_name.split(" - ", 1)
317 album_name_includes_artist = len(album_chunks) > 1
318 just_album_name = album_chunks[1] if album_name_includes_artist else None
319
320 # attempt matching using tokenized version of path and album name
321 # with _dir_contains_album_name()
322 if just_album_name and _dir_contains_album_name(just_album_name, dirname):
323 return parentdir
324
325 if _dir_contains_album_name(album_name, dirname):
326 return parentdir
327
328 if compare_strings(album_name.split("(", maxsplit=1)[0], dirname, False):
329 # account for AlbumName (Version) format in the album name
330 return parentdir
331 if compare_strings(album_name.split("(", maxsplit=1)[0], dirname.split(" - ")[-1], False):
332 # account for ArtistName - AlbumName (Version) format
333 return parentdir
334 if len(album_name) > 8 and album_name in dirname:
335 # dirname contains album name
336 # (could potentially lead to false positives, hence the length check)
337 return parentdir
338 parentdir = os.path.dirname(parentdir)
339 return None
340
341
342def get_relative_path(base_path: str, path: str) -> str:
343 """Return the relative path string for a path."""
344 if path.startswith(base_path):
345 path = path.split(base_path)[1]
346 for sep in ("/", "\\"):
347 if path.startswith(sep):
348 path = path[1:]
349 return path
350
351
352def get_absolute_path(base_path: str, path: str) -> str:
353 """
354 Return the absolute path for a path, constrained to base_path.
355
356 :raises MediaNotFoundError: If the resolved path escapes base_path
357 (e.g. via ``../`` traversal or an absolute path outside the base).
358 """
359 absolute_path = path if path.startswith(base_path) else os.path.join(base_path, path)
360 if not is_safe_path(absolute_path, base_path):
361 msg = f"Path is outside the configured base directory: {path}"
362 raise MediaNotFoundError(msg)
363 return absolute_path
364
365
366def recursive_iter(
367 path: str,
368 base_path: str,
369 supported_extensions: set[str],
370 log: logging.Logger,
371 scan_errors: ScanErrors | None = None,
372) -> Iterator[FileSystemItem]:
373 """
374 Recursively traverse directory entries yielding supported files.
375
376 :param path: The directory path to scan.
377 :param base_path: The root base path for constructing relative paths.
378 :param supported_extensions: Set of file extensions to include (lowercase, no dot).
379 :param log: Logger instance to use for warnings/debug messages.
380 :param scan_errors: Optional state object collecting the errors raised during this
381 scan. Callers treat ``fatal`` as "provider unreachable" and abort the sync.
382 """
383 if scan_errors is None:
384 scan_errors = ScanErrors()
385 try:
386 scan_iter = os.scandir(path)
387 except OSError as err:
388 if err.errno == errno.EINVAL:
389 log.warning(
390 "Skipping directory '%s' - unsupported characters in path",
391 path,
392 )
393 return
394 log.warning("Unable to scan directory %s: %s", path, err)
395 _record_dir_failure(scan_errors, err, path=path, base_path=base_path, log=log)
396 return
397 entry_error_logged = False
398 with scan_iter:
399 while True:
400 try:
401 item = next(scan_iter)
402 except StopIteration:
403 scan_errors.record_dir_read()
404 break
405 except OSError as err:
406 log.warning("Error while scanning directory %s: %s", path, err)
407 _record_dir_failure(scan_errors, err, path=path, base_path=base_path, log=log)
408 return
409 if (
410 item.name in IGNORE_DIRS
411 or item.name.startswith((".", "_"))
412 or _skip_undecodable_name(item.name, log)
413 ):
414 continue
415 try:
416 is_dir = item.is_dir(follow_symlinks=False)
417 is_file = item.is_file(follow_symlinks=False)
418 except OSError as err:
419 if err.errno == errno.EINVAL:
420 log.warning(
421 "Skipping '%s' - unsupported characters in name",
422 item.name,
423 )
424 else:
425 # the entry may well be a directory, so this can hide a whole subtree
426 entry_error_logged = _record_entry_failure(
427 scan_errors,
428 err,
429 entry_path=item.path,
430 base_path=base_path,
431 log=log,
432 already_logged=entry_error_logged,
433 )
434 continue
435 if is_dir:
436 yield from recursive_iter(
437 item.path,
438 base_path,
439 supported_extensions,
440 log,
441 scan_errors,
442 )
443 if scan_errors.aborted:
444 return
445 elif is_file:
446 if "." not in item.name:
447 continue
448 ext = item.name.rsplit(".", 1)[1].lower()
449 if ext not in supported_extensions:
450 continue
451 try:
452 yield FileSystemItem.from_dir_entry(item, base_path)
453 except OSError as err:
454 if err.errno == errno.EINVAL:
455 log.warning(
456 "Skipping '%s' - unsupported characters in name",
457 item.name,
458 )
459 else:
460 entry_error_logged = _record_entry_failure(
461 scan_errors,
462 err,
463 entry_path=item.path,
464 base_path=base_path,
465 log=log,
466 already_logged=entry_error_logged,
467 )
468
469
470def sorted_scandir(base_path: str, sub_path: str, sort: bool = False) -> list[FileSystemItem]:
471 """
472 Implement os.scandir that returns (optionally) sorted entries.
473
474 Not async friendly!
475 """
476
477 def nat_key(name: str) -> tuple[int | str, ...]:
478 """Sort key for natural sorting, case insensitive to match the frontend sorting."""
479 return tuple(int(s) if s.isdigit() else s.casefold() for s in re.split(r"(\d+)", name))
480
481 if base_path not in sub_path:
482 sub_path = os.path.join(base_path, sub_path)
483 items: list[FileSystemItem] = []
484 try:
485 entries = os.scandir(sub_path)
486 except OSError as err:
487 if err.errno == errno.EINVAL:
488 logger.warning(
489 "Skipping directory '%s' - unsupported characters in path",
490 sub_path,
491 )
492 return items
493 raise
494 with entries:
495 for entry in entries:
496 if (
497 entry.name in IGNORE_DIRS
498 or entry.name.startswith(".")
499 or _skip_undecodable_name(entry.name, logger)
500 ):
501 continue
502 try:
503 is_dir = entry.is_dir(follow_symlinks=False)
504 is_file = entry.is_file(follow_symlinks=False)
505 except OSError as err:
506 if err.errno == errno.EINVAL:
507 logger.warning(
508 "Skipping '%s' - unsupported characters in name",
509 entry.name,
510 )
511 continue
512 if not (is_dir or is_file):
513 continue
514 try:
515 items.append(FileSystemItem.from_dir_entry(entry, base_path))
516 except OSError as err:
517 if err.errno == errno.EINVAL:
518 logger.warning(
519 "Skipping '%s' - unsupported characters in name",
520 entry.name,
521 )
522 else:
523 logger.debug("Skipping '%s' due to OS error: %s", entry.name, err)
524 continue
525
526 if sort:
527 return sorted(
528 items,
529 # sort by (natural) name
530 key=lambda x: nat_key(x.name),
531 )
532 return items
533
534
535def _skip_undecodable_name(name: str, log: logging.Logger) -> bool:
536 """
537 Return True if the given filename is not valid UTF-8 and must be skipped.
538
539 A skipped name is logged in escaped form, so the caller only has to skip it.
540
541 :param name: Name of the file or directory, as returned by the os module.
542 :param log: Logger to report a skipped name on.
543 """
544 # such a path can be neither stored in the database nor sent to a client
545 if name.isascii():
546 return False
547 if (safe_name := make_utf8_safe(name)) == name:
548 return False
549 log.warning("Skipping '%s' - filename is not valid UTF-8", safe_name)
550 return True
551
552
553def _record_entry_failure(
554 scan_errors: ScanErrors,
555 err: OSError,
556 *,
557 entry_path: str,
558 base_path: str,
559 log: logging.Logger,
560 already_logged: bool,
561) -> bool:
562 """Register a directory entry that could not be read and report it once per directory."""
563 # a share that drops mid-listing fails every entry in the directory it was reading,
564 # so only the first one is a warning and the rest are debug to keep the log readable
565 log.log(
566 logging.DEBUG if already_logged else logging.WARNING,
567 "Skipping %s due to OS error: %s",
568 entry_path,
569 err,
570 )
571 scan_errors.record_entry_error(err, get_relative_path(base_path, entry_path))
572 return True
573
574
575def _record_dir_failure(
576 scan_errors: ScanErrors,
577 err: OSError,
578 *,
579 path: str,
580 base_path: str,
581 log: logging.Logger,
582) -> None:
583 """Register a directory that could not be read and report it if the scan gives up."""
584 is_root = path == base_path
585 # a folder we may not read is an ACL problem; the storage itself is still there
586 denied = err.errno in (errno.EACCES, errno.EPERM)
587 scan_errors.record_dir_error(
588 err,
589 is_root=is_root,
590 counts_toward_abort=not denied,
591 path=get_relative_path(base_path, path),
592 )
593 if scan_errors.aborted and not is_root:
594 log.error(
595 "Stopping the scan of %s: %d folders in a row could not be read",
596 base_path,
597 scan_errors.consecutive_failures,
598 )
599