navidrome/scanner/walk_dir_tree.go
Junker der Provinz 7c5268c119 fix(scanner): skip symlinks that point back into the folder being scanned - #5334
A symlink pointing at a folder that is already part of the current branch of
the walk (e.g. `music/tracks/music -> ..`) made the scanner walk the same
folders again and again, re-adding the same files until the OS hit its
recursion or path limits.

Each folder now carries its resolved path, with symlinks followed, and the
walk keeps the resolved paths of the branch it is currently in. A child whose
resolved path is already in that branch is a cycle and is skipped. The set is
per branch rather than global, so two sibling folders linking to the same
target are both walked: that is the same folder reached twice, not a loop.

Resolution uses the storage's own resolver where the filesystem is backed by a
real one, and falls back to following the link chain with fs.ReadLink, which
keeps the comparison inside the FS-relative path space.

The tests give the walk a deadline and a folder cap, because an undetected
cycle does not fail, it runs forever. Verified both ways: green with the
detection, and failing on the deadline without it.

Fixes #5334

Signed-off-by: Junker der Provinz <133605895+junkerderprovinz@users.noreply.github.com>
2026-10-02 03:08:46 +02:00

427 lines
15 KiB
Go

package scanner
import (
"context"
"io/fs"
"maps"
"path"
"path/filepath"
"slices"
"sort"
"strings"
"github.com/navidrome/navidrome/conf"
"github.com/navidrome/navidrome/core/storage"
"github.com/navidrome/navidrome/log"
"github.com/navidrome/navidrome/model"
"github.com/navidrome/navidrome/utils"
)
// walkDirTree recursively walks the directory tree starting from the given targetFolders.
// If no targetFolders are provided, it starts from the root folder (".").
// It returns a channel of folderEntry pointers representing each folder found.
func walkDirTree(ctx context.Context, job *scanJob, targetFolders ...string) (<-chan *folderEntry, error) {
results := make(chan *folderEntry)
folders := targetFolders
if len(targetFolders) == 0 {
// No specific folders provided, scan the root folder
folders = []string{"."}
}
go func() {
defer close(results)
for _, folderPath := range folders {
if utils.IsCtxDone(ctx) {
return
}
// Check if target folder exists before walking it
// If it doesn't exist (e.g., deleted between watcher detection and scan execution),
// skip it so it remains in job.lastUpdates and gets handled in following steps
_, err := fs.Stat(job.fs, folderPath)
if err != nil {
log.Warn(ctx, "Scanner: Target folder does not exist.", "path", folderPath, err)
continue
}
// Create checker and push patterns from root to this folder
checker := newIgnoreChecker(job.fs)
err = checker.PushAllParents(ctx, folderPath)
if err != nil {
log.Error(ctx, "Scanner: Error pushing ignore patterns for target folder", "path", folderPath, err)
continue
}
// Recursively walk this folder and all its children
err = walkFolder(ctx, job, newDirRef(job.fs, folderPath), checker, results, map[string]struct{}{})
if utils.IsCtxDone(ctx) {
return
}
if err != nil {
log.Error(ctx, "Scanner: Error walking target folder", "path", folderPath, err)
continue
}
}
log.Debug(ctx, "Scanner: Finished reading target folders", "lib", job.lib.Name, "path", job.lib.Path, "numFolders", job.numFolders.Load())
}()
return results, nil
}
// dirRef is a folder to be walked: its path in the library's filesystem and its resolved
// path, with all symlinks followed. The resolved path is the folder's identity: two paths
// that resolve to the same one are the same folder on disk, which is how symlink cycles
// are detected while walking (see #5334).
type dirRef struct {
path string
realPath string
}
func walkFolder(ctx context.Context, job *scanJob, dir dirRef, checker *IgnoreChecker, results chan<- *folderEntry, branch map[string]struct{}) error {
// Push patterns for this folder onto the stack
_ = checker.Push(ctx, dir.path)
defer checker.Pop() // Pop patterns when leaving this folder
// Keep track of the folders in the current branch, so any symlink pointing back into
// one of them is recognized as a cycle and skipped. Removed again on the way out, so
// two sibling folders linking to the same target are both walked - that is not a
// cycle, just the same folder reached twice.
branch[dir.realPath] = struct{}{}
defer delete(branch, dir.realPath)
folder, children, err := loadDir(ctx, job, dir, checker, branch)
if err != nil {
log.Warn(ctx, "Scanner: Error loading dir. Skipping", "path", dir.path, err)
return nil
}
for _, c := range children {
err := walkFolder(ctx, job, c, checker, results, branch)
if err != nil {
return err
}
}
cleanPath := path.Clean(dir.path)
log.Trace(ctx, "Scanner: Found directory", " path", cleanPath, "audioFiles", maps.Keys(folder.audioFiles),
"images", maps.Keys(folder.imageFiles), "playlists", len(folder.playlistFiles), "imagesUpdatedAt", folder.imagesUpdatedAt,
"updTime", folder.updTime, "modTime", folder.modTime, "numChildren", len(children))
folder.path = cleanPath
folder.elapsed.Start()
select {
case results <- folder:
return nil
case <-ctx.Done():
return ctx.Err()
}
}
func loadDir(ctx context.Context, job *scanJob, dir dirRef, checker *IgnoreChecker, branch map[string]struct{}) (folder *folderEntry, children []dirRef, err error) {
dirPath := dir.path
// Check if directory exists before creating the folder entry
// This is important to avoid removing the folder from lastUpdates if it doesn't exist
dirInfo, err := fs.Stat(job.fs, dirPath)
if err != nil {
log.Warn(ctx, "Scanner: Error stating dir", "path", dirPath, err)
return nil, nil, err
}
// Now that we know the folder exists, create the entry (which removes it from lastUpdates)
folder = job.createFolderEntry(dirPath)
folder.modTime = dirInfo.ModTime()
// Named dirFile rather than dir: dir is now the folder reference this function
// received, and shadowing it here would silently walk the wrong path.
dirFile, err := job.fs.Open(dirPath)
if err != nil {
log.Warn(ctx, "Scanner: Error in Opening directory", "path", dirPath, err)
return folder, children, err
}
defer dirFile.Close()
readDirFile, ok := dirFile.(fs.ReadDirFile)
if !ok {
log.Error(ctx, "Not a directory", "path", dirPath)
return folder, children, err
}
entries := fullReadDir(ctx, readDirFile)
children = make([]dirRef, 0, len(entries))
for _, entry := range entries {
entryPath := path.Join(dirPath, entry.Name())
if checker.ShouldIgnore(ctx, entryPath) {
log.Trace(ctx, "Scanner: Ignoring entry", "path", entryPath)
continue
}
if ctx.Err() != nil {
return folder, children, ctx.Err()
}
isDir, err := isDirOrSymlinkToDir(job.fs, dirPath, entry)
// Skip invalid symlinks
if err != nil {
log.Warn(ctx, "Scanner: Invalid symlink", "dir", entryPath, err)
continue
}
if isIgnoredEntry(entry.Name(), isDir) {
continue
}
if isDir && isDirReadable(ctx, job.fs, entryPath) {
child := newChildDirRef(job.fs, dir, entry)
// A symlink whose target is already in the current branch points back into a
// folder being walked right now: following it would walk the same folders
// forever. Everything else is walked normally.
if _, isCycle := branch[child.realPath]; isCycle {
log.Debug(ctx, "Scanner: Skipping symlink pointing back into a folder being scanned",
"path", entryPath, "target", child.realPath)
continue
}
children = append(children, child)
folder.numSubFolders++
} else {
fileInfo, err := entry.Info()
if err != nil {
log.Warn(ctx, "Scanner: Error getting fileInfo", "name", entry.Name(), err)
return folder, children, err
}
if fileInfo.ModTime().After(folder.modTime) {
folder.modTime = fileInfo.ModTime()
}
name, ok := resolveEntryName(ctx, job.fs, dirPath, entry)
if !ok {
continue
}
switch {
case model.IsAudioFile(name):
folder.audioFiles[entry.Name()] = entry
case model.IsValidPlaylist(name):
folder.playlistFiles[entry.Name()] = entry
case model.IsImageFile(name):
folder.imageFiles[entry.Name()] = entry
folder.imagesUpdatedAt = utils.TimeNewest(folder.imagesUpdatedAt, fileInfo.ModTime(), folder.modTime)
}
}
}
return folder, children, nil
}
// fullReadDir reads all files in the folder, skipping the ones with errors.
// It also detects when it is "stuck" with an error in the same directory over and over.
// In this case, it stops and returns whatever it was able to read until it got stuck.
// See discussion here: https://github.com/navidrome/navidrome/issues/1164#issuecomment-881922850
func fullReadDir(ctx context.Context, dir fs.ReadDirFile) []fs.DirEntry {
var allEntries []fs.DirEntry
var prevErrStr = ""
for {
if ctx.Err() != nil {
return nil
}
entries, err := dir.ReadDir(-1)
allEntries = append(allEntries, entries...)
if err == nil {
break
}
log.Warn(ctx, "Skipping DirEntry", err)
if prevErrStr == err.Error() {
log.Error(ctx, "Scanner: Duplicate DirEntry failure, bailing", err)
break
}
prevErrStr = err.Error()
}
sort.Slice(allEntries, func(i, j int) bool { return allEntries[i].Name() < allEntries[j].Name() })
return allEntries
}
// isDirOrSymlinkToDir returns true if and only if the dirEnt represents a file
// system directory, or a symbolic link to a directory. Note that if the dirEnt
// is not a directory but is a symbolic link, this method will resolve by
// sending a request to the operating system to follow the symbolic link.
// originally copied from github.com/karrick/godirwalk, modified to use dirEntry for
// efficiency for go 1.16 and beyond
func isDirOrSymlinkToDir(fsys fs.FS, baseDir string, dirEnt fs.DirEntry) (bool, error) {
if dirEnt.IsDir() {
return true, nil
}
if dirEnt.Type()&fs.ModeSymlink == 0 {
return false, nil
}
// If symlinks are disabled, return false for symlinks
if !conf.Server.Scanner.FollowSymlinks {
return false, nil
}
// Does this symlink point to a directory?
fileInfo, err := fs.Stat(fsys, path.Join(baseDir, dirEnt.Name()))
if err != nil {
return false, err
}
return fileInfo.IsDir(), nil
}
const maxSymlinkHops = 40
// newDirRef returns the reference for the folder a walk starts at, resolving it in case
// the folder itself is (or is reached through) a symlink.
func newDirRef(fsys fs.FS, dirPath string) dirRef {
dir := dirRef{path: dirPath, realPath: path.Clean(dirPath)}
dir.realPath = resolveDirPath(fsys, dir)
return dir
}
// newChildDirRef returns the reference for a subfolder of dir. Only symlinks need to be
// resolved: a real subfolder is always a new folder, it can never point back into one of
// its own ancestors.
func newChildDirRef(fsys fs.FS, dir dirRef, entry fs.DirEntry) dirRef {
child := dirRef{
path: path.Join(dir.path, entry.Name()),
realPath: path.Join(dir.realPath, entry.Name()),
}
if entry.Type()&fs.ModeSymlink != 0 {
child.realPath = resolveDirPath(fsys, child)
}
return child
}
// resolveDirPath returns the path of the folder dir points to, with all symlinks
// resolved. Storages backed by a real filesystem resolve the whole path at the OS level.
// For any other filesystem the symlink chain is followed with fs.ReadLink, starting from
// the parent's already resolved path, so the walk keeps comparing paths in the same
// (FS-relative) space. If the path cannot be resolved it is returned as is, and the
// folder is walked as any other.
func resolveDirPath(fsys fs.FS, dir dirRef) string {
if resolver, ok := fsys.(storage.SymlinkResolverFS); ok {
target, err := resolver.ResolveSymlink(dir.path)
if err != nil {
return dir.realPath
}
return filepath.ToSlash(target)
}
target, hops := followSymlinkChain(fsys, dir.realPath)
if hops >= maxSymlinkHops {
return dir.realPath
}
return target
}
// followSymlinkChain follows the symlink chain starting at linkPath using fs.ReadLink,
// and returns the last path it could reach, plus the number of hops it took to get there.
// Zero hops means linkPath is not a symlink this filesystem can read; maxSymlinkHops means
// the chain is too long to be resolved and is most likely a loop.
func followSymlinkChain(fsys fs.FS, linkPath string) (string, int) {
cur := linkPath
hop := 0
for ; hop < maxSymlinkHops; hop++ {
target, err := fs.ReadLink(fsys, cur)
if err != nil {
break
}
if path.IsAbs(target) {
// Absolute targets are not valid fs.FS paths, so the next ReadLink fails and
// resolution stops here, leaving cur as the final target.
cur = target
} else {
cur = path.Join(path.Dir(cur), target)
}
}
return cur, hop
}
// resolveEntryName returns the name to classify the entry by, and whether to
// consider it at all. Symlinks are resolved to their final target so the caller
// classifies by the target's extension, not the link's name. Returns ok=false
// when symlinks are disabled or the target can't be resolved.
func resolveEntryName(ctx context.Context, fsys fs.FS, dirPath string, entry fs.DirEntry) (string, bool) {
if entry.Type()&fs.ModeSymlink == 0 {
return entry.Name(), true
}
linkPath := path.Join(dirPath, entry.Name())
if !conf.Server.Scanner.FollowSymlinks {
log.Trace(ctx, "Scanner: Skipping symlink, following is disabled", "path", linkPath)
return "", false
}
// OS-backed filesystems can resolve the whole chain, even when it leaves the FS root
// (e.g. a link into another folder/drive), so the final target is always what gets
// classified. The fs.ReadLink loop below can't see past the root: it classifies by the
// last in-chain name it can reach.
if resolver, ok := fsys.(storage.SymlinkResolverFS); ok {
target, err := resolver.ResolveSymlink(linkPath)
if err != nil {
log.Trace(ctx, "Scanner: Skipping symlink, cannot resolve target", "path", linkPath, err)
return "", false
}
resolved := filepath.Base(target)
log.Trace(ctx, "Scanner: Resolved symlink", "path", linkPath, "target", target, "name", resolved)
return resolved, true
}
cur := linkPath
for hop := 0; hop < maxSymlinkHops; hop++ {
target, err := fs.ReadLink(fsys, cur)
if err != nil {
if hop == 0 {
log.Trace(ctx, "Scanner: Skipping symlink, cannot resolve target", "path", linkPath, err)
return "", false
}
resolved := path.Base(cur)
log.Trace(ctx, "Scanner: Resolved symlink", "path", linkPath, "target", cur, "name", resolved)
return resolved, true
}
if path.IsAbs(target) {
// Absolute targets are not valid fs.FS paths, so the next ReadLink fails and
// resolution stops here, leaving cur as the target to classify by name.
cur = target
} else {
cur = path.Join(path.Dir(cur), target)
}
}
log.Trace(ctx, "Scanner: Skipping symlink, too many hops (possible loop)", "path", linkPath)
return "", false
}
// isDirReadable returns true if the directory represented by dirEnt is readable
func isDirReadable(ctx context.Context, fsys fs.FS, dirPath string) bool {
dir, err := fsys.Open(dirPath)
if err != nil {
log.Warn("Scanner: Skipping unreadable directory", "path", dirPath, err)
return false
}
err = dir.Close()
if err != nil {
log.Warn(ctx, "Scanner: Error closing directory", "path", dirPath, err)
}
return true
}
// List of special directories to ignore
var ignoredDirs = []string{
"$RECYCLE.BIN",
"#snapshot",
"@Recycle",
"@eaDir",
"@Recently-Snapshot",
".git",
".streams",
"lost+found",
}
// isIgnoredEntry returns true if a directory entry with the given name should be
// skipped during scanning. It centralizes all name- and type-based ignore policy:
// - special system directories in ignoredDirs are always ignored;
// - dot-prefixed files are always ignored;
// - dot-prefixed folders are ignored unless Scanner.IgnoreDotFolders is disabled,
// allowing albums like ".Hack Sign" to be scanned when the option is off.
func isIgnoredEntry(name string, isDir bool) bool {
if isDir && isDirIgnored(name) {
return true
}
return isDotEntry(name) && (!isDir || conf.Server.Scanner.IgnoreDotFolders)
}
// isDirIgnored returns true if the directory name is in the explicit ignoredDirs
// blocklist. Used both while walking the tree and by the file watcher.
func isDirIgnored(name string) bool {
return slices.ContainsFunc(ignoredDirs, func(s string) bool { return strings.EqualFold(s, name) })
}
// isDotEntry returns true only for names with exactly one leading dot (the
// convention for hidden entries), e.g. ".hidden". Names with two or more leading
// dots are not considered hidden: "." and ".." are the special self/parent
// references, and anything like "..foo" or "...Album" is a regular name (album
// folders sometimes start with ellipses), so all of these return false.
func isDotEntry(name string) bool {
return name != "." && strings.HasPrefix(name, ".") && !strings.HasPrefix(name, "..")
}