package service import ( "bytes" "encoding/json" "fmt" "log/slog" "os" "path/filepath" "sort" "strconv" "strings" "sync" "time" "github.com/BurntSushi/toml" "vidarchive/internal/models" ) const ( itemMarkerName = ".vidarchive-item.toml" subtitlesDirName = "subtitles" ) var mediaExts = map[string]struct{}{ ".mp4": {}, ".webm": {}, ".mkv": {}, ".avi": {}, ".mov": {}, ".mp3": {}, ".wav": {}, ".flac": {}, ".aac": {}, ".opus": {}, ".m4a": {}, } var audioExts = map[string]struct{}{ ".mp3": {}, ".wav": {}, ".flac": {}, ".aac": {}, ".opus": {}, ".m4a": {}, } type LibraryService struct { libraryDir string ffmpegPath string ffprobePath string // thumbLocks serializes extraction per media file, so two callers never write // the same temp file at once. Entries are reference counted: evicting a lock // someone still holds would let the next caller take a fresh one and // reintroduce the race. thumbMu sync.Mutex thumbLocks map[string]*refLock thumbSem chan struct{} // thumbFailed remembers paths whose extraction already failed this run, so // ffmpeg is not re-run on every request. An evicted path costs one retry. thumbFailed *lru[bool] // scanCache memoizes item scans for a short TTL so the listing page does not // re-parse every marker + info.json per request. Only scans are cached: the // directory listing is always read fresh, so added and removed items appear // immediately. scanCache *lru[scanCacheEntry] scanTTL time.Duration } type scanCacheEntry struct { item *models.LibraryItem at time.Time } const scanCacheTTL = 10 * time.Second // maxCachedPaths bounds each per-path cache. Unbounded, both grow with every // distinct file touched over the process lifetime. const maxCachedPaths = 1024 func NewLibraryService(libraryDir, ffmpegPath, ffprobePath string) *LibraryService { return &LibraryService{ libraryDir: libraryDir, ffmpegPath: ffmpegPath, ffprobePath: ffprobePath, thumbLocks: make(map[string]*refLock), thumbSem: make(chan struct{}, maxConcurrentThumbnails), thumbFailed: newLRU[bool](maxCachedPaths), scanCache: newLRU[scanCacheEntry](maxCachedPaths), scanTTL: scanCacheTTL, } } func (s *LibraryService) getCachedScan(relPath string) (*models.LibraryItem, bool) { if s.scanTTL <= 0 { return nil, false } e, ok := s.scanCache.Get(relPath) if !ok || time.Since(e.at) > s.scanTTL { return nil, false } return e.item, true } func (s *LibraryService) putCachedScan(relPath string, item *models.LibraryItem) { if s.scanTTL <= 0 { return } s.scanCache.Put(relPath, scanCacheEntry{item: item, at: time.Now()}) } func (s *LibraryService) evictCachedScan(relPath string) { s.scanCache.Delete(relPath) } // evictCachedDir keys off the symlink-resolved root, because that is what // resolveItemDir stored the entry under. Against the raw libraryDir it misses // whenever that is a symlink. func (s *LibraryService) evictCachedDir(itemDir string) { rel, err := filepath.Rel(s.libraryRoot(), itemDir) if err != nil { return } s.evictCachedScan(filepath.ToSlash(rel)) } // refLock is a mutex plus the number of callers holding or waiting for it. type refLock struct { mu sync.Mutex refs int } func (s *LibraryService) lockThumbFile(path string) func() { s.thumbMu.Lock() l, ok := s.thumbLocks[path] if !ok { l = &refLock{} s.thumbLocks[path] = l } l.refs++ s.thumbMu.Unlock() l.mu.Lock() return func() { l.mu.Unlock() s.thumbMu.Lock() l.refs-- if l.refs == 0 { delete(s.thumbLocks, path) } s.thumbMu.Unlock() } } func (s *LibraryService) scannedItem(itemDir, relPath string) (*models.LibraryItem, error) { if item, ok := s.getCachedScan(relPath); ok { return item, nil } item, err := s.scanItem(itemDir, relPath) if err != nil { return nil, err } s.putCachedScan(relPath, item) return item, nil } // ResolveWithinLibrary rejects anything that escapes the library root. The // directory need not exist yet, so it also works for a download's output dir. func (s *LibraryService) ResolveWithinLibrary(relPath string) (string, error) { return s.resolveItemDir(filepath.Clean(relPath)) } func (s *LibraryService) libraryRoot() string { if base, err := filepath.EvalSymlinks(s.libraryDir); err == nil { return base } return filepath.Clean(s.libraryDir) } func (s *LibraryService) resolveItemDir(relPath string) (string, error) { relPath = strings.Trim(relPath, string(filepath.Separator)) base := s.libraryRoot() if relPath == "" || relPath == "." { return base, nil } itemDir := filepath.Join(base, relPath) cleanDir, err := filepath.EvalSymlinks(itemDir) if err != nil { cleanDir = filepath.Clean(itemDir) } if !strings.HasPrefix(cleanDir, base+string(filepath.Separator)) && cleanDir != base { return "", fmt.Errorf("invalid path") } // Return the path that passed the boundary check, so callers do I/O on // exactly that one. return cleanDir, nil } func (s *LibraryService) GetAll(path, sortBy, filter string) ([]*models.LibraryItem, []string, error) { if path != "" && !strings.HasSuffix(path, "/") { path += "/" } dir, err := s.resolveItemDir(path) if err != nil { slog.Warn("library listing: invalid path", "path", path, "err", err) return nil, nil, nil } entries, err := os.ReadDir(dir) if err != nil { if os.IsNotExist(err) { return nil, nil, nil } return nil, nil, err } var items []*models.LibraryItem var folders []string for _, entry := range entries { if !entry.IsDir() { continue } name := entry.Name() if name == subtitlesDirName { continue } itemDir := filepath.Join(dir, name) relPath := filepath.ToSlash(filepath.Join(path, name)) markerPath := filepath.Join(itemDir, itemMarkerName) if _, err := os.Stat(markerPath); err == nil { item, err := s.scannedItem(itemDir, relPath) if err != nil { slog.Warn("failed to scan item", "path", relPath, "err", err) continue } if filter != "" && !strings.Contains(strings.ToLower(item.Name), strings.ToLower(filter)) && !strings.Contains(strings.ToLower(item.RelPath), strings.ToLower(filter)) { continue } items = append(items, item) } else { folders = append(folders, name) } } switch sortBy { case "title": sort.Slice(items, func(i, j int) bool { return strings.ToLower(items[i].Name) < strings.ToLower(items[j].Name) }) case "duration": sort.Slice(items, func(i, j int) bool { return items[i].Duration > items[j].Duration }) case "date": fallthrough default: // Stat each item once up front rather than twice per comparison. modTime := make(map[string]int64, len(items)) for _, it := range items { if info, err := os.Stat(it.DirPath); err == nil { modTime[it.RelPath] = info.ModTime().UnixNano() } } sort.Slice(items, func(i, j int) bool { return modTime[items[i].RelPath] > modTime[items[j].RelPath] }) } sort.Strings(folders) return items, folders, nil } func (s *LibraryService) GetByRelPath(relPath string) (*models.LibraryItem, error) { relPath = strings.Trim(relPath, "/") if item, ok := s.getCachedScan(relPath); ok { return item, nil } itemDir, err := s.resolveItemDir(relPath) if err != nil { return nil, fmt.Errorf("item not found") } if _, err := os.Stat(filepath.Join(itemDir, itemMarkerName)); err != nil { return nil, fmt.Errorf("item not found") } return s.scannedItem(itemDir, relPath) } func (s *LibraryService) scanItem(itemDir, relPath string) (*models.LibraryItem, error) { metadata, err := s.readMetadata(itemDir) if err != nil { return nil, err } mediaFiles, infoJSONPath, err := s.listItemFiles(itemDir) if err != nil { return nil, err } var info map[string]interface{} if infoJSONPath != "" { data, err := os.ReadFile(infoJSONPath) if err == nil { if err := json.Unmarshal(data, &info); err != nil { slog.Warn("ignoring malformed info.json", "path", infoJSONPath, "err", err) } } } // Only derived metadata makes this dirty, so a plain listing does not rewrite // the marker on every read. dirty := false if metadata.Name == "" { if title, ok := infoString(info, "title"); ok && title != "" { metadata.Name = title } else if len(mediaFiles) > 0 { metadata.Name = mediaFileStem(mediaFiles[0]) } else { metadata.Name = filepath.Base(itemDir) } dirty = true } if metadata.SourceURL == "" { dirty = backfillString(&metadata.SourceURL, info, "webpage_url", "url") || dirty } if metadata.Description == "" { dirty = backfillString(&metadata.Description, info, "description") || dirty } // Backfilling the video id gives pre-existing items an identity on their next // scan; subscriptions match and prune by it. if metadata.VideoID == "" { dirty = backfillString(&metadata.VideoID, info, "id") || dirty } if metadata.FileDurations == nil { metadata.FileDurations = make(map[string]int) } // Durations come from the marker (written at import), plus info.json for a // single-file item. No ffprobe here: keeping it off the scan path is the point // of the marker cache. A file without a known duration shows no badge. for i := range mediaFiles { mf := &mediaFiles[i] if d, ok := metadata.FileDurations[mf.Filename]; ok { mf.Duration = d continue } if len(mediaFiles) == 1 { if d, ok := infoDuration(info); ok && d > 0 { mf.Duration = d metadata.FileDurations[mf.Filename] = d dirty = true } } } // The item-level duration only feeds the "duration" sort; nothing displays it. total := 0 for _, mf := range mediaFiles { if mf.Duration > 0 { total += mf.Duration } } item := &models.LibraryItem{ Name: metadata.Name, RelPath: relPath, DirPath: itemDir, SourceURL: metadata.SourceURL, Duration: total, Description: metadata.Description, YtdlpFlags: metadata.YtdlpFlags, MediaFiles: mediaFiles, } if dirty { if err := s.writeMetadata(itemDir, metadata); err != nil { slog.Warn("failed to persist derived metadata", "dir", itemDir, "err", err) } } return item, nil } // backfillString takes the first non-empty value among keys and reports whether // it changed anything. An empty value must not count as a change: the marker // would be rewritten on every scan forever without ever gaining a value. func backfillString(field *string, info map[string]interface{}, keys ...string) bool { for _, k := range keys { if v, ok := infoString(info, k); ok && v != "" { *field = v return true } } return false } func (s *LibraryService) readMetadata(itemDir string) (models.ItemMetadata, error) { markerPath := filepath.Join(itemDir, itemMarkerName) var metadata models.ItemMetadata data, err := os.ReadFile(markerPath) if err == nil { if _, err := toml.Decode(string(data), &metadata); err != nil { slog.Warn("failed to parse item marker", "path", markerPath, "err", err) } } return metadata, nil } func (s *LibraryService) writeMetadata(itemDir string, metadata models.ItemMetadata) error { var buf bytes.Buffer if err := toml.NewEncoder(&buf).Encode(metadata); err != nil { return err } return atomicWrite(filepath.Join(itemDir, itemMarkerName), buf.Bytes(), markerFileMode) } // markerFileMode matches the info.json sidecar: hand-editable by the operator. const markerFileMode = 0o644 // atomicWrite writes via a uniquely-named temp file in the same directory, so a // crash can't leave a truncated file and two writers can't collide. func atomicWrite(path string, data []byte, mode os.FileMode) error { f, err := os.CreateTemp(filepath.Dir(path), ".vidarchive-*.tmp") if err != nil { return err } tmpPath := f.Name() // A no-op once the rename below has succeeded. defer os.Remove(tmpPath) if err := f.Chmod(mode); err != nil { f.Close() return err } if _, err := f.Write(data); err != nil { f.Close() return err } // Close explicitly: a deferred close would hide a flush error on the write. if err := f.Close(); err != nil { return err } return os.Rename(tmpPath, path) } func (s *LibraryService) listItemFiles(itemDir string) ([]models.MediaFile, string, error) { entries, err := os.ReadDir(itemDir) if err != nil { return nil, "", err } var mediaFiles []models.MediaFile var infoJSONFiles []string for _, entry := range entries { if entry.IsDir() { continue } name := entry.Name() path := filepath.Join(itemDir, name) ext := strings.ToLower(filepath.Ext(name)) if name == itemMarkerName { continue } if name == "info.json" || strings.HasSuffix(name, ".info.json") { infoJSONFiles = append(infoJSONFiles, path) continue } if _, ok := mediaExts[ext]; !ok { continue } _, isAudio := audioExts[ext] mediaFiles = append(mediaFiles, models.MediaFile{ Filename: name, Filepath: path, IsAudio: isAudio, Duration: -1, }) } sort.Slice(mediaFiles, func(i, j int) bool { return mediaFiles[i].Filepath < mediaFiles[j].Filepath }) var infoJSONPath string if len(infoJSONFiles) > 0 { sort.Strings(infoJSONFiles) infoJSONPath = infoJSONFiles[0] if len(infoJSONFiles) > 1 { slog.Warn("multiple info.json files", "dir", itemDir, "using", infoJSONPath) } } return mediaFiles, infoJSONPath, nil } func infoString(info map[string]interface{}, key string) (string, bool) { if info == nil { return "", false } // Only genuine strings: stringifying a map or slice would store junk like // "map[...]" into the field. if s, ok := info[key].(string); ok { return s, true } return "", false } func infoDuration(info map[string]interface{}) (int, bool) { if info == nil { return 0, false } v, ok := info["duration"] if !ok { return 0, false } switch n := v.(type) { case float64: return int(n + 0.5), true case string: if f, err := strconv.ParseFloat(n, 64); err == nil { return int(f + 0.5), true } } return 0, false } func mediaFileStem(mf models.MediaFile) string { return strings.TrimSuffix(filepath.Base(mf.Filename), filepath.Ext(mf.Filename)) } // primaryMediaFile picks the largest video file, or the largest file overall if // the item has no video. It backs both the item thumbnail and the default // metadata target, which keeps those two consistent. func primaryMediaFile(item *models.LibraryItem) *models.MediaFile { size := func(mf *models.MediaFile) int64 { if info, err := os.Stat(mf.Filepath); err == nil { return info.Size() } return 0 } var best *models.MediaFile for i := range item.MediaFiles { mf := &item.MediaFiles[i] switch { case best == nil: best = mf case best.IsAudio && !mf.IsAudio: best = mf case best.IsAudio == mf.IsAudio && size(mf) > size(best): best = mf } } return best } func (s *LibraryService) GetMediaFile(relPath, filename string) (string, error) { item, err := s.GetByRelPath(relPath) if err != nil { return "", err } for _, mf := range item.MediaFiles { if mf.Filename == filename { return mf.Filepath, nil } } return "", fmt.Errorf("media file not found") } func (s *LibraryService) Delete(relPath string) error { itemDir, err := s.resolveItemDir(relPath) if err != nil { return err } // resolveItemDir maps ""/"." to the root and resolves symlinks, so the root is // reachable via a URL-encoded slash, "sub/..", or a symlink pointing back at // it. Deleting that would wipe the entire library. if itemDir == s.libraryRoot() { return fmt.Errorf("refusing to delete library root") } // Evict so the deletion shows immediately instead of at TTL expiry. s.evictCachedScan(strings.Trim(relPath, "/")) return os.RemoveAll(itemDir) } // FindByVideoID scans only baseDir, the folder one subscription owns. Matching // on the yt-dlp video id alone is safe there: a single source can't collide with // another extractor's ids. func (s *LibraryService) FindByVideoID(baseDir, id string) (string, bool) { if id == "" { return "", false } var match string // An unreadable base dir simply means no match here. _ = s.eachItemDir(baseDir, "FindByVideoID", func(itemDir string, meta models.ItemMetadata) bool { if meta.VideoID != id { return true } match = itemDir return false }) return match, match != "" } // eachItemDir calls fn for each marked item directly under baseDir until fn // returns false. Unreadable items are skipped: one bad item must not abort a // scan. logLabel names the caller in those skip messages. func (s *LibraryService) eachItemDir(baseDir, logLabel string, fn func(itemDir string, meta models.ItemMetadata) bool) error { entries, err := os.ReadDir(baseDir) if err != nil { return err } for _, entry := range entries { if !entry.IsDir() || entry.Name() == subtitlesDirName { continue } itemDir := filepath.Join(baseDir, entry.Name()) if _, err := os.Stat(filepath.Join(itemDir, itemMarkerName)); err != nil { continue } meta, err := s.readMetadata(itemDir) if err != nil { slog.Warn("skipping item", "op", logLabel, "dir", itemDir, "err", err) continue } if !fn(itemDir, meta) { return nil } } return nil } // PruneToIDSet deletes items under baseDir whose identity key is not in keep, // mirroring a subscription's source. An item without an identity key is left // alone: never delete what can't be positively identified. func (s *LibraryService) PruneToIDSet(baseDir string, keep map[string]bool) (int, error) { removed := 0 err := s.eachItemDir(baseDir, "PruneToIDSet", func(itemDir string, meta models.ItemMetadata) bool { if meta.VideoID == "" || keep[meta.VideoID] { return true } s.evictCachedDir(itemDir) if err := os.RemoveAll(itemDir); err != nil { slog.Warn("prune failed to remove item", "dir", itemDir, "err", err) return true } removed++ return true }) return removed, err }