package service import ( "bytes" "encoding/json" "fmt" "log" "os" "path/filepath" "sort" "strconv" "strings" "sync" "time" "github.com/BurntSushi/toml" "vidarchive/internal/models" ) const ( itemMarkerName = ".vidarchive-item.toml" subtitlesDirName = "subtitles" ) var mediaExts = map[string]struct{}{ ".mp4": {}, ".webm": {}, ".mkv": {}, ".avi": {}, ".mov": {}, ".mp3": {}, ".wav": {}, ".flac": {}, ".aac": {}, ".opus": {}, ".m4a": {}, } var audioExts = map[string]struct{}{ ".mp3": {}, ".wav": {}, ".flac": {}, ".aac": {}, ".opus": {}, ".m4a": {}, } type LibraryService struct { libraryDir string // ffmpegPath/ffprobePath are the binaries used for thumbnail extraction, // subtitle conversion, and media probing. Configurable so non-PATH installs // (e.g. a pinned build) can be pointed at directly. ffmpegPath string ffprobePath string // thumbLocks holds a per-media-file mutex serializing extraction so two // callers never write the same temp file at once. Entries are deliberately // never pruned: dropping one would reintroduce the race it prevents. thumbLocks sync.Map thumbSem chan struct{} // thumbFailed records media filepaths whose extraction already failed this // run, so we trust ffmpeg's verdict and don't re-run it on every request. thumbFailed sync.Map // scanCache memoizes scanned items for a short TTL so the listing page (which // fans out one thumbnail request per media file) and quick auto-refreshes // don't re-parse each item's marker + info.json on every request. Only item // scans are cached — the directory listing itself is always read fresh, so // newly added/removed items and subfolders appear immediately. scanMu sync.Mutex scanCache map[string]scanCacheEntry scanTTL time.Duration } type scanCacheEntry struct { item *models.LibraryItem at time.Time } // scanCacheTTL is how long a scanned item is reused before being re-read. const scanCacheTTL = 10 * time.Second func NewLibraryService(libraryDir, ffmpegPath, ffprobePath string) *LibraryService { return &LibraryService{ libraryDir: libraryDir, ffmpegPath: ffmpegPath, ffprobePath: ffprobePath, thumbSem: make(chan struct{}, maxConcurrentThumbnails), scanCache: make(map[string]scanCacheEntry), scanTTL: scanCacheTTL, } } func (s *LibraryService) getCachedScan(relPath string) (*models.LibraryItem, bool) { if s.scanTTL <= 0 { return nil, false } s.scanMu.Lock() defer s.scanMu.Unlock() e, ok := s.scanCache[relPath] if !ok || time.Since(e.at) > s.scanTTL { return nil, false } return e.item, true } func (s *LibraryService) putCachedScan(relPath string, item *models.LibraryItem) { if s.scanTTL <= 0 { return } s.scanMu.Lock() s.scanCache[relPath] = scanCacheEntry{item: item, at: time.Now()} s.scanMu.Unlock() } func (s *LibraryService) evictCachedScan(relPath string) { s.scanMu.Lock() delete(s.scanCache, relPath) s.scanMu.Unlock() } // scannedItem returns a cached scan if fresh, otherwise scans and caches it. func (s *LibraryService) scannedItem(itemDir, relPath string) (*models.LibraryItem, error) { if item, ok := s.getCachedScan(relPath); ok { return item, nil } item, err := s.scanItem(itemDir, relPath) if err != nil { return nil, err } s.putCachedScan(relPath, item) return item, nil } // ResolveWithinLibrary resolves a caller-supplied relative directory against the // library root and rejects anything that escapes it. The directory need not // exist yet, so it is safe to use when choosing a download's output location. func (s *LibraryService) ResolveWithinLibrary(relPath string) (string, error) { return s.resolveItemDir(filepath.Clean(relPath)) } // libraryRoot returns the symlink-resolved library root. func (s *LibraryService) libraryRoot() string { if base, err := filepath.EvalSymlinks(s.libraryDir); err == nil { return base } return filepath.Clean(s.libraryDir) } func (s *LibraryService) resolveItemDir(relPath string) (string, error) { relPath = strings.Trim(relPath, string(filepath.Separator)) base := s.libraryRoot() if relPath == "" || relPath == "." { return base, nil } itemDir := filepath.Join(base, relPath) cleanDir, err := filepath.EvalSymlinks(itemDir) if err != nil { cleanDir = filepath.Clean(itemDir) } if !strings.HasPrefix(cleanDir, base+string(filepath.Separator)) && cleanDir != base { return "", fmt.Errorf("invalid path") } // Return the cleaned/symlink-resolved path we just validated, so callers do // I/O on exactly the path that passed the boundary check. return cleanDir, nil } func (s *LibraryService) GetAll(path, sortBy, filter string) ([]*models.LibraryItem, []string, error) { if path != "" && !strings.HasSuffix(path, "/") { path += "/" } dir, err := s.resolveItemDir(path) if err != nil { log.Printf("GetAll: invalid path %q: %v", path, err) return nil, nil, nil } entries, err := os.ReadDir(dir) if err != nil { if os.IsNotExist(err) { return nil, nil, nil } return nil, nil, err } var items []*models.LibraryItem var folders []string for _, entry := range entries { if !entry.IsDir() { continue } name := entry.Name() if name == subtitlesDirName { continue } itemDir := filepath.Join(dir, name) relPath := filepath.ToSlash(filepath.Join(path, name)) markerPath := filepath.Join(itemDir, itemMarkerName) if _, err := os.Stat(markerPath); err == nil { item, err := s.scannedItem(itemDir, relPath) if err != nil { log.Printf("warning: failed to scan item %s: %v", relPath, err) continue } if filter != "" && !strings.Contains(strings.ToLower(item.Name), strings.ToLower(filter)) && !strings.Contains(strings.ToLower(item.RelPath), strings.ToLower(filter)) { continue } items = append(items, item) } else { folders = append(folders, name) } } switch sortBy { case "title": sort.Slice(items, func(i, j int) bool { return strings.ToLower(items[i].Name) < strings.ToLower(items[j].Name) }) case "duration": sort.Slice(items, func(i, j int) bool { return items[i].Duration > items[j].Duration }) case "date": fallthrough default: // Stat each item once up front rather than twice per comparison. modTime := make(map[string]int64, len(items)) for _, it := range items { if info, err := os.Stat(it.DirPath); err == nil { modTime[it.RelPath] = info.ModTime().UnixNano() } } sort.Slice(items, func(i, j int) bool { return modTime[items[i].RelPath] > modTime[items[j].RelPath] }) } sort.Strings(folders) return items, folders, nil } // GetByRelPath returns the item at relPath, reusing a recent cached scan when // available (see scanCache). func (s *LibraryService) GetByRelPath(relPath string) (*models.LibraryItem, error) { relPath = strings.Trim(relPath, "/") if item, ok := s.getCachedScan(relPath); ok { return item, nil } itemDir, err := s.resolveItemDir(relPath) if err != nil { return nil, fmt.Errorf("item not found") } if _, err := os.Stat(filepath.Join(itemDir, itemMarkerName)); err != nil { return nil, fmt.Errorf("item not found") } return s.scannedItem(itemDir, relPath) } func (s *LibraryService) scanItem(itemDir, relPath string) (*models.LibraryItem, error) { metadata, err := s.readMetadata(itemDir) if err != nil { return nil, err } mediaFiles, infoJSONPath, err := s.listItemFiles(itemDir) if err != nil { return nil, err } var info map[string]interface{} if infoJSONPath != "" { data, err := os.ReadFile(infoJSONPath) if err == nil { if err := json.Unmarshal(data, &info); err != nil { log.Printf("scanItem: ignoring malformed %s: %v", infoJSONPath, err) } } } // dirty tracks whether we derived any new metadata worth persisting, so a // plain listing or detail view doesn't rewrite the marker file on every read. dirty := false if metadata.Name == "" { if title, ok := infoString(info, "title"); ok && title != "" { metadata.Name = title } else if len(mediaFiles) > 0 { metadata.Name = mediaFileStem(mediaFiles[0]) } else { metadata.Name = filepath.Base(itemDir) } dirty = true } if metadata.SourceURL == "" { dirty = backfillString(&metadata.SourceURL, info, "webpage_url", "url") || dirty } if metadata.Description == "" { dirty = backfillString(&metadata.Description, info, "description") || dirty } // Backfill the stable identity (yt-dlp's video id) from info.json so // pre-existing items gain an identity on their next scan. Subscriptions match // and prune items by this id (see FindByVideoID / PruneToIDSet). if metadata.VideoID == "" { dirty = backfillString(&metadata.VideoID, info, "id") || dirty } if metadata.FileDurations == nil { metadata.FileDurations = make(map[string]int) } // Per-file durations come from the marker's file_durations map (populated at // import time). For a single-file item we also seed it from info.json's // duration, which covers the common case without a probe. We deliberately do // not run ffprobe here — keeping it off the scan/listing path is the point of // the marker cache. Files without a known duration simply show no badge. for i := range mediaFiles { mf := &mediaFiles[i] if d, ok := metadata.FileDurations[mf.Filename]; ok { mf.Duration = d continue } if len(mediaFiles) == 1 { if d, ok := infoDuration(info); ok && d > 0 { mf.Duration = d metadata.FileDurations[mf.Filename] = d dirty = true } } } // The item-level duration is the sum of known per-file durations, used only // for the "duration" sort — there is no single "overall" duration shown. total := 0 for _, mf := range mediaFiles { if mf.Duration > 0 { total += mf.Duration } } item := &models.LibraryItem{ Name: metadata.Name, RelPath: relPath, DirPath: itemDir, SourceURL: metadata.SourceURL, Duration: total, Description: metadata.Description, YtdlpFlags: metadata.YtdlpFlags, MediaFiles: mediaFiles, } if dirty { if err := s.writeMetadata(itemDir, metadata); err != nil { log.Printf("scanItem: failed to persist derived metadata for %s: %v", itemDir, err) } } return item, nil } // backfillString sets *field from the first non-empty string value among // info's keys, reporting whether it changed anything. An empty value in // info.json must not count as a change: the marker would be rewritten on every // scan forever without ever gaining a value. func backfillString(field *string, info map[string]interface{}, keys ...string) bool { for _, k := range keys { if v, ok := infoString(info, k); ok && v != "" { *field = v return true } } return false } func (s *LibraryService) readMetadata(itemDir string) (models.ItemMetadata, error) { markerPath := filepath.Join(itemDir, itemMarkerName) var metadata models.ItemMetadata data, err := os.ReadFile(markerPath) if err == nil { if _, err := toml.Decode(string(data), &metadata); err != nil { log.Printf("warning: failed to parse %s: %v", markerPath, err) } } return metadata, nil } func (s *LibraryService) writeMetadata(itemDir string, metadata models.ItemMetadata) error { var buf bytes.Buffer if err := toml.NewEncoder(&buf).Encode(metadata); err != nil { return err } return atomicWrite(filepath.Join(itemDir, itemMarkerName), buf.Bytes(), markerFileMode) } // markerFileMode matches the info.json sidecar: both sit in the library next to // the media and are meant to be readable (and hand-editable) by the operator. const markerFileMode = 0o644 // atomicWrite writes data to path via a uniquely-named temp file in the same // directory and a rename, so a crash mid-write can't leave a truncated file and // two concurrent writers never collide on a shared temp path. func atomicWrite(path string, data []byte, mode os.FileMode) error { f, err := os.CreateTemp(filepath.Dir(path), ".vidarchive-*.tmp") if err != nil { return err } tmpPath := f.Name() // A no-op once the rename below has succeeded. defer os.Remove(tmpPath) if err := f.Chmod(mode); err != nil { f.Close() return err } if _, err := f.Write(data); err != nil { f.Close() return err } // Close explicitly: a deferred close would hide a flush error on the write. if err := f.Close(); err != nil { return err } return os.Rename(tmpPath, path) } func (s *LibraryService) listItemFiles(itemDir string) ([]models.MediaFile, string, error) { entries, err := os.ReadDir(itemDir) if err != nil { return nil, "", err } var mediaFiles []models.MediaFile var infoJSONFiles []string for _, entry := range entries { if entry.IsDir() { continue } name := entry.Name() path := filepath.Join(itemDir, name) ext := strings.ToLower(filepath.Ext(name)) if name == itemMarkerName { continue } if name == "info.json" || strings.HasSuffix(name, ".info.json") { infoJSONFiles = append(infoJSONFiles, path) continue } if _, ok := mediaExts[ext]; !ok { continue } _, isAudio := audioExts[ext] mediaFiles = append(mediaFiles, models.MediaFile{ Filename: name, Filepath: path, IsAudio: isAudio, Duration: -1, }) } sort.Slice(mediaFiles, func(i, j int) bool { return mediaFiles[i].Filepath < mediaFiles[j].Filepath }) var infoJSONPath string if len(infoJSONFiles) > 0 { sort.Strings(infoJSONFiles) infoJSONPath = infoJSONFiles[0] if len(infoJSONFiles) > 1 { log.Printf("warning: multiple info.json files in %s, using %s", itemDir, infoJSONPath) } } return mediaFiles, infoJSONPath, nil } func infoString(info map[string]interface{}, key string) (string, bool) { if info == nil { return "", false } // Only accept genuine strings: title/url/description are always strings in // yt-dlp output, and stringifying an arbitrary JSON value (map, slice) would // store junk like "map[...]" into the field. if s, ok := info[key].(string); ok { return s, true } return "", false } func infoDuration(info map[string]interface{}) (int, bool) { if info == nil { return 0, false } v, ok := info["duration"] if !ok { return 0, false } switch n := v.(type) { case float64: return int(n + 0.5), true case string: if f, err := strconv.ParseFloat(n, 64); err == nil { return int(f + 0.5), true } } return 0, false } func mediaFileStem(mf models.MediaFile) string { return strings.TrimSuffix(filepath.Base(mf.Filename), filepath.Ext(mf.Filename)) } // primaryMediaFile picks the representative file for an item: the largest video // file, or — if there are none — the largest file overall. Returns nil for an // item with no media files. Used for the item-level thumbnail and as the default // target for metadata, keeping those two consistent. func primaryMediaFile(item *models.LibraryItem) *models.MediaFile { size := func(mf *models.MediaFile) int64 { if info, err := os.Stat(mf.Filepath); err == nil { return info.Size() } return 0 } var best *models.MediaFile for i := range item.MediaFiles { mf := &item.MediaFiles[i] switch { case best == nil: best = mf case best.IsAudio && !mf.IsAudio: // Prefer any video over audio. best = mf case best.IsAudio == mf.IsAudio && size(mf) > size(best): best = mf } } return best } func (s *LibraryService) GetMediaFile(relPath, filename string) (string, error) { item, err := s.GetByRelPath(relPath) if err != nil { return "", err } for _, mf := range item.MediaFiles { if mf.Filename == filename { return mf.Filepath, nil } } return "", fmt.Errorf("media file not found") } func (s *LibraryService) Delete(relPath string) error { itemDir, err := s.resolveItemDir(relPath) if err != nil { return err } // resolveItemDir maps ""/"." to the library root and resolves symlinks, so a // result equal to the root (reachable via a URL-encoded slash, "sub/..", or a // symlink pointing back at the root) must be refused — deleting it would // wipe the entire library. if itemDir == s.libraryRoot() { return fmt.Errorf("refusing to delete library root") } // Evict the cached scan so the deletion is reflected immediately rather than // lingering until the TTL expires. s.evictCachedScan(strings.Trim(relPath, "/")) return os.RemoveAll(itemDir) } // FindByVideoID returns the absolute directory of the item under baseDir whose // marker matches the given yt-dlp video id, scanning only that directory (the // subscription's owned folder). Matching on the id alone is safe here because // each subscription owns a single source, so ids don't collide across // extractors within the folder. ok is false when no match is found or id is // empty. func (s *LibraryService) FindByVideoID(baseDir, id string) (string, bool) { if id == "" { return "", false } var match string // An unreadable base dir simply means no match here. _ = s.eachItemDir(baseDir, "FindByVideoID", func(itemDir string, meta models.ItemMetadata) bool { if meta.VideoID != id { return true } match = itemDir return false }) return match, match != "" } // eachItemDir walks the marked library items directly under baseDir, reading // each one's metadata, and calls fn until it returns false. Entries that aren't // items, or whose metadata can't be read, are skipped — a single bad item must // not abort a scan. logLabel names the caller in those skip messages. func (s *LibraryService) eachItemDir(baseDir, logLabel string, fn func(itemDir string, meta models.ItemMetadata) bool) error { entries, err := os.ReadDir(baseDir) if err != nil { return err } for _, entry := range entries { if !entry.IsDir() || entry.Name() == subtitlesDirName { continue } itemDir := filepath.Join(baseDir, entry.Name()) if _, err := os.Stat(filepath.Join(itemDir, itemMarkerName)); err != nil { continue } meta, err := s.readMetadata(itemDir) if err != nil { log.Printf("%s: skipping %s: %v", logLabel, itemDir, err) continue } if !fn(itemDir, meta) { return nil } } return nil } // PruneToIDSet deletes items directly under baseDir whose identity key is not in // keep. It is used to mirror a subscription's source: entries removed upstream // are removed locally. Items without a known identity key are left untouched (we // never delete something we can't positively identify). Returns the number // removed. func (s *LibraryService) PruneToIDSet(baseDir string, keep map[string]bool) (int, error) { removed := 0 err := s.eachItemDir(baseDir, "PruneToIDSet", func(itemDir string, meta models.ItemMetadata) bool { if meta.VideoID == "" || keep[meta.VideoID] { return true } if rel, err := filepath.Rel(s.libraryDir, itemDir); err == nil { s.evictCachedScan(filepath.ToSlash(rel)) } if err := os.RemoveAll(itemDir); err != nil { log.Printf("warning: prune failed to remove %s: %v", itemDir, err) return true } removed++ return true }) return removed, err }