library.go
⎇
Raw
1package service
2
3import (
4 "bytes"
5 "encoding/json"
6 "fmt"
7 "log"
8 "os"
9 "path/filepath"
10 "sort"
11 "strconv"
12 "strings"
13 "sync"
14 "time"
15
16 "github.com/BurntSushi/toml"
17
18 "vidarchive/internal/models"
19)
20
21const (
22 itemMarkerName = ".vidarchive-item.toml"
23 subtitlesDirName = "subtitles"
24)
25
26var mediaExts = map[string]struct{}{
27 ".mp4": {}, ".webm": {}, ".mkv": {}, ".avi": {}, ".mov": {},
28 ".mp3": {}, ".wav": {}, ".flac": {}, ".aac": {}, ".opus": {}, ".m4a": {},
29}
30
31var audioExts = map[string]struct{}{
32 ".mp3": {}, ".wav": {}, ".flac": {}, ".aac": {}, ".opus": {}, ".m4a": {},
33}
34
35type LibraryService struct {
36 libraryDir string
37 // ffmpegPath/ffprobePath are the binaries used for thumbnail extraction,
38 // subtitle conversion, and media probing. Configurable so non-PATH installs
39 // (e.g. a pinned build) can be pointed at directly.
40 ffmpegPath string
41 ffprobePath string
42 // thumbLocks holds a per-media-file mutex serializing extraction so two
43 // callers never write the same temp file at once. Entries are deliberately
44 // never pruned: dropping one would reintroduce the race it prevents.
45 thumbLocks sync.Map
46 thumbSem chan struct{}
47
48 // thumbFailed records media filepaths whose extraction already failed this
49 // run, so we trust ffmpeg's verdict and don't re-run it on every request.
50 thumbFailed sync.Map
51
52 // scanCache memoizes scanned items for a short TTL so the listing page (which
53 // fans out one thumbnail request per media file) and quick auto-refreshes
54 // don't re-parse each item's marker + info.json on every request. Only item
55 // scans are cached — the directory listing itself is always read fresh, so
56 // newly added/removed items and subfolders appear immediately.
57 scanMu sync.Mutex
58 scanCache map[string]scanCacheEntry
59 scanTTL time.Duration
60}
61
62type scanCacheEntry struct {
63 item *models.LibraryItem
64 at time.Time
65}
66
67// scanCacheTTL is how long a scanned item is reused before being re-read.
68const scanCacheTTL = 10 * time.Second
69
70func NewLibraryService(libraryDir, ffmpegPath, ffprobePath string) *LibraryService {
71 return &LibraryService{
72 libraryDir: libraryDir,
73 ffmpegPath: ffmpegPath,
74 ffprobePath: ffprobePath,
75 thumbSem: make(chan struct{}, maxConcurrentThumbnails),
76 scanCache: make(map[string]scanCacheEntry),
77 scanTTL: scanCacheTTL,
78 }
79}
80
81func (s *LibraryService) getCachedScan(relPath string) (*models.LibraryItem, bool) {
82 if s.scanTTL <= 0 {
83 return nil, false
84 }
85 s.scanMu.Lock()
86 defer s.scanMu.Unlock()
87 e, ok := s.scanCache[relPath]
88 if !ok || time.Since(e.at) > s.scanTTL {
89 return nil, false
90 }
91 return e.item, true
92}
93
94func (s *LibraryService) putCachedScan(relPath string, item *models.LibraryItem) {
95 if s.scanTTL <= 0 {
96 return
97 }
98 s.scanMu.Lock()
99 s.scanCache[relPath] = scanCacheEntry{item: item, at: time.Now()}
100 s.scanMu.Unlock()
101}
102
103func (s *LibraryService) evictCachedScan(relPath string) {
104 s.scanMu.Lock()
105 delete(s.scanCache, relPath)
106 s.scanMu.Unlock()
107}
108
109// scannedItem returns a cached scan if fresh, otherwise scans and caches it.
110func (s *LibraryService) scannedItem(itemDir, relPath string) (*models.LibraryItem, error) {
111 if item, ok := s.getCachedScan(relPath); ok {
112 return item, nil
113 }
114 item, err := s.scanItem(itemDir, relPath)
115 if err != nil {
116 return nil, err
117 }
118 s.putCachedScan(relPath, item)
119 return item, nil
120}
121
122// ResolveWithinLibrary resolves a caller-supplied relative directory against the
123// library root and rejects anything that escapes it. The directory need not
124// exist yet, so it is safe to use when choosing a download's output location.
125func (s *LibraryService) ResolveWithinLibrary(relPath string) (string, error) {
126 return s.resolveItemDir(filepath.Clean(relPath))
127}
128
129// libraryRoot returns the symlink-resolved library root.
130func (s *LibraryService) libraryRoot() string {
131 if base, err := filepath.EvalSymlinks(s.libraryDir); err == nil {
132 return base
133 }
134 return filepath.Clean(s.libraryDir)
135}
136
137func (s *LibraryService) resolveItemDir(relPath string) (string, error) {
138 relPath = strings.Trim(relPath, string(filepath.Separator))
139 base := s.libraryRoot()
140 if relPath == "" || relPath == "." {
141 return base, nil
142 }
143 itemDir := filepath.Join(base, relPath)
144 cleanDir, err := filepath.EvalSymlinks(itemDir)
145 if err != nil {
146 cleanDir = filepath.Clean(itemDir)
147 }
148 if !strings.HasPrefix(cleanDir, base+string(filepath.Separator)) && cleanDir != base {
149 return "", fmt.Errorf("invalid path")
150 }
151 // Return the cleaned/symlink-resolved path we just validated, so callers do
152 // I/O on exactly the path that passed the boundary check.
153 return cleanDir, nil
154}
155
156func (s *LibraryService) GetAll(path, sortBy, filter string) ([]*models.LibraryItem, []string, error) {
157 if path != "" && !strings.HasSuffix(path, "/") {
158 path += "/"
159 }
160
161 dir, err := s.resolveItemDir(path)
162 if err != nil {
163 log.Printf("GetAll: invalid path %q: %v", path, err)
164 return nil, nil, nil
165 }
166 entries, err := os.ReadDir(dir)
167 if err != nil {
168 if os.IsNotExist(err) {
169 return nil, nil, nil
170 }
171 return nil, nil, err
172 }
173
174 var items []*models.LibraryItem
175 var folders []string
176
177 for _, entry := range entries {
178 if !entry.IsDir() {
179 continue
180 }
181 name := entry.Name()
182 if name == subtitlesDirName {
183 continue
184 }
185 itemDir := filepath.Join(dir, name)
186 relPath := filepath.ToSlash(filepath.Join(path, name))
187
188 markerPath := filepath.Join(itemDir, itemMarkerName)
189 if _, err := os.Stat(markerPath); err == nil {
190 item, err := s.scannedItem(itemDir, relPath)
191 if err != nil {
192 log.Printf("warning: failed to scan item %s: %v", relPath, err)
193 continue
194 }
195 if filter != "" && !strings.Contains(strings.ToLower(item.Name), strings.ToLower(filter)) && !strings.Contains(strings.ToLower(item.RelPath), strings.ToLower(filter)) {
196 continue
197 }
198 items = append(items, item)
199 } else {
200 folders = append(folders, name)
201 }
202 }
203
204 switch sortBy {
205 case "title":
206 sort.Slice(items, func(i, j int) bool { return strings.ToLower(items[i].Name) < strings.ToLower(items[j].Name) })
207 case "duration":
208 sort.Slice(items, func(i, j int) bool { return items[i].Duration > items[j].Duration })
209 case "date":
210 fallthrough
211 default:
212 // Stat each item once up front rather than twice per comparison.
213 modTime := make(map[string]int64, len(items))
214 for _, it := range items {
215 if info, err := os.Stat(it.DirPath); err == nil {
216 modTime[it.RelPath] = info.ModTime().UnixNano()
217 }
218 }
219 sort.Slice(items, func(i, j int) bool {
220 return modTime[items[i].RelPath] > modTime[items[j].RelPath]
221 })
222 }
223
224 sort.Strings(folders)
225
226 return items, folders, nil
227}
228
229// GetByRelPath returns the item at relPath, reusing a recent cached scan when
230// available (see scanCache).
231func (s *LibraryService) GetByRelPath(relPath string) (*models.LibraryItem, error) {
232 relPath = strings.Trim(relPath, "/")
233
234 if item, ok := s.getCachedScan(relPath); ok {
235 return item, nil
236 }
237
238 itemDir, err := s.resolveItemDir(relPath)
239 if err != nil {
240 return nil, fmt.Errorf("item not found")
241 }
242 if _, err := os.Stat(filepath.Join(itemDir, itemMarkerName)); err != nil {
243 return nil, fmt.Errorf("item not found")
244 }
245 return s.scannedItem(itemDir, relPath)
246}
247
248func (s *LibraryService) scanItem(itemDir, relPath string) (*models.LibraryItem, error) {
249 metadata, err := s.readMetadata(itemDir)
250 if err != nil {
251 return nil, err
252 }
253
254 mediaFiles, infoJSONPath, err := s.listItemFiles(itemDir)
255 if err != nil {
256 return nil, err
257 }
258
259 var info map[string]interface{}
260 if infoJSONPath != "" {
261 data, err := os.ReadFile(infoJSONPath)
262 if err == nil {
263 if err := json.Unmarshal(data, &info); err != nil {
264 log.Printf("scanItem: ignoring malformed %s: %v", infoJSONPath, err)
265 }
266 }
267 }
268
269 // dirty tracks whether we derived any new metadata worth persisting, so a
270 // plain listing or detail view doesn't rewrite the marker file on every read.
271 dirty := false
272
273 if metadata.Name == "" {
274 if title, ok := infoString(info, "title"); ok && title != "" {
275 metadata.Name = title
276 } else if len(mediaFiles) > 0 {
277 metadata.Name = mediaFileStem(mediaFiles[0])
278 } else {
279 metadata.Name = filepath.Base(itemDir)
280 }
281 dirty = true
282 }
283
284 if metadata.SourceURL == "" {
285 dirty = backfillString(&metadata.SourceURL, info, "webpage_url", "url") || dirty
286 }
287
288 if metadata.Description == "" {
289 dirty = backfillString(&metadata.Description, info, "description") || dirty
290 }
291
292 // Backfill the stable identity (yt-dlp's video id) from info.json so
293 // pre-existing items gain an identity on their next scan. Subscriptions match
294 // and prune items by this id (see FindByVideoID / PruneToIDSet).
295 if metadata.VideoID == "" {
296 dirty = backfillString(&metadata.VideoID, info, "id") || dirty
297 }
298
299 if metadata.FileDurations == nil {
300 metadata.FileDurations = make(map[string]int)
301 }
302
303 // Per-file durations come from the marker's file_durations map (populated at
304 // import time). For a single-file item we also seed it from info.json's
305 // duration, which covers the common case without a probe. We deliberately do
306 // not run ffprobe here — keeping it off the scan/listing path is the point of
307 // the marker cache. Files without a known duration simply show no badge.
308 for i := range mediaFiles {
309 mf := &mediaFiles[i]
310 if d, ok := metadata.FileDurations[mf.Filename]; ok {
311 mf.Duration = d
312 continue
313 }
314 if len(mediaFiles) == 1 {
315 if d, ok := infoDuration(info); ok && d > 0 {
316 mf.Duration = d
317 metadata.FileDurations[mf.Filename] = d
318 dirty = true
319 }
320 }
321 }
322
323 // The item-level duration is the sum of known per-file durations, used only
324 // for the "duration" sort — there is no single "overall" duration shown.
325 total := 0
326 for _, mf := range mediaFiles {
327 if mf.Duration > 0 {
328 total += mf.Duration
329 }
330 }
331
332 item := &models.LibraryItem{
333 Name: metadata.Name,
334 RelPath: relPath,
335 DirPath: itemDir,
336 SourceURL: metadata.SourceURL,
337 Duration: total,
338 Description: metadata.Description,
339 YtdlpFlags: metadata.YtdlpFlags,
340 MediaFiles: mediaFiles,
341 }
342
343 if dirty {
344 if err := s.writeMetadata(itemDir, metadata); err != nil {
345 log.Printf("scanItem: failed to persist derived metadata for %s: %v", itemDir, err)
346 }
347 }
348
349 return item, nil
350}
351
352// backfillString sets *field from the first non-empty string value among
353// info's keys, reporting whether it changed anything. An empty value in
354// info.json must not count as a change: the marker would be rewritten on every
355// scan forever without ever gaining a value.
356func backfillString(field *string, info map[string]interface{}, keys ...string) bool {
357 for _, k := range keys {
358 if v, ok := infoString(info, k); ok && v != "" {
359 *field = v
360 return true
361 }
362 }
363 return false
364}
365
366func (s *LibraryService) readMetadata(itemDir string) (models.ItemMetadata, error) {
367 markerPath := filepath.Join(itemDir, itemMarkerName)
368 var metadata models.ItemMetadata
369 data, err := os.ReadFile(markerPath)
370 if err == nil {
371 if _, err := toml.Decode(string(data), &metadata); err != nil {
372 log.Printf("warning: failed to parse %s: %v", markerPath, err)
373 }
374 }
375 return metadata, nil
376}
377
378func (s *LibraryService) writeMetadata(itemDir string, metadata models.ItemMetadata) error {
379 var buf bytes.Buffer
380 if err := toml.NewEncoder(&buf).Encode(metadata); err != nil {
381 return err
382 }
383 return atomicWrite(filepath.Join(itemDir, itemMarkerName), buf.Bytes(), markerFileMode)
384}
385
386// markerFileMode matches the info.json sidecar: both sit in the library next to
387// the media and are meant to be readable (and hand-editable) by the operator.
388const markerFileMode = 0o644
389
390// atomicWrite writes data to path via a uniquely-named temp file in the same
391// directory and a rename, so a crash mid-write can't leave a truncated file and
392// two concurrent writers never collide on a shared temp path.
393func atomicWrite(path string, data []byte, mode os.FileMode) error {
394 f, err := os.CreateTemp(filepath.Dir(path), ".vidarchive-*.tmp")
395 if err != nil {
396 return err
397 }
398 tmpPath := f.Name()
399 // A no-op once the rename below has succeeded.
400 defer os.Remove(tmpPath)
401
402 if err := f.Chmod(mode); err != nil {
403 f.Close()
404 return err
405 }
406 if _, err := f.Write(data); err != nil {
407 f.Close()
408 return err
409 }
410 // Close explicitly: a deferred close would hide a flush error on the write.
411 if err := f.Close(); err != nil {
412 return err
413 }
414 return os.Rename(tmpPath, path)
415}
416
417func (s *LibraryService) listItemFiles(itemDir string) ([]models.MediaFile, string, error) {
418 entries, err := os.ReadDir(itemDir)
419 if err != nil {
420 return nil, "", err
421 }
422
423 var mediaFiles []models.MediaFile
424 var infoJSONFiles []string
425
426 for _, entry := range entries {
427 if entry.IsDir() {
428 continue
429 }
430 name := entry.Name()
431 path := filepath.Join(itemDir, name)
432 ext := strings.ToLower(filepath.Ext(name))
433
434 if name == itemMarkerName {
435 continue
436 }
437 if name == "info.json" || strings.HasSuffix(name, ".info.json") {
438 infoJSONFiles = append(infoJSONFiles, path)
439 continue
440 }
441 if _, ok := mediaExts[ext]; !ok {
442 continue
443 }
444
445 _, isAudio := audioExts[ext]
446
447 mediaFiles = append(mediaFiles, models.MediaFile{
448 Filename: name,
449 Filepath: path,
450 IsAudio: isAudio,
451 Duration: -1,
452 })
453 }
454
455 sort.Slice(mediaFiles, func(i, j int) bool {
456 return mediaFiles[i].Filepath < mediaFiles[j].Filepath
457 })
458
459 var infoJSONPath string
460 if len(infoJSONFiles) > 0 {
461 sort.Strings(infoJSONFiles)
462 infoJSONPath = infoJSONFiles[0]
463 if len(infoJSONFiles) > 1 {
464 log.Printf("warning: multiple info.json files in %s, using %s", itemDir, infoJSONPath)
465 }
466 }
467
468 return mediaFiles, infoJSONPath, nil
469}
470
471func infoString(info map[string]interface{}, key string) (string, bool) {
472 if info == nil {
473 return "", false
474 }
475 // Only accept genuine strings: title/url/description are always strings in
476 // yt-dlp output, and stringifying an arbitrary JSON value (map, slice) would
477 // store junk like "map[...]" into the field.
478 if s, ok := info[key].(string); ok {
479 return s, true
480 }
481 return "", false
482}
483
484func infoDuration(info map[string]interface{}) (int, bool) {
485 if info == nil {
486 return 0, false
487 }
488 v, ok := info["duration"]
489 if !ok {
490 return 0, false
491 }
492 switch n := v.(type) {
493 case float64:
494 return int(n + 0.5), true
495 case string:
496 if f, err := strconv.ParseFloat(n, 64); err == nil {
497 return int(f + 0.5), true
498 }
499 }
500 return 0, false
501}
502
503func mediaFileStem(mf models.MediaFile) string {
504 return strings.TrimSuffix(filepath.Base(mf.Filename), filepath.Ext(mf.Filename))
505}
506
507// primaryMediaFile picks the representative file for an item: the largest video
508// file, or — if there are none — the largest file overall. Returns nil for an
509// item with no media files. Used for the item-level thumbnail and as the default
510// target for metadata, keeping those two consistent.
511func primaryMediaFile(item *models.LibraryItem) *models.MediaFile {
512 size := func(mf *models.MediaFile) int64 {
513 if info, err := os.Stat(mf.Filepath); err == nil {
514 return info.Size()
515 }
516 return 0
517 }
518 var best *models.MediaFile
519 for i := range item.MediaFiles {
520 mf := &item.MediaFiles[i]
521 switch {
522 case best == nil:
523 best = mf
524 case best.IsAudio && !mf.IsAudio:
525 // Prefer any video over audio.
526 best = mf
527 case best.IsAudio == mf.IsAudio && size(mf) > size(best):
528 best = mf
529 }
530 }
531 return best
532}
533
534func (s *LibraryService) GetMediaFile(relPath, filename string) (string, error) {
535 item, err := s.GetByRelPath(relPath)
536 if err != nil {
537 return "", err
538 }
539 for _, mf := range item.MediaFiles {
540 if mf.Filename == filename {
541 return mf.Filepath, nil
542 }
543 }
544 return "", fmt.Errorf("media file not found")
545}
546
547func (s *LibraryService) Delete(relPath string) error {
548 itemDir, err := s.resolveItemDir(relPath)
549 if err != nil {
550 return err
551 }
552 // resolveItemDir maps ""/"." to the library root and resolves symlinks, so a
553 // result equal to the root (reachable via a URL-encoded slash, "sub/..", or a
554 // symlink pointing back at the root) must be refused — deleting it would
555 // wipe the entire library.
556 if itemDir == s.libraryRoot() {
557 return fmt.Errorf("refusing to delete library root")
558 }
559 // Evict the cached scan so the deletion is reflected immediately rather than
560 // lingering until the TTL expires.
561 s.evictCachedScan(strings.Trim(relPath, "/"))
562 return os.RemoveAll(itemDir)
563}
564
565// FindByVideoID returns the absolute directory of the item under baseDir whose
566// marker matches the given yt-dlp video id, scanning only that directory (the
567// subscription's owned folder). Matching on the id alone is safe here because
568// each subscription owns a single source, so ids don't collide across
569// extractors within the folder. ok is false when no match is found or id is
570// empty.
571func (s *LibraryService) FindByVideoID(baseDir, id string) (string, bool) {
572 if id == "" {
573 return "", false
574 }
575
576 var match string
577 // An unreadable base dir simply means no match here.
578 _ = s.eachItemDir(baseDir, "FindByVideoID", func(itemDir string, meta models.ItemMetadata) bool {
579 if meta.VideoID != id {
580 return true
581 }
582 match = itemDir
583 return false
584 })
585
586 return match, match != ""
587}
588
589// eachItemDir walks the marked library items directly under baseDir, reading
590// each one's metadata, and calls fn until it returns false. Entries that aren't
591// items, or whose metadata can't be read, are skipped — a single bad item must
592// not abort a scan. logLabel names the caller in those skip messages.
593func (s *LibraryService) eachItemDir(baseDir, logLabel string, fn func(itemDir string, meta models.ItemMetadata) bool) error {
594 entries, err := os.ReadDir(baseDir)
595 if err != nil {
596 return err
597 }
598
599 for _, entry := range entries {
600 if !entry.IsDir() || entry.Name() == subtitlesDirName {
601 continue
602 }
603 itemDir := filepath.Join(baseDir, entry.Name())
604 if _, err := os.Stat(filepath.Join(itemDir, itemMarkerName)); err != nil {
605 continue
606 }
607 meta, err := s.readMetadata(itemDir)
608 if err != nil {
609 log.Printf("%s: skipping %s: %v", logLabel, itemDir, err)
610 continue
611 }
612 if !fn(itemDir, meta) {
613 return nil
614 }
615 }
616
617 return nil
618}
619
620// PruneToIDSet deletes items directly under baseDir whose identity key is not in
621// keep. It is used to mirror a subscription's source: entries removed upstream
622// are removed locally. Items without a known identity key are left untouched (we
623// never delete something we can't positively identify). Returns the number
624// removed.
625func (s *LibraryService) PruneToIDSet(baseDir string, keep map[string]bool) (int, error) {
626 removed := 0
627 err := s.eachItemDir(baseDir, "PruneToIDSet", func(itemDir string, meta models.ItemMetadata) bool {
628 if meta.VideoID == "" || keep[meta.VideoID] {
629 return true
630 }
631 if rel, err := filepath.Rel(s.libraryDir, itemDir); err == nil {
632 s.evictCachedScan(filepath.ToSlash(rel))
633 }
634 if err := os.RemoveAll(itemDir); err != nil {
635 log.Printf("warning: prune failed to remove %s: %v", itemDir, err)
636 return true
637 }
638 removed++
639 return true
640 })
641
642 return removed, err
643}
644