package service import ( "bufio" "bytes" "context" "database/sql" "encoding/json" "errors" "fmt" "io" "log" "os" "os/exec" "path/filepath" "sort" "strconv" "strings" "sync" "syscall" "time" "github.com/gabriel-vasile/mimetype" "vidarchive/internal/config" "vidarchive/internal/models" "vidarchive/internal/repository" "vidarchive/internal/util" ) type DownloadService struct { repo *repository.DownloadRepository librarySvc *LibraryService presetSvc *PresetService settingsSvc *SettingsService subscriptionSvc *SubscriptionService cfg *config.Config cache *ProgressCache activeMu sync.Mutex active map[int64]context.CancelFunc } func NewDownloadService(repo *repository.DownloadRepository, librarySvc *LibraryService, presetSvc *PresetService, settingsSvc *SettingsService, subscriptionSvc *SubscriptionService, cfg *config.Config) *DownloadService { return &DownloadService{ repo: repo, librarySvc: librarySvc, presetSvc: presetSvc, settingsSvc: settingsSvc, subscriptionSvc: subscriptionSvc, cfg: cfg, cache: NewProgressCache(), active: make(map[int64]context.CancelFunc), } } func (s *DownloadService) Create(url string, presetID *int64, formatOverride, customFlags, outputDir string) (*models.Download, error) { d := &models.Download{ URL: url, Status: "queued", FormatOverride: formatOverride, CustomFlags: customFlags, OutputDir: sql.NullString{String: outputDir, Valid: outputDir != ""}, } if presetID != nil { d.PresetID = sqlNullInt64(*presetID) } if err := s.repo.Create(d); err != nil { return nil, err } return d, nil } // CreateForSubscription queues a download for a subscription run, copying its // download options and tagging it with the subscription id so ExecuteDownload // applies the right refresh mode and pruning. func (s *DownloadService) CreateForSubscription(sub *models.Subscription) (*models.Download, error) { d := &models.Download{ URL: sub.URL, Status: "queued", FormatOverride: sub.FormatOverride, CustomFlags: sub.CustomFlags, OutputDir: sql.NullString{String: sub.OutputDir, Valid: sub.OutputDir != ""}, PresetID: sub.PresetID, SubscriptionID: sqlNullInt64(sub.ID), } if err := s.repo.Create(d); err != nil { return nil, err } return d, nil } func (s *DownloadService) GetByID(id int64) (*models.Download, error) { d, err := s.repo.GetByID(id) if err != nil { return nil, err } if logs := s.cache.Snapshot(id); logs != "" { d.Logs = sql.NullString{String: logs, Valid: true} } return d, nil } func (s *DownloadService) GetAll(status, sortBy string) ([]*models.Download, error) { downloads, err := s.repo.GetAll(status, sortBy) if err != nil { return nil, err } for _, d := range downloads { if logs := s.cache.Snapshot(d.ID); logs != "" { d.Logs = sql.NullString{String: logs, Valid: true} } } return downloads, nil } func (s *DownloadService) GetQueued(limit int) ([]*models.Download, error) { return s.repo.GetQueued(limit) } // HasActiveForSubscription reports whether the subscription already has a queued // or in-progress download, so the scheduler can skip stacking another run. func (s *DownloadService) HasActiveForSubscription(subID int64) (bool, error) { return s.repo.HasActiveForSubscription(subID) } func (s *DownloadService) Delete(id int64) error { s.cancelDownload(id) s.cache.Delete(id) return s.repo.Delete(id) } // registerActive records the cancel func for a claimed download and returns a // release func. Registration happens at claim time rather than after the process // spawns, so a delete arriving during setup, between yt-dlp and the import, or // mid-import still stops the work instead of silently letting it finish. func (s *DownloadService) registerActive(id int64, cancel context.CancelFunc) func() { s.activeMu.Lock() s.active[id] = cancel s.activeMu.Unlock() return func() { s.activeMu.Lock() delete(s.active, id) s.activeMu.Unlock() } } func (s *DownloadService) cancelDownload(id int64) { s.activeMu.Lock() cancel, ok := s.active[id] delete(s.active, id) s.activeMu.Unlock() if ok { cancel() } } // CancelAll stops every download currently in flight. Used on shutdown and when // clearing the queue, so no yt-dlp child outlives the rows that described it. func (s *DownloadService) CancelAll() { s.activeMu.Lock() cancels := make([]context.CancelFunc, 0, len(s.active)) for id, cancel := range s.active { cancels = append(cancels, cancel) delete(s.active, id) } s.activeMu.Unlock() for _, cancel := range cancels { cancel() } } func (s *DownloadService) DeleteAll() error { // Clearing the queue must also stop what is running; otherwise yt-dlp keeps // going and imports into the library after its row is gone. s.CancelAll() return s.repo.DeleteAll() } func (s *DownloadService) CountByStatus(ctx context.Context) (map[string]int, error) { return s.repo.CountByStatus(ctx) } // ResetStalledDownloads re-queues downloads left mid-flight by a previous run and // discards their temp directories. Without the cleanup the re-run imports into a // fresh uniqueDir and the library ends up with a duplicate of the same item. func (s *DownloadService) ResetStalledDownloads() error { ids, err := s.repo.IDsByStatus("downloading") if err != nil { return err } for _, id := range ids { for _, dir := range s.tempDirsFor(id) { if err := os.RemoveAll(dir); err != nil { log.Printf("warning: failed to remove stale temp dir %s: %v", dir, err) } } } return s.repo.UpdateStatusWhere("downloading", "queued") } // tempDirFor returns the scratch directory a download writes into. func (s *DownloadService) tempDirFor(id int64) string { return filepath.Join(s.cfg.TempDir, strconv.FormatInt(id, 10)) } // tempNewDirFor returns the second-pass scratch directory used by metadata mode. func (s *DownloadService) tempNewDirFor(id int64) string { return s.tempDirFor(id) + "-new" } // tempDirsFor returns every scratch directory a download owns. ResetStalledDownloads // clears these, so the two builders above must stay the only places that name them. func (s *DownloadService) tempDirsFor(id int64) []string { return []string{s.tempDirFor(id), s.tempNewDirFor(id)} } func (s *DownloadService) ListFormats(url string) ([]*models.FormatInfo, error) { // Use machine-readable JSON (-J) rather than scraping the human "-F" table, // whose columns/separators shift between yt-dlp versions. stderr is captured // separately so warnings can't corrupt the JSON on stdout. cmd := exec.Command(s.cfg.YTDLPPath, "-J", "--no-warnings", url) var stderr bytes.Buffer cmd.Stderr = &stderr output, err := cmd.Output() if err != nil { return nil, fmt.Errorf("yt-dlp -J failed: %w\n%s", err, stderr.String()) } return parseFormatJSON(output) } // ExecuteDownload runs the download for d. The bool reports whether this call // actually processed it: false means another worker already claimed it (Submit // and the queue checker can both enqueue the same row within the 2s poll window), // so the caller should not log it as completed. A cancelled run returns // ErrCancelled. // // parent belongs to the worker pool: deriving from it means a shutdown cancels // the download even if it lands before this call registers its own cancel func. func (s *DownloadService) ExecuteDownload(parent context.Context, d *models.Download) (bool, error) { claimed, err := s.repo.MarkStarted(d.ID) if err != nil { return false, err } if !claimed { return false, nil } // Registered before any work starts so Delete/CancelAll can interrupt every // phase, not just the window where yt-dlp happens to be running. ctx, cancel := context.WithCancel(parent) defer cancel() defer s.registerActive(d.ID, cancel)() s.cache.Set(d.ID, &LiveDownload{LastUpdate: time.Now()}) defer s.cache.Delete(d.ID) var preset *models.Preset if d.PresetID.Valid { preset, err = s.presetSvc.GetByID(d.PresetID.Int64) if err != nil { log.Printf("download %d: preset %d lookup failed (%v); falling back to default", d.ID, d.PresetID.Int64, err) preset = nil } } if preset == nil { var derr error if preset, derr = s.presetSvc.GetDefault(); derr != nil { log.Printf("download %d: no default preset available (%v); using built-in defaults", d.ID, derr) preset = &models.Preset{} } } var sub *models.Subscription if d.SubscriptionID.Valid && s.subscriptionSvc != nil { var serr error if sub, serr = s.subscriptionSvc.GetByID(d.SubscriptionID.Int64); serr != nil { log.Printf("download %d: subscription %d lookup failed: %v", d.ID, d.SubscriptionID.Int64, serr) } } // Reject custom flags that clash with options VidArchive sets itself, before // spending any work — the download fails with a message naming the offender. isSubscription := d.SubscriptionID.Valid for _, flags := range []string{d.CustomFlags, preset.CustomFlags} { if err := checkReservedFlags(flags, isSubscription); err != nil { s.finalizeError(d.ID, err) return false, err } } tempDownloadDir := s.tempDirFor(d.ID) if err := os.MkdirAll(tempDownloadDir, 0755); err != nil { return false, fmt.Errorf("create temp download dir: %w", err) } // Own the temp dir's lifetime here, where it's created, so it's removed on // every exit path — including a failed yt-dlp run or an early return that // crashes mid-import. The import helpers below no longer clean it up. defer os.RemoveAll(tempDownloadDir) args := s.presetSvc.BuildArgs(preset, d.FormatOverride, d.CustomFlags) // Record the meaningful flags (format/audio/subs/custom) that shaped this // download, before the internal plumbing (cookies, -P/-o, URL) is appended, // so each imported item can show how it was fetched. ytdlpFlags := strings.Join(args, " ") var cookieCleanup func() args, cookieCleanup = s.appendCookies(args) defer cookieCleanup() if sub != nil { // Always write info.json so the import step can read the stable identity // (yt-dlp's video id) used to match/replace existing items. args = append(args, "--write-info-json") switch sub.RefreshMode { case "skip": // Let yt-dlp skip entries already recorded — no re-download. archive := s.subscriptionSvc.ArchivePath(sub.ID) if err := os.MkdirAll(filepath.Dir(archive), 0755); err == nil { args = append(args, "--download-archive", archive) } case "metadata": // Refresh metadata only; don't fetch media. args = append(args, "--skip-download") } } args = append(args, "-P", tempDownloadDir) args = append(args, "-o", "item-%(autonumber)05d/%(title)s.%(ext)s") args = append(args, d.URL) runErr := s.runYTDLP(ctx, d, args) // Cancellation wins over both the run error and the import: a cancel that // lands just after yt-dlp exited 0 leaves runErr nil, and the item must not // reach the library after the user removed it. if ctx.Err() != nil { return false, s.finalizeCancelled(parent, d.ID) } if runErr != nil { s.finalizeError(d.ID, runErr) return false, runErr } mode := "" if sub != nil { mode = sub.RefreshMode } // Post-process before marking completed, so the download stays "downloading" // until everything is really done — including metadata mode's second pass, // which downloads any genuinely new entries as full items. var postErr error if mode == "metadata" { // The main pass ran with --skip-download, so the temp dir holds only // info.json files: refresh existing items in place and fetch new ones. postErr = s.refreshAndAddNew(ctx, d, preset, tempDownloadDir, ytdlpFlags) } else { imported, err := s.importDownloadedItems(ctx, d, tempDownloadDir, mode, ytdlpFlags) switch { case err != nil: postErr = err // A plain (non-subscription) download that yields nothing is a failure, not // a silent "completed". Subscription modes legitimately import zero (skip // mode, or a metadata refresh with no new entries), so only enforce this for // plain runs. case sub == nil && imported == 0: postErr = fmt.Errorf("yt-dlp finished but no media files were downloaded") } } if postErr == nil && sub != nil && sub.PruneRemoved { s.pruneSubscription(ctx, d, sub) } // Same cancellation check as above: a stop during post-processing must not // be recorded as "completed". if ctx.Err() != nil { return false, s.finalizeCancelled(parent, d.ID) } if postErr != nil { s.finalizeError(d.ID, postErr) return false, postErr } if err := s.repo.MarkCompleted(d.ID, "completed"); err != nil { return false, err } return true, nil } // runYTDLP executes yt-dlp with args, streaming combined output into the live // progress cache and periodically flushing it to the download's persisted log. // Cancelling ctx kills the whole process group and makes this return. func (s *DownloadService) runYTDLP(ctx context.Context, d *models.Download, args []string) error { // --newline forces yt-dlp to emit each progress update on its own line. Without // it, progress is rewritten in place with carriage returns, so a long download // becomes one ever-growing line that overflows the reader's buffer and stalls // the pipe — hanging the download. See the hardened scanner below. fullArgs := append([]string{"--newline"}, args...) cmd := exec.CommandContext(ctx, s.cfg.YTDLPPath, fullArgs...) cmd.SysProcAttr = &syscall.SysProcAttr{Setpgid: true} // yt-dlp spawns helpers (ffmpeg, external downloaders). Kill the whole group // rather than just the parent, which would leave those orphaned. cmd.Cancel = func() error { return syscall.Kill(-cmd.Process.Pid, syscall.SIGKILL) } stdout, err := cmd.StdoutPipe() if err != nil { return err } cmd.Stderr = cmd.Stdout if err := cmd.Start(); err != nil { return err } ticker := time.NewTicker(10 * time.Second) defer ticker.Stop() done := make(chan struct{}) go func() { for { select { case <-ticker.C: s.flushLogs(d.ID) case <-done: return } } }() scanner := bufio.NewScanner(stdout) // Allow long lines (a single yt-dlp message can exceed the 64 KiB default) // rather than letting the scanner abort and leave the pipe unread. scanner.Buffer(make([]byte, 0, 64*1024), 1024*1024) for scanner.Scan() { s.cache.AppendLog(d.ID, scanner.Text()) } if err := scanner.Err(); err != nil { log.Printf("download %d: error reading yt-dlp output: %v", d.ID, err) } close(done) s.flushLogs(d.ID) return cmd.Wait() } // appendCookies writes the saved cookies (if any) to a temp file and appends a // --cookies flag. The returned cleanup removes the temp file and is always safe // to call, even when no cookies were configured. func (s *DownloadService) appendCookies(args []string) ([]string, func()) { cookies, err := s.settingsSvc.GetCookies() if err != nil || strings.TrimSpace(cookies) == "" { return args, func() {} } path, err := s.writeCookiesFile(cookies) if err != nil { return args, func() {} } return append(args, "--cookies", path), func() { os.Remove(path) } } // writeCookiesFile writes cookies to a temp file in the app's own temp dir (the // same volume the rest of the run uses). On any failure the partial file is // removed — a truncated cookies file must not be handed to yt-dlp. func (s *DownloadService) writeCookiesFile(cookies string) (string, error) { if err := os.MkdirAll(s.cfg.TempDir, 0755); err != nil { return "", err } tmpFile, err := os.CreateTemp(s.cfg.TempDir, "cookies-*.txt") if err != nil { return "", err } if _, err := tmpFile.WriteString(cookies); err != nil { tmpFile.Close() os.Remove(tmpFile.Name()) return "", err } if err := tmpFile.Close(); err != nil { os.Remove(tmpFile.Name()) return "", err } return tmpFile.Name(), nil } func (s *DownloadService) finalizeError(id int64, err error) { s.flushLogs(id) if markErr := s.repo.MarkError(id, err.Error()); markErr != nil { log.Printf("download %d: failed to record error: %v", id, markErr) } } // ErrCancelled reports that a download was deliberately stopped (deleted, queue // cleared, or shutdown) rather than having failed. Callers distinguish it so a // cancellation isn't logged as an error. var ErrCancelled = errors.New("download cancelled") // finalizeCancelled records a stopped download. // // A shutdown (parent already cancelled) deliberately leaves the row // "downloading": ResetStalledDownloads re-queues it on the next start, so // stopping the server resumes the download instead of losing it. Only a // user-initiated cancel is terminal. The row may already be deleted in that // case — cancellation usually arrives via Delete — so a missing row is fine. func (s *DownloadService) finalizeCancelled(parent context.Context, id int64) error { s.flushLogs(id) if parent.Err() != nil { return ErrCancelled } if err := s.repo.MarkCompleted(id, "cancelled"); err != nil { log.Printf("download %d: failed to record cancellation: %v", id, err) } return ErrCancelled } // flushLogs persists whatever output has accumulated for a download. func (s *DownloadService) flushLogs(id int64) { logs := s.cache.FlushLogs(id) if logs == "" { return } if err := s.repo.AppendLogs(id, logs); err != nil { log.Printf("download %d: failed to persist logs: %v", id, err) } } // resolveBaseLibraryDir returns the absolute library directory a download writes // into, applying the optional per-download OutputDir while rejecting any path // that escapes the library root. func (s *DownloadService) resolveBaseLibraryDir(d *models.Download) (string, error) { if !d.OutputDir.Valid || d.OutputDir.String == "" { return s.cfg.LibraryDir, nil } // Reuse the library service's guard so both entry points enforce the boundary // the same way — it resolves symlinks, which a plain prefix check does not. dir, err := s.librarySvc.ResolveWithinLibrary(d.OutputDir.String) if err != nil { return "", fmt.Errorf("invalid output directory: %w", err) } return dir, nil } // importDownloadedItems moves each downloaded item from the temp dir into the // library and returns the number of items successfully imported. Per-item // failures are logged and skipped (a playlist with a few bad entries still // imports the rest); a non-nil error means the import couldn't even start. // // A cancel stops the import between items and returns ctx.Err() with the count // imported so far. Importing a long playlist takes real time (a move plus an // ffprobe per file), so a deleted download must not keep filling the library. func (s *DownloadService) importDownloadedItems(ctx context.Context, d *models.Download, tempDownloadDir, mode, ytdlpFlags string) (int, error) { entries, err := os.ReadDir(tempDownloadDir) if err != nil { return 0, err } baseLibraryDir, err := s.resolveBaseLibraryDir(d) if err != nil { return 0, err } if err := os.MkdirAll(baseLibraryDir, 0755); err != nil { return 0, err } var itemDirs []string for _, entry := range entries { if !entry.IsDir() { continue } name := entry.Name() if strings.HasPrefix(name, "item-") { itemDirs = append(itemDirs, filepath.Join(tempDownloadDir, name)) } } sort.Strings(itemDirs) imported := 0 for _, itemDir := range itemDirs { if err := ctx.Err(); err != nil { return imported, err } if err := s.importItemDir(ctx, d.URL, itemDir, baseLibraryDir, mode, ytdlpFlags); err != nil { // A cancelled item isn't a bad item: stop instead of logging a warning // for it and every one that follows. if ctx.Err() != nil { return imported, ctx.Err() } log.Printf("warning: failed to import item %s: %v", itemDir, err) continue } imported++ } // The temp dir (and any leftovers from failed imports) is removed by the // caller's deferred cleanup, so partial state never leaks even on a crash. return imported, nil } func (s *DownloadService) importItemDir(ctx context.Context, url, itemDir, baseLibraryDir, mode, ytdlpFlags string) error { entries, err := os.ReadDir(itemDir) if err != nil { return err } var mediaFiles []os.DirEntry var infoJSONPath string var subtitleFiles []string for _, entry := range entries { if entry.IsDir() { continue } name := entry.Name() path := filepath.Join(itemDir, name) ext := strings.ToLower(filepath.Ext(name)) if isInfoJSON(name) { infoJSONPath = path continue } if ext == ".vtt" || ext == ".srt" || ext == ".ass" || ext == ".ssa" { subtitleFiles = append(subtitleFiles, path) continue } mtype, err := mimetype.DetectFile(path) if err == nil && mtype != nil && (strings.HasPrefix(mtype.String(), "audio/") || strings.HasPrefix(mtype.String(), "video/")) { mediaFiles = append(mediaFiles, entry) } } if len(mediaFiles) == 0 { return fmt.Errorf("no media files found in %s", itemDir) } info := readInfoJSON(infoJSONPath) name := s.deriveItemName(itemDir, info, mediaFiles) videoID := info.ID // Last point at which nothing has been written to the library yet: give up // here on a cancel rather than part-way through, which would leave a folder // with some of its files and no marker — or, in overwrite mode, delete the // existing item and not replace it. if err := ctx.Err(); err != nil { return err } // Overwrite mode: replace the existing copy of this video in place rather than // creating a duplicate folder. Removing the old dir lets uniqueDir reuse its // name (or land on the new title if it changed upstream). if mode == "overwrite" && videoID != "" { if existing, ok := s.librarySvc.FindByVideoID(baseLibraryDir, videoID); ok { if rel, err := filepath.Rel(s.cfg.LibraryDir, existing); err == nil { s.librarySvc.evictCachedScan(filepath.ToSlash(rel)) } os.RemoveAll(existing) } } targetDir := s.uniqueDir(baseLibraryDir, name) if err := os.MkdirAll(targetDir, 0755); err != nil { return err } if infoJSONPath != "" { if err := moveFile(infoJSONPath, filepath.Join(targetDir, "info.json")); err != nil { return err } } for _, entry := range mediaFiles { if err := moveFile(filepath.Join(itemDir, entry.Name()), filepath.Join(targetDir, entry.Name())); err != nil { return err } } // Probe each media file's duration once, here in the worker (off the request // path), and cache it in the marker so the library never has to probe while // serving pages. Files we can't probe simply get no duration. fileDurations := make(map[string]int) for _, entry := range mediaFiles { if d, ok := probeDuration(ctx, s.cfg.FFprobePath, filepath.Join(targetDir, entry.Name())); ok { fileDurations[entry.Name()] = d } } if len(subtitleFiles) > 0 { subtitlesDir := filepath.Join(targetDir, subtitlesDirName) if err := os.MkdirAll(subtitlesDir, 0755); err != nil { return err } for _, sf := range subtitleFiles { if err := moveFile(sf, filepath.Join(subtitlesDir, filepath.Base(sf))); err != nil { return err } } } metadata := models.ItemMetadata{ Name: name, SourceURL: url, VideoID: videoID, YtdlpFlags: ytdlpFlags, FileDurations: fileDurations, } return s.librarySvc.writeMetadata(targetDir, metadata) } // infoJSON is the subset of yt-dlp's info.json VidArchive reads. ID is the // stable item identity; within a single subscription's own directory it is // enough to match items, so the extractor is not needed. type infoJSON struct { ID string `json:"id"` Title string `json:"title"` Description string `json:"description"` WebpageURL string `json:"webpage_url"` } // readInfoJSON parses an info.json. A missing, unreadable or malformed file // yields a zero-value struct: every caller treats absent fields as "unknown" // and falls back, so there is nothing to distinguish. func readInfoJSON(infoJSONPath string) infoJSON { var info infoJSON if infoJSONPath == "" { return info } data, err := os.ReadFile(infoJSONPath) if err != nil { return info } if err := json.Unmarshal(data, &info); err != nil { log.Printf("ignoring malformed %s: %v", infoJSONPath, err) return infoJSON{} } return info } // refreshAndAddNew handles a metadata-mode run. The main pass used // --skip-download, so tempDownloadDir holds only info.json files. Existing // library items have their markers refreshed in place; entries with no existing // match are genuinely new and are downloaded as full items in a second pass. func (s *DownloadService) refreshAndAddNew(ctx context.Context, d *models.Download, preset *models.Preset, tempDownloadDir, ytdlpFlags string) error { // The library dir is not created here: a refresh that matches everything // writes nothing, and importDownloadedItems creates it when a second pass // actually has an item to add. baseLibraryDir, err := s.resolveBaseLibraryDir(d) if err != nil { return err } entries, err := os.ReadDir(tempDownloadDir) if err != nil { return err } var newURLs []string for _, entry := range entries { if err := ctx.Err(); err != nil { return err } if !entry.IsDir() || !strings.HasPrefix(entry.Name(), "item-") { continue } itemDir := filepath.Join(tempDownloadDir, entry.Name()) infoJSONPath := findInfoJSON(itemDir) if infoJSONPath == "" { continue } info := readInfoJSON(infoJSONPath) if info.ID == "" { continue } if existing, ok := s.librarySvc.FindByVideoID(baseLibraryDir, info.ID); ok { if err := s.applyMetadata(existing, info, infoJSONPath); err != nil { log.Printf("warning: failed to refresh metadata for %s: %v", itemDir, err) } continue } if info.WebpageURL != "" { newURLs = append(newURLs, info.WebpageURL) } } if len(newURLs) == 0 { return nil } return s.downloadFresh(ctx, d, preset, newURLs, ytdlpFlags) } // downloadFresh fetches the given item URLs as full downloads (media + info.json) // and imports them into the download's library directory. Metadata mode uses this // to add entries that don't exist in the library yet. func (s *DownloadService) downloadFresh(ctx context.Context, d *models.Download, preset *models.Preset, urls []string, ytdlpFlags string) error { tempDir := s.tempNewDirFor(d.ID) if err := os.MkdirAll(tempDir, 0755); err != nil { return err } defer os.RemoveAll(tempDir) args := s.presetSvc.BuildArgs(preset, d.FormatOverride, d.CustomFlags) args, cleanup := s.appendCookies(args) defer cleanup() args = append(args, "--write-info-json") args = append(args, "-P", tempDir) args = append(args, "-o", "item-%(autonumber)05d/%(title)s.%(ext)s") args = append(args, urls...) runErr := s.runYTDLP(ctx, d, args) // A cancelled second pass has nothing worth importing. if ctx.Err() != nil { return runErr } // Import whatever succeeded even if some entries errored. if _, err := s.importDownloadedItems(ctx, d, tempDir, "", ytdlpFlags); err != nil { log.Printf("warning: failed to import new metadata-mode items: %v", err) } return runErr } // mergeInfoJSON keeps fields from the existing sidecar that are absent from a // metadata-only refresh. In particular, comments and heatmap data are expensive // to reacquire and must not disappear just because the refresh preset does not // request them. func mergeInfoJSON(oldData, newData []byte) ([]byte, error) { var oldObject, newObject map[string]json.RawMessage if err := json.Unmarshal(newData, &newObject); err != nil { return nil, err } if err := json.Unmarshal(oldData, &oldObject); err != nil { return newData, nil } if newObject == nil { return newData, nil } merged := make(map[string]json.RawMessage, len(oldObject)+len(newObject)) for key, value := range oldObject { merged[key] = value } for key, value := range newObject { merged[key] = value } for _, key := range []string{"comments", "heatmap"} { oldValue, hadOldValue := oldObject[key] newValue, hasNewValue := newObject[key] if hadOldValue && (!hasNewValue || isEmptyJSONArray(newValue)) { merged[key] = oldValue } } return json.Marshal(merged) } func isEmptyJSONArray(value json.RawMessage) bool { var values []json.RawMessage if err := json.Unmarshal(value, &values); err != nil { return false } return len(values) == 0 } // restoreFileAtomically puts data back at path without exposing a partial file. // It is used to roll back the marker if installing the staged info sidecar fails. func restoreFileAtomically(path string, data []byte) error { tmp, err := os.CreateTemp(filepath.Dir(path), ".vidarchive-restore-*.tmp") if err != nil { return err } tmpPath := tmp.Name() defer os.Remove(tmpPath) if err := tmp.Chmod(0644); err != nil { tmp.Close() return err } if _, err := tmp.Write(data); err != nil { tmp.Close() return err } if err := tmp.Close(); err != nil { return err } if err := os.Rename(tmpPath, path); err != nil { return err } return nil } // applyMetadata refreshes an existing item's marker and info sidecar from a // fresh info.json without touching its media. func (s *DownloadService) applyMetadata(existing string, info infoJSON, sourceInfoJSON string) error { meta, err := s.librarySvc.readMetadata(existing) if err != nil { return fmt.Errorf("read metadata for %s: %w", existing, err) } if info.Title != "" { meta.Name = info.Title } if info.Description != "" { meta.Description = info.Description } if meta.SourceURL == "" && info.WebpageURL != "" { meta.SourceURL = info.WebpageURL } meta.VideoID = info.ID markerPath := filepath.Join(existing, itemMarkerName) oldMarker, err := os.ReadFile(markerPath) if err != nil { return fmt.Errorf("read existing marker: %w", err) } // Stage the sidecar before changing the marker. The marker is committed first; // if installing the sidecar then fails, restore the old marker so an ordinary // I/O error cannot leave the two metadata files out of sync. stagedInfo := "" defer func() { if stagedInfo != "" { _ = os.Remove(stagedInfo) } }() if sourceInfoJSON != "" { data, err := os.ReadFile(sourceInfoJSON) if err != nil { return fmt.Errorf("read refreshed info JSON: %w", err) } if oldInfoJSON := findInfoJSON(existing); oldInfoJSON != "" { oldData, err := os.ReadFile(oldInfoJSON) if err != nil { return fmt.Errorf("read existing info JSON: %w", err) } data, err = mergeInfoJSON(oldData, data) if err != nil { return fmt.Errorf("merge refreshed info JSON: %w", err) } } tmp, err := os.CreateTemp(existing, ".info-json-*.tmp") if err != nil { return fmt.Errorf("create refreshed info JSON: %w", err) } stagedInfo = tmp.Name() if err := tmp.Chmod(0644); err != nil { tmp.Close() return fmt.Errorf("set refreshed info JSON permissions: %w", err) } if _, err := tmp.Write(data); err != nil { tmp.Close() return fmt.Errorf("write refreshed info JSON: %w", err) } if err := tmp.Close(); err != nil { return fmt.Errorf("close refreshed info JSON: %w", err) } } if err := s.librarySvc.writeMetadata(existing, meta); err != nil { return err } if stagedInfo != "" { if err := os.Rename(stagedInfo, filepath.Join(existing, "info.json")); err != nil { if restoreErr := restoreFileAtomically(markerPath, oldMarker); restoreErr != nil { return fmt.Errorf("install refreshed info JSON: %v; restore marker: %w", err, restoreErr) } return fmt.Errorf("install refreshed info JSON: %w", err) } stagedInfo = "" } if rel, err := filepath.Rel(s.cfg.LibraryDir, existing); err == nil { s.librarySvc.evictCachedScan(filepath.ToSlash(rel)) } return nil } // isInfoJSON reports whether a file name is yt-dlp's metadata sidecar. yt-dlp // writes ".info.json" next to the media, but a bare "info.json" is what // an already-imported item holds. func isInfoJSON(name string) bool { return name == "info.json" || strings.HasSuffix(name, ".info.json") } // findInfoJSON returns the path to an info.json directly inside itemDir, or "". func findInfoJSON(itemDir string) string { entries, err := os.ReadDir(itemDir) if err != nil { return "" } for _, entry := range entries { if entry.IsDir() { continue } name := entry.Name() if isInfoJSON(name) { return filepath.Join(itemDir, name) } } return "" } // pruneSubscription mirrors the source by deleting items in the subscription's // directory that are no longer present upstream. It enumerates the current id // set with a cheap flat-playlist listing; it never prunes when that enumeration // fails or returns nothing, so a dead URL or network error can't wipe the dir. func (s *DownloadService) pruneSubscription(ctx context.Context, d *models.Download, sub *models.Subscription) { baseLibraryDir, err := s.resolveBaseLibraryDir(d) if err != nil { log.Printf("subscription %d prune skipped: %v", sub.ID, err) return } keep, err := s.enumeratePlaylistIDs(ctx, sub.URL) if err != nil { log.Printf("subscription %d prune skipped: enumeration failed: %v", sub.ID, err) return } if len(keep) == 0 { log.Printf("subscription %d prune skipped: source returned no entries", sub.ID) return } removed, err := s.librarySvc.PruneToIDSet(baseLibraryDir, keep) if err != nil { log.Printf("subscription %d prune error: %v", sub.ID, err) return } if removed > 0 { log.Printf("subscription %d pruned %d item(s) removed upstream", sub.ID, removed) } } // enumeratePlaylistIDs lists the current video-id set for a URL without // downloading, using yt-dlp --flat-playlist. Cookies are applied so private // playlists enumerate correctly. Ids alone are sufficient to match items within // a subscription's own directory (see FindByVideoID / PruneToIDSet). func (s *DownloadService) enumeratePlaylistIDs(ctx context.Context, url string) (map[string]bool, error) { args := []string{"--flat-playlist", "--no-warnings", "--print", "%(id)s"} args, cleanup := s.appendCookies(args) defer cleanup() args = append(args, url) out, err := exec.CommandContext(ctx, s.cfg.YTDLPPath, args...).Output() if err != nil { return nil, err } keep := make(map[string]bool) for _, line := range strings.Split(string(out), "\n") { id := strings.TrimSpace(line) // yt-dlp prints "NA" for a missing field; never treat that as a real id. if id == "" || id == "NA" { continue } keep[id] = true } return keep, nil } // deriveItemName names the imported item after its title, falling back to the // largest media file's base name when there is no usable info.json. func (s *DownloadService) deriveItemName(itemDir string, info infoJSON, mediaFiles []os.DirEntry) string { if info.Title != "" { return sanitizeDirName(info.Title) } var largest os.DirEntry var maxSize int64 for _, f := range mediaFiles { st, err := os.Stat(filepath.Join(itemDir, f.Name())) if err == nil && (largest == nil || st.Size() > maxSize) { largest, maxSize = f, st.Size() } } if largest == nil { largest = mediaFiles[0] } base := strings.TrimSuffix(largest.Name(), filepath.Ext(largest.Name())) return sanitizeDirName(base) } func (s *DownloadService) uniqueDir(base, name string) string { dir := filepath.Join(base, name) if _, err := os.Stat(dir); os.IsNotExist(err) { return dir } for i := 1; ; i++ { candidate := fmt.Sprintf("%s-%d", dir, i) if _, err := os.Stat(candidate); os.IsNotExist(err) { return candidate } } } // moveFile moves src to dst, falling back to copy-and-delete when the two are on // different filesystems. The temp and library directories are independently // configurable, so they can legitimately live on separate mounts — where a plain // rename fails with EXDEV. func moveFile(src, dst string) error { if err := os.Rename(src, dst); err == nil { return nil } else if !errors.Is(err, syscall.EXDEV) { return err } in, err := os.Open(src) if err != nil { return err } defer in.Close() info, err := in.Stat() if err != nil { return err } out, err := os.OpenFile(dst, os.O_WRONLY|os.O_CREATE|os.O_TRUNC, info.Mode()) if err != nil { return err } if _, err := io.Copy(out, in); err != nil { out.Close() os.Remove(dst) return err } // Close explicitly: a deferred close would hide a flush error on the copy. if err := out.Close(); err != nil { os.Remove(dst) return err } return os.Remove(src) } func sanitizeDirName(name string) string { name = strings.TrimSpace(name) replacer := strings.NewReplacer( "/", "-", "\\", "-", ":", "-", "*", "-", "?", "-", "\"", "-", "<", "-", ">", "-", "|", "-", ) name = replacer.Replace(name) name = strings.TrimSpace(name) if name == "" { name = "untitled" } return name } // probeDuration returns the duration of a media file in whole seconds. The bool // is false when ffprobe is unavailable or the file has no usable duration. func probeDuration(ctx context.Context, ffprobePath, path string) (int, bool) { out, err := exec.CommandContext(ctx, ffprobePath, "-v", "error", "-show_entries", "format=duration", "-of", "default=nw=1:nk=1", path).Output() if err != nil { return 0, false } f, err := strconv.ParseFloat(strings.TrimSpace(string(out)), 64) if err != nil || f <= 0 { return 0, false } return int(f + 0.5), true } // ytFormat mirrors the subset of yt-dlp's per-format JSON (-J) we surface. // Numeric fields are pointers so an absent value (null/omitted) is distinct // from a real zero. type ytFormat struct { FormatID string `json:"format_id"` Ext string `json:"ext"` Resolution string `json:"resolution"` Width *int `json:"width"` Height *int `json:"height"` FPS *float64 `json:"fps"` VCodec string `json:"vcodec"` ACodec string `json:"acodec"` AudioChannels *int `json:"audio_channels"` Filesize *int64 `json:"filesize"` FilesizeApprox *int64 `json:"filesize_approx"` FormatNote string `json:"format_note"` } // parseFormatJSON reads yt-dlp's single-JSON dump (-J) and returns the available // formats. For a single video the formats live at the top level; for a playlist // URL we fall back to the first entry's formats so the picker still shows // something useful. func parseFormatJSON(data []byte) ([]*models.FormatInfo, error) { var top struct { Formats []ytFormat `json:"formats"` Entries []struct { Formats []ytFormat `json:"formats"` } `json:"entries"` } if err := json.Unmarshal(data, &top); err != nil { return nil, fmt.Errorf("parse yt-dlp JSON: %w", err) } raw := top.Formats if len(raw) == 0 && len(top.Entries) > 0 { raw = top.Entries[0].Formats } formats := make([]*models.FormatInfo, 0, len(raw)) for _, f := range raw { formats = append(formats, f.toFormatInfo()) } return formats, nil } func (f ytFormat) toFormatInfo() *models.FormatInfo { fi := &models.FormatInfo{ ID: f.FormatID, Ext: f.Ext, Note: f.FormatNote, } switch { case f.Resolution != "": fi.Resolution = f.Resolution case f.Width != nil && f.Height != nil && *f.Width > 0 && *f.Height > 0: fi.Resolution = fmt.Sprintf("%dx%d", *f.Width, *f.Height) } if f.FPS != nil && *f.FPS > 0 { fi.FPS = strconv.FormatFloat(*f.FPS, 'f', -1, 64) } if f.AudioChannels != nil && *f.AudioChannels > 0 { fi.Channels = strconv.Itoa(*f.AudioChannels) } // Prefer the video codec; fall back to the audio codec for audio-only formats. if f.VCodec != "" && f.VCodec != "none" { fi.Codec = f.VCodec } else if f.ACodec != "" && f.ACodec != "none" { fi.Codec = f.ACodec } if f.Filesize != nil && *f.Filesize > 0 { fi.FileSize = util.FormatBytes(*f.Filesize) } else if f.FilesizeApprox != nil && *f.FilesizeApprox > 0 { fi.FileSize = "~" + util.FormatBytes(*f.FilesizeApprox) } return fi } func sqlNullInt64(v int64) sql.NullInt64 { return sql.NullInt64{Int64: v, Valid: true} } // reservedFlags are yt-dlp options VidArchive always sets itself; user custom // flags must not pass them (or a conflicting inverse). The value describes what // the option controls, for the failure message. var reservedFlags = map[string]string{ "-o": "the output template", "--output": "the output template", "-P": "the download path", "--paths": "the download path", "--cookies": "cookies (set these in Settings instead)", "--no-cookies": "cookies (set these in Settings instead)", "--newline": "progress output formatting (VidArchive sets this to stream logs)", // These hand yt-dlp an arbitrary command or binary to run. VidArchive passes // custom flags through verbatim, so allowing them would turn the preset form // into remote command execution. "--exec": "running external commands (not permitted)", "--exec-before-download": "running external commands (not permitted)", "--postprocessor-args": "post-processor arguments (not permitted)", "--ppa": "post-processor arguments (not permitted)", "--downloader": "selecting an external downloader (not permitted)", "--external-downloader": "selecting an external downloader (not permitted)", "--downloader-args": "external downloader arguments (not permitted)", "--external-downloader-args": "external downloader arguments (not permitted)", } // reservedSubscriptionFlags are additionally reserved for subscription runs, // where VidArchive drives info-json writing and the refresh mode. var reservedSubscriptionFlags = map[string]string{ "--write-info-json": "info-json writing (needed to track item identity)", "--no-write-info-json": "info-json writing (needed to track item identity)", "--download-archive": "the download archive (managed by Skip mode)", "--no-download-archive": "the download archive (managed by Skip mode)", "--skip-download": "media downloading (managed by Metadata mode)", "--no-skip-download": "media downloading (managed by Metadata mode)", } // checkReservedFlags rejects custom flags that clash with options VidArchive // controls, naming the offender. It matches both "--flag" and "--flag=value". func checkReservedFlags(customFlags string, isSubscription bool) error { for _, tok := range strings.Fields(customFlags) { // Both "--flag value" and "--flag=value" name the same option. name, _, _ := strings.Cut(tok, "=") desc, ok := reservedFlags[name] if !ok && isSubscription { desc, ok = reservedSubscriptionFlags[name] } if ok { return fmt.Errorf("custom flag %q conflicts with VidArchive's handling of %s; remove it and try again", tok, desc) } } return nil }