package service import ( "bufio" "bytes" "database/sql" "encoding/json" "fmt" "log" "os" "os/exec" "path/filepath" "sort" "strconv" "strings" "sync" "syscall" "time" "github.com/BurntSushi/toml" "github.com/gabriel-vasile/mimetype" "vidarchive/internal/config" "vidarchive/internal/models" "vidarchive/internal/repository" ) type DownloadService struct { repo *repository.DownloadRepository librarySvc *LibraryService presetSvc *PresetService settingsSvc *SettingsService subscriptionSvc *SubscriptionService cfg *config.Config cache *ProgressCache processMu sync.Mutex processes map[int64]*os.Process } func NewDownloadService(repo *repository.DownloadRepository, librarySvc *LibraryService, presetSvc *PresetService, settingsSvc *SettingsService, subscriptionSvc *SubscriptionService, cfg *config.Config) *DownloadService { return &DownloadService{ repo: repo, librarySvc: librarySvc, presetSvc: presetSvc, settingsSvc: settingsSvc, subscriptionSvc: subscriptionSvc, cfg: cfg, cache: NewProgressCache(), processes: make(map[int64]*os.Process), } } func (s *DownloadService) Create(url string, presetID *int64, formatOverride, customFlags, outputDir string) (*models.Download, error) { d := &models.Download{ URL: url, Status: "queued", FormatOverride: formatOverride, CustomFlags: customFlags, OutputDir: sql.NullString{String: outputDir, Valid: outputDir != ""}, } if presetID != nil { d.PresetID = sqlNullInt64(*presetID) } if err := s.repo.Create(d); err != nil { return nil, err } return d, nil } // CreateForSubscription queues a download for a subscription run, copying its // download options and tagging it with the subscription id so ExecuteDownload // applies the right refresh mode and pruning. func (s *DownloadService) CreateForSubscription(sub *models.Subscription) (*models.Download, error) { d := &models.Download{ URL: sub.URL, Status: "queued", FormatOverride: sub.FormatOverride, CustomFlags: sub.CustomFlags, OutputDir: sql.NullString{String: sub.OutputDir, Valid: sub.OutputDir != ""}, PresetID: sub.PresetID, SubscriptionID: sqlNullInt64(sub.ID), } if err := s.repo.Create(d); err != nil { return nil, err } return d, nil } func (s *DownloadService) GetByID(id int64) (*models.Download, error) { d, err := s.repo.GetByID(id) if err != nil { return nil, err } if logs := s.cache.Snapshot(id); logs != "" { d.Logs = sql.NullString{String: logs, Valid: true} } return d, nil } func (s *DownloadService) GetAll(status, sortBy string) ([]*models.Download, error) { downloads, err := s.repo.GetAll(status, sortBy) if err != nil { return nil, err } for _, d := range downloads { if logs := s.cache.Snapshot(d.ID); logs != "" { d.Logs = sql.NullString{String: logs, Valid: true} } } return downloads, nil } func (s *DownloadService) GetQueued(limit int) ([]*models.Download, error) { return s.repo.GetQueued(limit) } // HasActiveForSubscription reports whether the subscription already has a queued // or in-progress download, so the scheduler can skip stacking another run. func (s *DownloadService) HasActiveForSubscription(subID int64) (bool, error) { return s.repo.HasActiveForSubscription(subID) } func (s *DownloadService) Delete(id int64) error { s.killProcess(id) s.cache.Delete(id) return s.repo.Delete(id) } func (s *DownloadService) killProcess(id int64) { s.processMu.Lock() proc, ok := s.processes[id] delete(s.processes, id) s.processMu.Unlock() if !ok || proc == nil { return } _ = syscall.Kill(-proc.Pid, syscall.SIGKILL) } func (s *DownloadService) DeleteAll() error { return s.repo.DeleteAll() } func (s *DownloadService) ResetStalledDownloads() error { return s.repo.UpdateStatusWhere("downloading", "queued") } func (s *DownloadService) ListFormats(url string) ([]*models.FormatInfo, error) { // Use machine-readable JSON (-J) rather than scraping the human "-F" table, // whose columns/separators shift between yt-dlp versions. stderr is captured // separately so warnings can't corrupt the JSON on stdout. cmd := exec.Command(s.cfg.YTDLPPath, "-J", "--no-warnings", url) var stderr bytes.Buffer cmd.Stderr = &stderr output, err := cmd.Output() if err != nil { return nil, fmt.Errorf("yt-dlp -J failed: %w\n%s", err, stderr.String()) } return parseFormatJSON(output) } // ExecuteDownload runs the download for d. The bool reports whether this call // actually processed it: false means another worker already claimed it (Submit // and the queue checker can both enqueue the same row within the 2s poll window), // so the caller should not log it as completed. func (s *DownloadService) ExecuteDownload(d *models.Download) (bool, error) { // Atomically claim the download. If it's no longer queued, another worker // already took it — bail rather than download it twice. claimed, err := s.repo.MarkStarted(d.ID) if err != nil { return false, err } if !claimed { return false, nil } s.cache.Set(d.ID, &LiveDownload{LastUpdate: time.Now()}) defer s.cache.Delete(d.ID) var preset *models.Preset if d.PresetID.Valid { preset, err = s.presetSvc.GetByID(d.PresetID.Int64) if err != nil { preset, _ = s.presetSvc.GetDefault() } } else { preset, _ = s.presetSvc.GetDefault() } if preset == nil { preset = &models.Preset{} } var sub *models.Subscription if d.SubscriptionID.Valid && s.subscriptionSvc != nil { sub, _ = s.subscriptionSvc.GetByID(d.SubscriptionID.Int64) } // Reject custom flags that clash with options VidArchive sets itself, before // spending any work — the download fails with a message naming the offender. isSubscription := d.SubscriptionID.Valid if err := checkReservedFlags(d.CustomFlags, isSubscription); err != nil { s.finalizeError(d.ID, err) return false, err } if err := checkReservedFlags(preset.CustomFlags, isSubscription); err != nil { s.finalizeError(d.ID, err) return false, err } tempDownloadDir := filepath.Join(s.cfg.TempDir, fmt.Sprintf("%d", d.ID)) if err := os.MkdirAll(tempDownloadDir, 0755); err != nil { return false, fmt.Errorf("create temp download dir: %w", err) } args := s.presetSvc.BuildArgs(preset, d.FormatOverride, d.CustomFlags) // Record the meaningful flags (format/audio/subs/custom) that shaped this // download, before the internal plumbing (cookies, -P/-o, URL) is appended, // so each imported item can show how it was fetched. ytdlpFlags := strings.Join(args, " ") var cookieCleanup func() args, cookieCleanup = s.appendCookies(args) defer cookieCleanup() if sub != nil { // Always write info.json so the import step can read the stable identity // (yt-dlp's video id) used to match/replace existing items. args = append(args, "--write-info-json") switch sub.RefreshMode { case "skip": // Let yt-dlp skip entries already recorded — no re-download. archive := s.subscriptionSvc.ArchivePath(sub.ID) if err := os.MkdirAll(filepath.Dir(archive), 0755); err == nil { args = append(args, "--download-archive", archive) } case "metadata": // Refresh metadata only; don't fetch media. args = append(args, "--skip-download") } } args = append(args, "-P", tempDownloadDir) args = append(args, "-o", "item-%(autonumber)05d/%(title)s.%(ext)s") args = append(args, d.URL) if err := s.runYTDLP(d, args); err != nil { s.finalizeError(d.ID, err) return false, err } mode := "" if sub != nil { mode = sub.RefreshMode } // Post-process before marking completed, so the download stays "downloading" // until everything is really done — including metadata mode's second pass, // which downloads any genuinely new entries as full items. if mode == "metadata" { // The main pass ran with --skip-download, so the temp dir holds only // info.json files: refresh existing items in place and fetch new ones. if err := s.refreshAndAddNew(d, preset, tempDownloadDir, ytdlpFlags); err != nil { s.finalizeError(d.ID, err) return false, err } } else { imported, err := s.importDownloadedItems(d, tempDownloadDir, mode, ytdlpFlags) if err != nil { s.finalizeError(d.ID, err) return false, err } // A plain (non-subscription) download that yields nothing is a failure, not // a silent "completed". Subscription modes legitimately import zero (skip // mode, or a metadata refresh with no new entries), so only enforce this for // plain runs. if sub == nil && imported == 0 { err := fmt.Errorf("yt-dlp finished but no media files were downloaded") s.finalizeError(d.ID, err) return false, err } } if sub != nil && sub.PruneRemoved { s.pruneSubscription(d, sub) } if err := s.repo.MarkCompleted(d.ID, "completed"); err != nil { return false, err } return true, nil } // runYTDLP executes yt-dlp with args, streaming combined output into the live // progress cache and periodically flushing it to the download's persisted log. // It registers the process so Delete can kill it, and returns the exit error. func (s *DownloadService) runYTDLP(d *models.Download, args []string) error { // --newline forces yt-dlp to emit each progress update on its own line. Without // it, progress is rewritten in place with carriage returns, so a long download // becomes one ever-growing line that overflows the reader's buffer and stalls // the pipe — hanging the download. See the hardened scanner below. fullArgs := append([]string{"--newline"}, args...) cmd := exec.Command(s.cfg.YTDLPPath, fullArgs...) cmd.SysProcAttr = &syscall.SysProcAttr{Setpgid: true} stdout, err := cmd.StdoutPipe() if err != nil { return err } cmd.Stderr = cmd.Stdout if err := cmd.Start(); err != nil { return err } s.processMu.Lock() s.processes[d.ID] = cmd.Process s.processMu.Unlock() defer func() { s.processMu.Lock() delete(s.processes, d.ID) s.processMu.Unlock() }() ticker := time.NewTicker(10 * time.Second) defer ticker.Stop() done := make(chan struct{}) go func() { for { select { case <-ticker.C: if logs := s.cache.FlushLogs(d.ID); logs != "" { s.repo.AppendLogs(d.ID, logs) } case <-done: return } } }() scanner := bufio.NewScanner(stdout) // Allow long lines (a single yt-dlp message can exceed the 64 KiB default) // rather than letting the scanner abort and leave the pipe unread. scanner.Buffer(make([]byte, 0, 64*1024), 1024*1024) for scanner.Scan() { s.cache.AppendLog(d.ID, scanner.Text()) } if err := scanner.Err(); err != nil { log.Printf("download %d: error reading yt-dlp output: %v", d.ID, err) } close(done) if logs := s.cache.FlushLogs(d.ID); logs != "" { s.repo.AppendLogs(d.ID, logs) } return cmd.Wait() } // appendCookies writes the saved cookies (if any) to a temp file and appends a // --cookies flag. The returned cleanup removes the temp file and is always safe // to call, even when no cookies were configured. func (s *DownloadService) appendCookies(args []string) ([]string, func()) { cleanup := func() {} cookies, err := s.settingsSvc.GetCookies() if err != nil || strings.TrimSpace(cookies) == "" { return args, cleanup } tmpFile, err := os.CreateTemp("", "cookies-*.txt") if err != nil { return args, cleanup } // A short write would hand yt-dlp a truncated cookies file; on any write/close // failure, drop the temp file and proceed without cookies rather than silently // using a broken one. if _, err := tmpFile.WriteString(cookies); err != nil { tmpFile.Close() os.Remove(tmpFile.Name()) return args, cleanup } if err := tmpFile.Close(); err != nil { os.Remove(tmpFile.Name()) return args, cleanup } return append(args, "--cookies", tmpFile.Name()), func() { os.Remove(tmpFile.Name()) } } func (s *DownloadService) finalizeError(id int64, err error) { logs := s.cache.FlushLogs(id) if logs != "" { s.repo.AppendLogs(id, logs) } s.repo.MarkError(id, err.Error()) } // resolveBaseLibraryDir returns the absolute library directory a download writes // into, applying the optional per-download OutputDir while rejecting any path // that escapes the library root. func (s *DownloadService) resolveBaseLibraryDir(d *models.Download) (string, error) { baseLibraryDir := s.cfg.LibraryDir if d.OutputDir.Valid && d.OutputDir.String != "" { cleanDir := filepath.Clean(d.OutputDir.String) fullPath := filepath.Join(baseLibraryDir, cleanDir) resolvedPath, err := filepath.Abs(fullPath) if err != nil { return "", fmt.Errorf("invalid output directory: %w", err) } resolvedLibraryDir, _ := filepath.Abs(baseLibraryDir) if !strings.HasPrefix(resolvedPath, resolvedLibraryDir+string(filepath.Separator)) && resolvedPath != resolvedLibraryDir { return "", fmt.Errorf("invalid output directory: path traversal attempt detected") } baseLibraryDir = fullPath } return baseLibraryDir, nil } // importDownloadedItems moves each downloaded item from the temp dir into the // library and returns the number of items successfully imported. Per-item // failures are logged and skipped (a playlist with a few bad entries still // imports the rest); a non-nil error means the import couldn't even start. func (s *DownloadService) importDownloadedItems(d *models.Download, tempDownloadDir, mode, ytdlpFlags string) (int, error) { entries, err := os.ReadDir(tempDownloadDir) if err != nil { return 0, err } baseLibraryDir, err := s.resolveBaseLibraryDir(d) if err != nil { return 0, err } if err := os.MkdirAll(baseLibraryDir, 0755); err != nil { return 0, err } var itemDirs []string for _, entry := range entries { if !entry.IsDir() { continue } name := entry.Name() if strings.HasPrefix(name, "item-") { itemDirs = append(itemDirs, filepath.Join(tempDownloadDir, name)) } } sort.Strings(itemDirs) imported := 0 for _, itemDir := range itemDirs { if err := s.importItemDir(d.URL, itemDir, baseLibraryDir, mode, ytdlpFlags); err != nil { log.Printf("warning: failed to import item %s: %v", itemDir, err) continue } imported++ } // RemoveAll (not Remove): leftover item dirs from failed imports, plus any // orphaned thumbnail/image files yt-dlp left behind, would otherwise keep the // temp dir non-empty and leak it forever. os.RemoveAll(tempDownloadDir) return imported, nil } func (s *DownloadService) importItemDir(url, itemDir, baseLibraryDir, mode, ytdlpFlags string) error { entries, err := os.ReadDir(itemDir) if err != nil { return err } var mediaFiles []os.DirEntry var infoJSONPath string var subtitleFiles []string for _, entry := range entries { if entry.IsDir() { continue } name := entry.Name() path := filepath.Join(itemDir, name) ext := strings.ToLower(filepath.Ext(name)) if name == "info.json" || strings.HasSuffix(name, ".info.json") { infoJSONPath = path continue } if ext == ".vtt" || ext == ".srt" || ext == ".ass" || ext == ".ssa" { subtitleFiles = append(subtitleFiles, path) continue } mtype, err := mimetype.DetectFile(path) if err == nil && mtype != nil && (strings.HasPrefix(mtype.String(), "audio/") || strings.HasPrefix(mtype.String(), "video/")) { mediaFiles = append(mediaFiles, entry) } } if len(mediaFiles) == 0 { return fmt.Errorf("no media files found in %s", itemDir) } name := s.deriveItemName(itemDir, infoJSONPath, mediaFiles) videoID := readInfoID(infoJSONPath) // Overwrite mode: replace the existing copy of this video in place rather than // creating a duplicate folder. Removing the old dir lets uniqueDir reuse its // name (or land on the new title if it changed upstream). if mode == "overwrite" && videoID != "" { if existing, ok := s.librarySvc.FindByVideoID(baseLibraryDir, videoID); ok { if rel, err := filepath.Rel(s.cfg.LibraryDir, existing); err == nil { s.librarySvc.evictCachedScan(filepath.ToSlash(rel)) } os.RemoveAll(existing) } } targetDir := s.uniqueDir(baseLibraryDir, name) if err := os.MkdirAll(targetDir, 0755); err != nil { return err } if infoJSONPath != "" { if err := os.Rename(infoJSONPath, filepath.Join(targetDir, "info.json")); err != nil { return err } } for _, entry := range mediaFiles { if err := os.Rename(filepath.Join(itemDir, entry.Name()), filepath.Join(targetDir, entry.Name())); err != nil { return err } } // Probe each media file's duration once, here in the worker (off the request // path), and cache it in the marker so the library never has to probe while // serving pages. Files we can't probe simply get no duration. fileDurations := make(map[string]int) for _, entry := range mediaFiles { if d, ok := probeDuration(s.cfg.FFprobePath, filepath.Join(targetDir, entry.Name())); ok { fileDurations[entry.Name()] = d } } if len(subtitleFiles) > 0 { subtitlesDir := filepath.Join(targetDir, subtitlesDirName) if err := os.MkdirAll(subtitlesDir, 0755); err != nil { return err } for _, sf := range subtitleFiles { if err := os.Rename(sf, filepath.Join(subtitlesDir, filepath.Base(sf))); err != nil { return err } } } metadata := models.ItemMetadata{ Name: name, SourceURL: url, VideoID: videoID, YtdlpFlags: ytdlpFlags, FileDurations: fileDurations, } markerPath := filepath.Join(targetDir, itemMarkerName) f, err := os.Create(markerPath) if err != nil { return err } defer f.Close() if err := toml.NewEncoder(f).Encode(metadata); err != nil { return err } return nil } // readInfoID extracts the stable item identity (yt-dlp's video id) from an // info.json. The id alone is sufficient to match items within a single // subscription's owned directory, so the extractor is not used. Returns an // empty string when the file is absent or unreadable. func readInfoID(infoJSONPath string) string { if infoJSONPath == "" { return "" } data, err := os.ReadFile(infoJSONPath) if err != nil { return "" } var info struct { ID string `json:"id"` } if err := json.Unmarshal(data, &info); err != nil { return "" } return info.ID } // refreshAndAddNew handles a metadata-mode run. The main pass used // --skip-download, so tempDownloadDir holds only info.json files. Existing // library items have their markers refreshed in place; entries with no existing // match are genuinely new and are downloaded as full items in a second pass. func (s *DownloadService) refreshAndAddNew(d *models.Download, preset *models.Preset, tempDownloadDir, ytdlpFlags string) error { baseLibraryDir, err := s.resolveBaseLibraryDir(d) if err != nil { return err } if err := os.MkdirAll(baseLibraryDir, 0755); err != nil { return err } entries, err := os.ReadDir(tempDownloadDir) if err != nil { return err } var newURLs []string for _, entry := range entries { if !entry.IsDir() || !strings.HasPrefix(entry.Name(), "item-") { continue } itemDir := filepath.Join(tempDownloadDir, entry.Name()) infoJSONPath := findInfoJSON(itemDir) if infoJSONPath == "" { continue } videoID := readInfoID(infoJSONPath) if videoID == "" { continue } if existing, ok := s.librarySvc.FindByVideoID(baseLibraryDir, videoID); ok { if err := s.applyMetadata(existing, infoJSONPath, videoID); err != nil { log.Printf("warning: failed to refresh metadata for %s: %v", itemDir, err) } continue } if u := readWebpageURL(infoJSONPath); u != "" { newURLs = append(newURLs, u) } } os.RemoveAll(tempDownloadDir) if len(newURLs) == 0 { return nil } return s.downloadFresh(d, preset, newURLs, ytdlpFlags) } // downloadFresh fetches the given item URLs as full downloads (media + info.json) // and imports them into the download's library directory. Metadata mode uses this // to add entries that don't exist in the library yet. func (s *DownloadService) downloadFresh(d *models.Download, preset *models.Preset, urls []string, ytdlpFlags string) error { tempDir := filepath.Join(s.cfg.TempDir, fmt.Sprintf("%d-new", d.ID)) if err := os.MkdirAll(tempDir, 0755); err != nil { return err } args := s.presetSvc.BuildArgs(preset, d.FormatOverride, d.CustomFlags) args, cleanup := s.appendCookies(args) defer cleanup() args = append(args, "--write-info-json") args = append(args, "-P", tempDir) args = append(args, "-o", "item-%(autonumber)05d/%(title)s.%(ext)s") args = append(args, urls...) runErr := s.runYTDLP(d, args) // Import whatever succeeded even if some entries errored. if _, err := s.importDownloadedItems(d, tempDir, "", ytdlpFlags); err != nil { log.Printf("warning: failed to import new metadata-mode items: %v", err) } return runErr } // applyMetadata rewrites an existing item's marker (name/description/identity) // from a fresh info.json without touching its media. func (s *DownloadService) applyMetadata(existing, infoJSONPath, videoID string) error { data, err := os.ReadFile(infoJSONPath) if err != nil { return err } var info struct { Title string `json:"title"` Description string `json:"description"` WebpageURL string `json:"webpage_url"` } _ = json.Unmarshal(data, &info) meta, _ := s.librarySvc.readOrCreateMetadata(existing) if info.Title != "" { meta.Name = info.Title } if info.Description != "" { meta.Description = info.Description } if meta.SourceURL == "" && info.WebpageURL != "" { meta.SourceURL = info.WebpageURL } meta.VideoID = videoID if err := s.librarySvc.writeMetadata(existing, meta); err != nil { return err } if rel, err := filepath.Rel(s.cfg.LibraryDir, existing); err == nil { s.librarySvc.evictCachedScan(filepath.ToSlash(rel)) } return nil } // readWebpageURL returns the canonical entry URL from an info.json, or "". func readWebpageURL(infoJSONPath string) string { data, err := os.ReadFile(infoJSONPath) if err != nil { return "" } var info struct { WebpageURL string `json:"webpage_url"` } if err := json.Unmarshal(data, &info); err != nil { return "" } return info.WebpageURL } // findInfoJSON returns the path to an info.json directly inside itemDir, or "". func findInfoJSON(itemDir string) string { entries, err := os.ReadDir(itemDir) if err != nil { return "" } for _, entry := range entries { if entry.IsDir() { continue } name := entry.Name() if name == "info.json" || strings.HasSuffix(name, ".info.json") { return filepath.Join(itemDir, name) } } return "" } // pruneSubscription mirrors the source by deleting items in the subscription's // directory that are no longer present upstream. It enumerates the current id // set with a cheap flat-playlist listing; it never prunes when that enumeration // fails or returns nothing, so a dead URL or network error can't wipe the dir. func (s *DownloadService) pruneSubscription(d *models.Download, sub *models.Subscription) { baseLibraryDir, err := s.resolveBaseLibraryDir(d) if err != nil { log.Printf("subscription %d prune skipped: %v", sub.ID, err) return } keep, err := s.enumeratePlaylistIDs(sub.URL) if err != nil { log.Printf("subscription %d prune skipped: enumeration failed: %v", sub.ID, err) return } if len(keep) == 0 { log.Printf("subscription %d prune skipped: source returned no entries", sub.ID) return } removed, err := s.librarySvc.PruneToIDSet(baseLibraryDir, keep) if err != nil { log.Printf("subscription %d prune error: %v", sub.ID, err) return } if removed > 0 { log.Printf("subscription %d pruned %d item(s) removed upstream", sub.ID, removed) } } // enumeratePlaylistIDs lists the current video-id set for a URL without // downloading, using yt-dlp --flat-playlist. Cookies are applied so private // playlists enumerate correctly. Ids alone are sufficient to match items within // a subscription's own directory (see FindByVideoID / PruneToIDSet). func (s *DownloadService) enumeratePlaylistIDs(url string) (map[string]bool, error) { args := []string{"--flat-playlist", "--no-warnings", "--print", "%(id)s"} args, cleanup := s.appendCookies(args) defer cleanup() args = append(args, url) out, err := exec.Command(s.cfg.YTDLPPath, args...).Output() if err != nil { return nil, err } keep := make(map[string]bool) for _, line := range strings.Split(string(out), "\n") { id := strings.TrimSpace(line) // yt-dlp prints "NA" for a missing field; never treat that as a real id. if id == "" || id == "NA" { continue } keep[id] = true } return keep, nil } func (s *DownloadService) deriveItemName(itemDir, infoJSONPath string, mediaFiles []os.DirEntry) string { if infoJSONPath != "" { data, err := os.ReadFile(infoJSONPath) if err == nil { var info struct { Title string `json:"title"` } if err := json.Unmarshal(data, &info); err == nil && info.Title != "" { return sanitizeDirName(info.Title) } } } sort.Slice(mediaFiles, func(i, j int) bool { ii, _ := os.Stat(filepath.Join(itemDir, mediaFiles[i].Name())) jj, _ := os.Stat(filepath.Join(itemDir, mediaFiles[j].Name())) if ii == nil || jj == nil { return false } return ii.Size() > jj.Size() }) base := strings.TrimSuffix(mediaFiles[0].Name(), filepath.Ext(mediaFiles[0].Name())) return sanitizeDirName(base) } func (s *DownloadService) uniqueDir(base, name string) string { dir := filepath.Join(base, name) if _, err := os.Stat(dir); os.IsNotExist(err) { return dir } for i := 1; ; i++ { candidate := fmt.Sprintf("%s-%d", dir, i) if _, err := os.Stat(candidate); os.IsNotExist(err) { return candidate } } } func sanitizeDirName(name string) string { name = strings.TrimSpace(name) replacer := strings.NewReplacer( "/", "-", "\\", "-", ":", "-", "*", "-", "?", "-", "\"", "-", "<", "-", ">", "-", "|", "-", ) name = replacer.Replace(name) name = strings.TrimSpace(name) if name == "" { name = "untitled" } return name } // probeDuration returns the duration of a media file in whole seconds. The bool // is false when ffprobe is unavailable or the file has no usable duration. func probeDuration(ffprobePath, path string) (int, bool) { out, err := exec.Command(ffprobePath, "-v", "error", "-show_entries", "format=duration", "-of", "default=nw=1:nk=1", path).Output() if err != nil { return 0, false } f, err := strconv.ParseFloat(strings.TrimSpace(string(out)), 64) if err != nil || f <= 0 { return 0, false } return int(f + 0.5), true } // ytFormat mirrors the subset of yt-dlp's per-format JSON (-J) we surface. // Numeric fields are pointers so an absent value (null/omitted) is distinct // from a real zero. type ytFormat struct { FormatID string `json:"format_id"` Ext string `json:"ext"` Resolution string `json:"resolution"` Width *int `json:"width"` Height *int `json:"height"` FPS *float64 `json:"fps"` VCodec string `json:"vcodec"` ACodec string `json:"acodec"` AudioChannels *int `json:"audio_channels"` Filesize *int64 `json:"filesize"` FilesizeApprox *int64 `json:"filesize_approx"` FormatNote string `json:"format_note"` } // parseFormatJSON reads yt-dlp's single-JSON dump (-J) and returns the available // formats. For a single video the formats live at the top level; for a playlist // URL we fall back to the first entry's formats so the picker still shows // something useful. func parseFormatJSON(data []byte) ([]*models.FormatInfo, error) { var top struct { Formats []ytFormat `json:"formats"` Entries []struct { Formats []ytFormat `json:"formats"` } `json:"entries"` } if err := json.Unmarshal(data, &top); err != nil { return nil, fmt.Errorf("parse yt-dlp JSON: %w", err) } raw := top.Formats if len(raw) == 0 && len(top.Entries) > 0 { raw = top.Entries[0].Formats } formats := make([]*models.FormatInfo, 0, len(raw)) for _, f := range raw { formats = append(formats, f.toFormatInfo()) } return formats, nil } func (f ytFormat) toFormatInfo() *models.FormatInfo { fi := &models.FormatInfo{ ID: f.FormatID, Ext: f.Ext, Note: f.FormatNote, } switch { case f.Resolution != "": fi.Resolution = f.Resolution case f.Width != nil && f.Height != nil && *f.Width > 0 && *f.Height > 0: fi.Resolution = fmt.Sprintf("%dx%d", *f.Width, *f.Height) } if f.FPS != nil && *f.FPS > 0 { fi.FPS = strconv.FormatFloat(*f.FPS, 'f', -1, 64) } if f.AudioChannels != nil && *f.AudioChannels > 0 { fi.Channels = strconv.Itoa(*f.AudioChannels) } // Prefer the video codec; fall back to the audio codec for audio-only formats. if f.VCodec != "" && f.VCodec != "none" { fi.Codec = f.VCodec } else if f.ACodec != "" && f.ACodec != "none" { fi.Codec = f.ACodec } if f.Filesize != nil && *f.Filesize > 0 { fi.FileSize = humanizeBytes(*f.Filesize) } else if f.FilesizeApprox != nil && *f.FilesizeApprox > 0 { fi.FileSize = "~" + humanizeBytes(*f.FilesizeApprox) } return fi } // humanizeBytes renders a byte count as a compact human-readable size. func humanizeBytes(n int64) string { const unit = 1024 if n < unit { return fmt.Sprintf("%dB", n) } div, exp := int64(unit), 0 for m := n / unit; m >= unit; m /= unit { div *= unit exp++ } return fmt.Sprintf("%.1f%ciB", float64(n)/float64(div), "KMGTPE"[exp]) } func sqlNullInt64(v int64) sql.NullInt64 { return sql.NullInt64{Int64: v, Valid: true} } // reservedFlags are yt-dlp options VidArchive always sets itself; user custom // flags must not pass them (or a conflicting inverse). The value describes what // the option controls, for the failure message. var reservedFlags = map[string]string{ "-o": "the output template", "--output": "the output template", "-P": "the download path", "--paths": "the download path", "--cookies": "cookies (set these in Settings instead)", "--no-cookies": "cookies (set these in Settings instead)", "--newline": "progress output formatting (VidArchive sets this to stream logs)", } // reservedSubscriptionFlags are additionally reserved for subscription runs, // where VidArchive drives info-json writing and the refresh mode. var reservedSubscriptionFlags = map[string]string{ "--write-info-json": "info-json writing (needed to track item identity)", "--no-write-info-json": "info-json writing (needed to track item identity)", "--download-archive": "the download archive (managed by Skip mode)", "--no-download-archive": "the download archive (managed by Skip mode)", "--skip-download": "media downloading (managed by Metadata mode)", "--no-skip-download": "media downloading (managed by Metadata mode)", } // checkReservedFlags rejects custom flags that clash with options VidArchive // controls, naming the offender. It matches both "--flag" and "--flag=value". func checkReservedFlags(customFlags string, isSubscription bool) error { for _, tok := range strings.Fields(customFlags) { name := tok if i := strings.IndexByte(name, '='); i >= 0 { name = name[:i] } if desc, ok := reservedFlags[name]; ok { return fmt.Errorf("custom flag %q conflicts with VidArchive's handling of %s; remove it and try again", tok, desc) } if isSubscription { if desc, ok := reservedSubscriptionFlags[name]; ok { return fmt.Errorf("custom flag %q conflicts with VidArchive's handling of %s; remove it and try again", tok, desc) } } } return nil }