rssd/state/state.go
Greg Pomerantz bc26e73a90 Add YouTube feed support via yt-dlp, MIME-based extensions, robust state
- ytdlp: thin wrapper around yt-dlp CLI for channel/playlist discovery
  (--flat-playlist) and per-video metadata (--print-json); defaults to
  player_client=web_embedded to avoid SABR-only format restrictions.
- downloader: YouTube path extracts audio via yt-dlp -x, probes duration
  with ffprobe and rejects clips far shorter than the expected length.
- main: YouTube channel/playlist feeds, incremental discovery with
  pre-start-date boundary for newest-first channels, two-phase
  record-then-enqueue so discovered jobs survive crashes.
- poller: propagate enclosure MIME type; computeDestPath appends the
  matching file extension.
- state: jobs carry expected duration; AddJobWithStatus for skipped items.
- cmd/probe: dry-run discovery/date-resolution diagnostics.
2026-09-23 09:06:36 -04:00

290 lines
7.5 KiB
Go

package state
import (
"encoding/json"
"fmt"
"os"
"path/filepath"
"sync"
"time"
)
// State is the full persisted state of the daemon.
type State struct {
Feeds []FeedState `json:"feeds"`
Jobs []Job `json:"jobs"`
}
// FeedState holds per-feed metadata from HTTP responses.
type FeedState struct {
URL string `json:"url"`
LastModified string `json:"last_modified,omitempty"`
ETag string `json:"etag,omitempty"`
}
// Job represents a download job with retry state.
type Job struct {
FeedURL string `json:"feed_url"`
EnclosureURL string `json:"enclosure_url"`
DestPath string `json:"dest_path"`
Status string `json:"status"`
AttemptCount int `json:"attempt_count"`
MaxRetries int `json:"max_retries"`
NextAttemptAt *time.Time `json:"next_attempt_at,omitempty"` // nil means ready now
Error string `json:"error,omitempty"`
Duration float64 `json:"duration,omitempty"` // expected media duration in seconds (YouTube)
}
// JobStatus values.
const (
StatusPending = "pending"
StatusDownloading = "downloading"
StatusRetrying = "retrying"
StatusDownloaded = "downloaded"
StatusFailed = "failed"
StatusSkipped = "skipped" // resolved but excluded by start_date; never downloaded
)
// Manager provides thread-safe access to persisted state.
type Manager struct {
mu sync.Mutex
path string
state *State
}
// NewManager creates a new state manager pointing at the given file path.
func NewManager(path string) (*Manager, error) {
dir := filepath.Dir(path)
if err := os.MkdirAll(dir, 0755); err != nil {
return nil, fmt.Errorf("create state directory: %w", err)
}
return &Manager{path: path, state: &State{}}, nil
}
// Load reads the state file from disk. If it doesn't exist or is empty, returns an empty state.
func (m *Manager) Load() error {
m.mu.Lock()
defer m.mu.Unlock()
data, err := os.ReadFile(m.path)
if err != nil {
if os.IsNotExist(err) {
m.state = &State{}
return nil
}
return fmt.Errorf("read state: %w", err)
}
if len(data) == 0 {
m.state = &State{}
return nil
}
var s State
if err := json.Unmarshal(data, &s); err != nil {
return fmt.Errorf("parse state: %w", err)
}
m.state = &s
return nil
}
// Save atomically writes the current state to disk with pretty-printed JSON.
func (m *Manager) Save() error {
m.mu.Lock()
defer m.mu.Unlock()
data, err := json.MarshalIndent(m.state, "", " ")
if err != nil {
return fmt.Errorf("marshal state: %w", err)
}
dir := filepath.Dir(m.path)
tmpFile, err := os.CreateTemp(dir, ".rssd-state-*.tmp")
if err != nil {
return fmt.Errorf("create temp file: %w", err)
}
tmpName := tmpFile.Name()
if _, err := tmpFile.Write(data); err != nil {
tmpFile.Close()
os.Remove(tmpName)
return fmt.Errorf("write temp state: %w", err)
}
if err := tmpFile.Close(); err != nil {
os.Remove(tmpName)
return fmt.Errorf("close temp state: %w", err)
}
if err := os.Rename(tmpName, m.path); err != nil {
os.Remove(tmpName)
return fmt.Errorf("rename state file: %w", err)
}
return nil
}
// JobSnapshot is a copy of a Job suitable for read-only inspection.
type JobSnapshot struct {
FeedURL string
EnclosureURL string
DestPath string
Status string
AttemptCount int
MaxRetries int
NextAttemptAt *time.Time // nil means ready now
Error string
Duration float64
}
// ForEachJob iterates over all jobs and calls fn for each one.
// The callback receives a pointer to a copy of the job (not the live state),
// so it must not modify fields. Use UpdateJob or AddJob for mutations.
// fn may perform I/O without risk of deadlock — the lock is released before
// any callback invocation.
func (m *Manager) ForEachJob(fn func(*JobSnapshot)) {
m.mu.Lock()
jobsCopy := make([]JobSnapshot, len(m.state.Jobs))
for i := range m.state.Jobs {
j := &m.state.Jobs[i]
jobsCopy[i] = JobSnapshot{
FeedURL: j.FeedURL,
EnclosureURL: j.EnclosureURL,
DestPath: j.DestPath,
Status: j.Status,
AttemptCount: j.AttemptCount,
MaxRetries: j.MaxRetries,
NextAttemptAt: j.NextAttemptAt,
Error: j.Error,
Duration: j.Duration,
}
}
m.mu.Unlock()
for i := range jobsCopy {
fn(&jobsCopy[i])
}
}
// State returns a copy of the current in-memory state for iteration.
func (m *Manager) State() *State {
m.mu.Lock()
defer m.mu.Unlock()
return m.state
}
// AddJob adds a new download job if one with this enclosure URL doesn't already exist.
// duration is the expected media duration in seconds (0 = unknown, used for
// YouTube duration sanity checks). Returns true if the job was added.
func (m *Manager) AddJob(feedURL, enclosureURL, destPath string, maxRetries int, duration float64) bool {
return m.addJob(feedURL, enclosureURL, destPath, StatusPending, maxRetries, duration)
}
// AddJobWithStatus adds a new job with an explicit initial status (used to
// record videos resolved but excluded by the start_date filter, so they are
// not re-resolved on every poll). Returns true if the job was added.
func (m *Manager) AddJobWithStatus(feedURL, enclosureURL, destPath, status string, duration float64) bool {
return m.addJob(feedURL, enclosureURL, destPath, status, 0, duration)
}
func (m *Manager) addJob(feedURL, enclosureURL, destPath, status string, maxRetries int, duration float64) bool {
m.mu.Lock()
defer m.mu.Unlock()
// Check if this enclosure URL already has a job.
for _, j := range m.state.Jobs {
if j.EnclosureURL == enclosureURL {
return false // already exists
}
}
job := Job{
FeedURL: feedURL,
EnclosureURL: enclosureURL,
DestPath: destPath,
Status: status,
AttemptCount: 0,
MaxRetries: maxRetries,
NextAttemptAt: nil,
Error: "",
Duration: duration,
}
m.state.Jobs = append(m.state.Jobs, job)
return true
}
// UpdateJob finds a job by enclosure URL and applies the update function.
// Returns true if the job was found and updated.
func (m *Manager) UpdateJob(enclosureURL string, fn func(*Job)) bool {
m.mu.Lock()
defer m.mu.Unlock()
for i := range m.state.Jobs {
if m.state.Jobs[i].EnclosureURL == enclosureURL {
fn(&m.state.Jobs[i])
return true
}
}
return false
}
// FindJobByURL returns a pointer to the job for the given enclosure URL, or nil.
func (m *Manager) FindJobByURL(enclosureURL string) *Job {
m.mu.Lock()
defer m.mu.Unlock()
for i := range m.state.Jobs {
if m.state.Jobs[i].EnclosureURL == enclosureURL {
return &m.state.Jobs[i]
}
}
return nil
}
// HasJob returns true if a job with the given enclosure URL exists.
func (m *Manager) HasJob(enclosureURL string) bool {
m.mu.Lock()
defer m.mu.Unlock()
for _, j := range m.state.Jobs {
if j.EnclosureURL == enclosureURL {
return true
}
}
return false
}
// FindFeedByURL returns the FeedState for a given feed URL, or nil.
func (m *Manager) FindFeedByURL(feedURL string) *FeedState {
m.mu.Lock()
defer m.mu.Unlock()
for i := range m.state.Feeds {
if m.state.Feeds[i].URL == feedURL {
return &m.state.Feeds[i]
}
}
return nil
}
// UpsertFeed adds or updates a feed's metadata (ETag, Last-Modified).
func (m *Manager) UpsertFeed(feedURL, lastModified, etag string) {
m.mu.Lock()
defer m.mu.Unlock()
for i := range m.state.Feeds {
if m.state.Feeds[i].URL == feedURL {
m.state.Feeds[i].LastModified = lastModified
m.state.Feeds[i].ETag = etag
return
}
}
m.state.Feeds = append(m.state.Feeds, FeedState{
URL: feedURL,
LastModified: lastModified,
ETag: etag,
})
}