feat: data lifecycle rewrite, download clients, wanted list, and central catalog index
Ships the fresh-start schema cleanup: rebuilt explore catalog index pipeline (dump import, artifact fetch/build, incremental listen-count refresh), a new download subsystem (Lidarr/Prowlarr/qBittorrent/SABnzbd/ slskd/yt-dlp providers, staging, reconciliation, wanted list), and the supporting schema/query/store changes across backend and frontend. Also includes two smaller follow-ups: bump the central index's rebuild-after cadence from 90 to 180 days, and remove the Explore "library only" online/offline toggle entirely (frontend-only, no backend counterpart) rather than carry unused UI/state. Co-Authored-By: Claude Sonnet 5 <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_01Y2Agd9af5hE7qzti2ackiS
This commit is contained in:
@@ -0,0 +1,214 @@
|
||||
// Package maintenance runs the janitorial work that keeps persisted
|
||||
// data from accumulating without bound.
|
||||
//
|
||||
// It exists because cleanup used to have no owner. Functions that
|
||||
// deleted expired rows were written and then never called; files written
|
||||
// by one package had no counterpart anywhere that removed them. A
|
||||
// registry makes the set of janitorial jobs a single visible list, so a
|
||||
// new cache that forgets to register is obvious in review rather than
|
||||
// discovered years later as unbounded growth.
|
||||
//
|
||||
// Policies come from the classification in backend/datamap:
|
||||
//
|
||||
// - Derived data is swept against a live set computed from the data it
|
||||
// was derived from. Anything not in the live set is garbage.
|
||||
// - Cache data is evicted by age, because it has no owner to be
|
||||
// compared against and is merely expensive — not impossible — to
|
||||
// re-fetch.
|
||||
//
|
||||
// Sweeps are idempotent and safe to interrupt: each deletes only what it
|
||||
// has positively identified as unreferenced, so a partial run simply
|
||||
// leaves work for the next one.
|
||||
package maintenance
|
||||
|
||||
import (
|
||||
"context"
|
||||
"log/slog"
|
||||
"sync"
|
||||
"time"
|
||||
)
|
||||
|
||||
// Result reports what a single job reclaimed.
|
||||
type Result struct {
|
||||
// RowsDeleted counts database rows removed.
|
||||
RowsDeleted int64
|
||||
// FilesDeleted counts files removed from disk.
|
||||
FilesDeleted int64
|
||||
// BytesFreed is the size of those files.
|
||||
BytesFreed int64
|
||||
}
|
||||
|
||||
// empty reports whether the job found nothing to do, so quiet runs can
|
||||
// be logged at a lower level.
|
||||
func (r Result) empty() bool {
|
||||
return r.RowsDeleted == 0 && r.FilesDeleted == 0
|
||||
}
|
||||
|
||||
// Job is one unit of janitorial work.
|
||||
type Job struct {
|
||||
// Name identifies the job in logs and in the run record.
|
||||
Name string
|
||||
// MinInterval is the minimum time between runs. A job is skipped if
|
||||
// it ran more recently than this, so hooking the runner to a
|
||||
// frequently-firing trigger stays cheap.
|
||||
MinInterval time.Duration
|
||||
// Run performs the work. It must be idempotent and must respect
|
||||
// context cancellation.
|
||||
Run func(ctx context.Context) (Result, error)
|
||||
}
|
||||
|
||||
// Runner holds the registered jobs and enforces their intervals.
|
||||
type Runner struct {
|
||||
mu sync.Mutex
|
||||
jobs []Job
|
||||
lastRun map[string]time.Time
|
||||
logger *slog.Logger
|
||||
}
|
||||
|
||||
// NewRunner returns an empty runner.
|
||||
func NewRunner(logger *slog.Logger) *Runner {
|
||||
return &Runner{
|
||||
lastRun: make(map[string]time.Time),
|
||||
logger: logger,
|
||||
}
|
||||
}
|
||||
|
||||
// Register adds a job. Registering a name twice replaces the earlier
|
||||
// job, so wiring code can be re-run without accumulating duplicates.
|
||||
func (r *Runner) Register(job Job) {
|
||||
r.mu.Lock()
|
||||
defer r.mu.Unlock()
|
||||
|
||||
for i, existing := range r.jobs {
|
||||
if existing.Name == job.Name {
|
||||
r.jobs[i] = job
|
||||
|
||||
return
|
||||
}
|
||||
}
|
||||
|
||||
r.jobs = append(r.jobs, job)
|
||||
}
|
||||
|
||||
// JobNames returns the registered job names, for tests and diagnostics.
|
||||
func (r *Runner) JobNames() []string {
|
||||
r.mu.Lock()
|
||||
defer r.mu.Unlock()
|
||||
|
||||
names := make([]string, 0, len(r.jobs))
|
||||
for _, j := range r.jobs {
|
||||
names = append(names, j.Name)
|
||||
}
|
||||
|
||||
return names
|
||||
}
|
||||
|
||||
// due reports whether a job's interval has elapsed.
|
||||
func (r *Runner) due(job Job, now time.Time) bool {
|
||||
r.mu.Lock()
|
||||
defer r.mu.Unlock()
|
||||
|
||||
last, ran := r.lastRun[job.Name]
|
||||
if !ran {
|
||||
return true
|
||||
}
|
||||
|
||||
return now.Sub(last) >= job.MinInterval
|
||||
}
|
||||
|
||||
func (r *Runner) markRun(name string, at time.Time) {
|
||||
r.mu.Lock()
|
||||
defer r.mu.Unlock()
|
||||
|
||||
r.lastRun[name] = at
|
||||
}
|
||||
|
||||
// snapshot copies the job list so a run does not hold the lock while
|
||||
// executing jobs.
|
||||
func (r *Runner) snapshot() []Job {
|
||||
r.mu.Lock()
|
||||
defer r.mu.Unlock()
|
||||
|
||||
out := make([]Job, len(r.jobs))
|
||||
copy(out, r.jobs)
|
||||
|
||||
return out
|
||||
}
|
||||
|
||||
// RunDue runs every job whose interval has elapsed. A job that fails is
|
||||
// logged and does not prevent the others from running; janitorial work
|
||||
// is best-effort by nature and the next run will retry.
|
||||
func (r *Runner) RunDue(ctx context.Context) Result {
|
||||
var total Result
|
||||
|
||||
for _, job := range r.snapshot() {
|
||||
if ctx.Err() != nil {
|
||||
r.logger.Info("maintenance cancelled", "after", job.Name)
|
||||
|
||||
break
|
||||
}
|
||||
|
||||
now := time.Now()
|
||||
if !r.due(job, now) {
|
||||
continue
|
||||
}
|
||||
|
||||
start := time.Now()
|
||||
|
||||
result, err := job.Run(ctx)
|
||||
|
||||
r.markRun(job.Name, now)
|
||||
|
||||
if err != nil {
|
||||
r.logger.Warn("maintenance job failed",
|
||||
"job", job.Name, "err", err,
|
||||
"duration", time.Since(start),
|
||||
)
|
||||
|
||||
continue
|
||||
}
|
||||
|
||||
total.RowsDeleted += result.RowsDeleted
|
||||
total.FilesDeleted += result.FilesDeleted
|
||||
total.BytesFreed += result.BytesFreed
|
||||
|
||||
if result.empty() {
|
||||
r.logger.Debug("maintenance job found nothing",
|
||||
"job", job.Name, "duration", time.Since(start),
|
||||
)
|
||||
|
||||
continue
|
||||
}
|
||||
|
||||
r.logger.Info("maintenance job reclaimed",
|
||||
"job", job.Name,
|
||||
"rows", result.RowsDeleted,
|
||||
"files", result.FilesDeleted,
|
||||
"bytes", result.BytesFreed,
|
||||
"duration", time.Since(start),
|
||||
)
|
||||
}
|
||||
|
||||
return total
|
||||
}
|
||||
|
||||
// Start runs the due jobs immediately and then on every tick until the
|
||||
// context is cancelled. It returns straight away; the loop runs in its
|
||||
// own goroutine.
|
||||
func (r *Runner) Start(ctx context.Context, tick time.Duration) {
|
||||
go func() {
|
||||
r.RunDue(ctx)
|
||||
|
||||
ticker := time.NewTicker(tick)
|
||||
defer ticker.Stop()
|
||||
|
||||
for {
|
||||
select {
|
||||
case <-ctx.Done():
|
||||
return
|
||||
case <-ticker.C:
|
||||
r.RunDue(ctx)
|
||||
}
|
||||
}
|
||||
}()
|
||||
}
|
||||
@@ -0,0 +1,434 @@
|
||||
package maintenance
|
||||
|
||||
import (
|
||||
"context"
|
||||
"errors"
|
||||
"log/slog"
|
||||
"os"
|
||||
"path/filepath"
|
||||
"slices"
|
||||
"testing"
|
||||
"time"
|
||||
|
||||
"yellowjacket/backend/database"
|
||||
)
|
||||
|
||||
// errTestJobFailed stands in for a job returning an error.
|
||||
var errTestJobFailed = errors.New("job failed")
|
||||
|
||||
func testRunner() *Runner {
|
||||
return NewRunner(slog.Default())
|
||||
}
|
||||
|
||||
func TestRunnerRunsRegisteredJobs(t *testing.T) {
|
||||
t.Parallel()
|
||||
|
||||
r := testRunner()
|
||||
|
||||
var ran int
|
||||
|
||||
r.Register(Job{
|
||||
Name: "counter",
|
||||
Run: func(_ context.Context) (Result, error) {
|
||||
ran++
|
||||
|
||||
return Result{RowsDeleted: 3}, nil
|
||||
},
|
||||
})
|
||||
|
||||
total := r.RunDue(context.Background())
|
||||
|
||||
if ran != 1 {
|
||||
t.Errorf("job ran %d times, want 1", ran)
|
||||
}
|
||||
|
||||
if total.RowsDeleted != 3 {
|
||||
t.Errorf("RowsDeleted = %d, want 3", total.RowsDeleted)
|
||||
}
|
||||
}
|
||||
|
||||
// A job must not run again before its interval has elapsed, so hooking
|
||||
// the runner to a frequent trigger stays cheap.
|
||||
func TestRunnerRespectsMinInterval(t *testing.T) {
|
||||
t.Parallel()
|
||||
|
||||
r := testRunner()
|
||||
|
||||
var ran int
|
||||
|
||||
r.Register(Job{
|
||||
Name: "throttled",
|
||||
MinInterval: time.Hour,
|
||||
Run: func(_ context.Context) (Result, error) {
|
||||
ran++
|
||||
|
||||
return Result{}, nil
|
||||
},
|
||||
})
|
||||
|
||||
r.RunDue(context.Background())
|
||||
r.RunDue(context.Background())
|
||||
r.RunDue(context.Background())
|
||||
|
||||
if ran != 1 {
|
||||
t.Errorf("job ran %d times despite 1h interval, want 1", ran)
|
||||
}
|
||||
}
|
||||
|
||||
// One failing job must not prevent the others from running — janitorial
|
||||
// work is best-effort and the next run retries.
|
||||
func TestRunnerContinuesAfterFailure(t *testing.T) {
|
||||
t.Parallel()
|
||||
|
||||
r := testRunner()
|
||||
|
||||
var secondRan bool
|
||||
|
||||
r.Register(Job{
|
||||
Name: "failing",
|
||||
Run: func(_ context.Context) (Result, error) {
|
||||
return Result{}, errTestJobFailed
|
||||
},
|
||||
})
|
||||
r.Register(Job{
|
||||
Name: "healthy",
|
||||
Run: func(_ context.Context) (Result, error) {
|
||||
secondRan = true
|
||||
|
||||
return Result{}, nil
|
||||
},
|
||||
})
|
||||
|
||||
r.RunDue(context.Background())
|
||||
|
||||
if !secondRan {
|
||||
t.Error("second job did not run after the first failed")
|
||||
}
|
||||
}
|
||||
|
||||
// Registering the same name twice replaces the job rather than
|
||||
// accumulating duplicates, so wiring code is safe to re-run.
|
||||
func TestRunnerRegisterReplaces(t *testing.T) {
|
||||
t.Parallel()
|
||||
|
||||
r := testRunner()
|
||||
|
||||
noop := func(_ context.Context) (Result, error) { return Result{}, nil }
|
||||
|
||||
r.Register(Job{Name: "dup", Run: noop})
|
||||
r.Register(Job{Name: "dup", Run: noop})
|
||||
|
||||
if names := r.JobNames(); len(names) != 1 {
|
||||
t.Errorf("JobNames() = %v, want one entry", names)
|
||||
}
|
||||
}
|
||||
|
||||
func TestRunnerStopsOnCancelledContext(t *testing.T) {
|
||||
t.Parallel()
|
||||
|
||||
r := testRunner()
|
||||
|
||||
ctx, cancel := context.WithCancel(context.Background())
|
||||
cancel()
|
||||
|
||||
var ran bool
|
||||
|
||||
r.Register(Job{
|
||||
Name: "should-not-run",
|
||||
Run: func(_ context.Context) (Result, error) {
|
||||
ran = true
|
||||
|
||||
return Result{}, nil
|
||||
},
|
||||
})
|
||||
|
||||
r.RunDue(ctx)
|
||||
|
||||
if ran {
|
||||
t.Error("job ran despite cancelled context")
|
||||
}
|
||||
}
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// Sweeps
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
func TestExpiredHTTPCacheJob(t *testing.T) {
|
||||
t.Parallel()
|
||||
|
||||
db := database.NewTestDB(t)
|
||||
|
||||
if _, err := db.ExecContext(
|
||||
`INSERT INTO http_cache (url_key, response, expires_at)
|
||||
VALUES ('stale', '{}', datetime('now', '-1 day')),
|
||||
('fresh', '{}', datetime('now', '+1 day'))`,
|
||||
); err != nil {
|
||||
t.Fatalf("seed http_cache: %v", err)
|
||||
}
|
||||
|
||||
result, err := ExpiredHTTPCacheJob(db).Run(context.Background())
|
||||
if err != nil {
|
||||
t.Fatalf("run job: %v", err)
|
||||
}
|
||||
|
||||
if result.RowsDeleted != 1 {
|
||||
t.Errorf("RowsDeleted = %d, want 1", result.RowsDeleted)
|
||||
}
|
||||
|
||||
var remaining string
|
||||
|
||||
rows, err := db.QueryContext("SELECT url_key FROM http_cache")
|
||||
if err != nil {
|
||||
t.Fatalf("query http_cache: %v", err)
|
||||
}
|
||||
|
||||
defer func() { _ = rows.Close() }()
|
||||
|
||||
if rows.Next() {
|
||||
_ = rows.Scan(&remaining)
|
||||
}
|
||||
|
||||
if remaining != "fresh" {
|
||||
t.Errorf("remaining row = %q, want the unexpired one", remaining)
|
||||
}
|
||||
}
|
||||
|
||||
// The covers sweep must delete files no cover_art row references while
|
||||
// keeping the referenced original and every derived variant.
|
||||
func TestOrphanedCoverFilesJob(t *testing.T) {
|
||||
t.Parallel()
|
||||
|
||||
db := database.NewTestDB(t)
|
||||
dir := t.TempDir()
|
||||
|
||||
// Stand-in for library.CoverArtFileSet.
|
||||
expand := func(original string) []string {
|
||||
base := filepath.Base(original)
|
||||
base = base[:len(base)-len(filepath.Ext(base))]
|
||||
|
||||
return []string{
|
||||
original,
|
||||
filepath.Join(filepath.Dir(original), base+"_sm.jpg"),
|
||||
filepath.Join(filepath.Dir(original), base+"_md.jpg"),
|
||||
}
|
||||
}
|
||||
|
||||
keep := []string{"live.jpg", "live_sm.jpg", "live_md.jpg"}
|
||||
drop := []string{"orphan.jpg", "orphan_sm.jpg", "stray_md.jpg"}
|
||||
|
||||
for _, name := range slices.Concat(keep, drop) {
|
||||
if err := os.WriteFile(
|
||||
filepath.Join(dir, name), []byte("img"), 0o600,
|
||||
); err != nil {
|
||||
t.Fatalf("write %s: %v", name, err)
|
||||
}
|
||||
}
|
||||
|
||||
if _, err := db.ExecContext(
|
||||
`INSERT INTO cover_art (is_embedded, file_path, mime_type)
|
||||
VALUES (0, ?, 'image/jpeg')`,
|
||||
filepath.Join(dir, "live.jpg"),
|
||||
); err != nil {
|
||||
t.Fatalf("seed cover_art: %v", err)
|
||||
}
|
||||
|
||||
result, err := OrphanedCoverFilesJob(db, dir, expand).
|
||||
Run(context.Background())
|
||||
if err != nil {
|
||||
t.Fatalf("run job: %v", err)
|
||||
}
|
||||
|
||||
if result.FilesDeleted != int64(len(drop)) {
|
||||
t.Errorf("FilesDeleted = %d, want %d", result.FilesDeleted, len(drop))
|
||||
}
|
||||
|
||||
for _, name := range keep {
|
||||
if _, err := os.Stat(filepath.Join(dir, name)); err != nil {
|
||||
t.Errorf("referenced file %s was deleted", name)
|
||||
}
|
||||
}
|
||||
|
||||
for _, name := range drop {
|
||||
if _, err := os.Stat(filepath.Join(dir, name)); !os.IsNotExist(err) {
|
||||
t.Errorf("orphan %s survived the sweep", name)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// An empty live set means the query saw nothing, not that every cover is
|
||||
// garbage. The sweep must refuse to empty the directory in that case.
|
||||
func TestOrphanedCoverFilesJob_EmptyLiveSetIsNoOp(t *testing.T) {
|
||||
t.Parallel()
|
||||
|
||||
db := database.NewTestDB(t)
|
||||
dir := t.TempDir()
|
||||
|
||||
path := filepath.Join(dir, "something.jpg")
|
||||
if err := os.WriteFile(path, []byte("img"), 0o600); err != nil {
|
||||
t.Fatalf("write file: %v", err)
|
||||
}
|
||||
|
||||
expand := func(p string) []string { return []string{p} }
|
||||
|
||||
result, err := OrphanedCoverFilesJob(db, dir, expand).
|
||||
Run(context.Background())
|
||||
if err != nil {
|
||||
t.Fatalf("run job: %v", err)
|
||||
}
|
||||
|
||||
if result.FilesDeleted != 0 {
|
||||
t.Errorf("FilesDeleted = %d, want 0", result.FilesDeleted)
|
||||
}
|
||||
|
||||
if _, err := os.Stat(path); err != nil {
|
||||
t.Error("sweep emptied the directory on an empty live set")
|
||||
}
|
||||
}
|
||||
|
||||
// Artwork for an artist in the library is kept regardless of age;
|
||||
// artwork for a browsed artist ages out.
|
||||
func TestOrphanedArtistImagesJob(t *testing.T) {
|
||||
t.Parallel()
|
||||
|
||||
db := database.NewTestDB(t)
|
||||
dir := t.TempDir()
|
||||
|
||||
const (
|
||||
ownedMBID = "11111111-1111-1111-1111-111111111111"
|
||||
browsedMBID = "22222222-2222-2222-2222-222222222222"
|
||||
recentMBID = "33333333-3333-3333-3333-333333333333"
|
||||
)
|
||||
|
||||
for _, mbid := range []string{ownedMBID, browsedMBID, recentMBID} {
|
||||
artistDir := filepath.Join(dir, mbid)
|
||||
if err := os.MkdirAll(artistDir, 0o755); err != nil {
|
||||
t.Fatalf("mkdir %s: %v", mbid, err)
|
||||
}
|
||||
|
||||
if err := os.WriteFile(
|
||||
filepath.Join(artistDir, "primary.jpg"), []byte("img"), 0o600,
|
||||
); err != nil {
|
||||
t.Fatalf("write image: %v", err)
|
||||
}
|
||||
}
|
||||
|
||||
// The owned artist is in the library.
|
||||
if _, err := db.ExecContext(
|
||||
`INSERT INTO artists (name, mbid) VALUES ('Owned', ?)`, ownedMBID,
|
||||
); err != nil {
|
||||
t.Fatalf("seed artists: %v", err)
|
||||
}
|
||||
|
||||
old := time.Now().Add(-200 * 24 * time.Hour)
|
||||
|
||||
for _, tc := range []struct {
|
||||
mbid string
|
||||
created time.Time
|
||||
}{
|
||||
{ownedMBID, old},
|
||||
{browsedMBID, old},
|
||||
{recentMBID, time.Now()},
|
||||
} {
|
||||
if _, err := db.ExecContext(
|
||||
`INSERT INTO artist_images
|
||||
(artist_mbid, source, source_url, file_path, created_at)
|
||||
VALUES (?, 'test', 'http://x', ?, ?)`,
|
||||
tc.mbid,
|
||||
filepath.Join(dir, tc.mbid, "primary.jpg"),
|
||||
tc.created,
|
||||
); err != nil {
|
||||
t.Fatalf("seed artist_images for %s: %v", tc.mbid, err)
|
||||
}
|
||||
}
|
||||
|
||||
if _, err := OrphanedArtistImagesJob(db, dir).
|
||||
Run(context.Background()); err != nil {
|
||||
t.Fatalf("run job: %v", err)
|
||||
}
|
||||
|
||||
if _, err := os.Stat(filepath.Join(dir, ownedMBID)); err != nil {
|
||||
t.Error("artwork for a library artist was evicted")
|
||||
}
|
||||
|
||||
if _, err := os.Stat(filepath.Join(dir, recentMBID)); err != nil {
|
||||
t.Error("recently fetched artwork was evicted")
|
||||
}
|
||||
|
||||
if _, err := os.Stat(filepath.Join(dir, browsedMBID)); !os.IsNotExist(err) {
|
||||
t.Error("stale browsed artwork survived the sweep")
|
||||
}
|
||||
|
||||
// The rows must go with the files.
|
||||
rows, err := db.QueryContext(
|
||||
"SELECT COUNT(*) FROM artist_images WHERE artist_mbid = ?",
|
||||
browsedMBID,
|
||||
)
|
||||
if err != nil {
|
||||
t.Fatalf("count rows: %v", err)
|
||||
}
|
||||
|
||||
defer func() { _ = rows.Close() }()
|
||||
|
||||
var n int
|
||||
|
||||
if rows.Next() {
|
||||
_ = rows.Scan(&n)
|
||||
}
|
||||
|
||||
if n != 0 {
|
||||
t.Errorf("artist_images rows for evicted artist = %d, want 0", n)
|
||||
}
|
||||
}
|
||||
|
||||
func TestExpiredProxyCacheJob(t *testing.T) {
|
||||
t.Parallel()
|
||||
|
||||
dir := t.TempDir()
|
||||
|
||||
stale := filepath.Join(dir, "stale.jpg")
|
||||
fresh := filepath.Join(dir, "fresh.jpg")
|
||||
|
||||
for _, p := range []string{stale, fresh} {
|
||||
if err := os.WriteFile(p, []byte("img"), 0o600); err != nil {
|
||||
t.Fatalf("write %s: %v", p, err)
|
||||
}
|
||||
}
|
||||
|
||||
old := time.Now().Add(-60 * 24 * time.Hour)
|
||||
if err := os.Chtimes(stale, old, old); err != nil {
|
||||
t.Fatalf("chtimes: %v", err)
|
||||
}
|
||||
|
||||
result, err := ExpiredProxyCacheJob(dir).Run(context.Background())
|
||||
if err != nil {
|
||||
t.Fatalf("run job: %v", err)
|
||||
}
|
||||
|
||||
if result.FilesDeleted != 1 {
|
||||
t.Errorf("FilesDeleted = %d, want 1", result.FilesDeleted)
|
||||
}
|
||||
|
||||
if _, err := os.Stat(fresh); err != nil {
|
||||
t.Error("recently written thumbnail was evicted")
|
||||
}
|
||||
|
||||
if _, err := os.Stat(stale); !os.IsNotExist(err) {
|
||||
t.Error("stale thumbnail survived the sweep")
|
||||
}
|
||||
}
|
||||
|
||||
// A missing directory is normal on a fresh install and must not error.
|
||||
func TestSweepMissingDirectory(t *testing.T) {
|
||||
t.Parallel()
|
||||
|
||||
result, err := ExpiredProxyCacheJob(
|
||||
filepath.Join(t.TempDir(), "does-not-exist"),
|
||||
).Run(context.Background())
|
||||
if err != nil {
|
||||
t.Fatalf("missing directory returned an error: %v", err)
|
||||
}
|
||||
|
||||
if result.FilesDeleted != 0 {
|
||||
t.Errorf("FilesDeleted = %d, want 0", result.FilesDeleted)
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,356 @@
|
||||
package maintenance
|
||||
|
||||
import (
|
||||
"context"
|
||||
"fmt"
|
||||
"os"
|
||||
"path/filepath"
|
||||
"time"
|
||||
|
||||
"yellowjacket/backend/database"
|
||||
)
|
||||
|
||||
// Retention windows for cache data. Cached artwork for artists the user
|
||||
// actually owns is kept indefinitely; art fetched while browsing Explore
|
||||
// is transient and ages out.
|
||||
const (
|
||||
// browsedArtRetention is how long artwork for a non-library artist
|
||||
// survives after it was fetched.
|
||||
browsedArtRetention = 90 * 24 * time.Hour
|
||||
|
||||
// proxyCacheRetention is how long an Explore cover-art thumbnail
|
||||
// survives after it was last written.
|
||||
proxyCacheRetention = 30 * 24 * time.Hour
|
||||
)
|
||||
|
||||
// Default intervals. These are minimums, not schedules — the runner
|
||||
// skips a job that ran more recently.
|
||||
const (
|
||||
frequentInterval = 6 * time.Hour
|
||||
dailyInterval = 24 * time.Hour
|
||||
)
|
||||
|
||||
// ExpiredHTTPCacheJob deletes HTTP cache rows past their TTL.
|
||||
//
|
||||
// Reads already filter on expires_at, so expired rows are inert — but
|
||||
// nothing was deleting them, so the table grew without bound for the
|
||||
// life of the install.
|
||||
func ExpiredHTTPCacheJob(db *database.DB) Job {
|
||||
return Job{
|
||||
Name: "http-cache-evict",
|
||||
MinInterval: frequentInterval,
|
||||
Run: func(_ context.Context) (Result, error) {
|
||||
res, err := db.ExecContext(
|
||||
"DELETE FROM http_cache WHERE expires_at < datetime('now')",
|
||||
)
|
||||
if err != nil {
|
||||
return Result{}, fmt.Errorf(
|
||||
"delete expired http_cache rows: %w", err,
|
||||
)
|
||||
}
|
||||
|
||||
rows, _ := res.RowsAffected()
|
||||
|
||||
return Result{RowsDeleted: rows}, nil
|
||||
},
|
||||
}
|
||||
}
|
||||
|
||||
// OrphanedCoverFilesJob removes files from the covers directory that no
|
||||
// cover_art row references.
|
||||
//
|
||||
// Cover art is derived data, so the live set is authoritative: every
|
||||
// file that is not the original named by a cover_art row, or one of that
|
||||
// original's derived size variants, is garbage. This reclaims art left
|
||||
// behind by earlier versions that deleted only the original and left its
|
||||
// thumbnails.
|
||||
func OrphanedCoverFilesJob(
|
||||
db *database.DB,
|
||||
coversDir string,
|
||||
expandVariants func(originalPath string) []string,
|
||||
) Job {
|
||||
return Job{
|
||||
Name: "covers-sweep",
|
||||
MinInterval: dailyInterval,
|
||||
Run: func(ctx context.Context) (Result, error) {
|
||||
live, err := liveCoverFiles(db, coversDir, expandVariants)
|
||||
if err != nil {
|
||||
return Result{}, err
|
||||
}
|
||||
|
||||
// A covers directory with no live entries almost certainly
|
||||
// means the query failed to see the real table rather than
|
||||
// that every cover is garbage. Refuse to empty the
|
||||
// directory on that basis.
|
||||
if len(live) == 0 {
|
||||
return Result{}, nil
|
||||
}
|
||||
|
||||
return sweepDir(ctx, coversDir, func(name string) bool {
|
||||
return !live[name]
|
||||
})
|
||||
},
|
||||
}
|
||||
}
|
||||
|
||||
// liveCoverFiles returns the basenames of every file the covers
|
||||
// directory is supposed to contain.
|
||||
func liveCoverFiles(
|
||||
db *database.DB,
|
||||
coversDir string,
|
||||
expandVariants func(string) []string,
|
||||
) (map[string]bool, error) {
|
||||
rows, err := db.QueryContext("SELECT file_path FROM cover_art")
|
||||
if err != nil {
|
||||
return nil, fmt.Errorf("read cover_art paths: %w", err)
|
||||
}
|
||||
|
||||
defer func() { _ = rows.Close() }()
|
||||
|
||||
live := make(map[string]bool)
|
||||
|
||||
for rows.Next() {
|
||||
var path string
|
||||
|
||||
if err := rows.Scan(&path); err != nil {
|
||||
return nil, fmt.Errorf("scan cover_art path: %w", err)
|
||||
}
|
||||
|
||||
// Rows may store an absolute path from a previous install
|
||||
// location, so compare by basename within the covers directory.
|
||||
original := filepath.Join(coversDir, filepath.Base(path))
|
||||
|
||||
for _, variant := range expandVariants(original) {
|
||||
live[filepath.Base(variant)] = true
|
||||
}
|
||||
}
|
||||
|
||||
if err := rows.Err(); err != nil {
|
||||
return nil, fmt.Errorf("iterate cover_art paths: %w", err)
|
||||
}
|
||||
|
||||
return live, nil
|
||||
}
|
||||
|
||||
// OrphanedArtistImagesJob evicts cached artist artwork.
|
||||
//
|
||||
// Artist images are cache data with no owner to compare against — they
|
||||
// are fetched for any artist the user browses in Explore, most of whom
|
||||
// are not in the library. The policy is therefore twofold: artwork for
|
||||
// an artist the user owns is kept indefinitely, and everything else ages
|
||||
// out. Rows whose file has vanished are dropped so the table matches
|
||||
// what is actually on disk.
|
||||
func OrphanedArtistImagesJob(db *database.DB, artistImagesDir string) Job {
|
||||
return Job{
|
||||
Name: "artist-images-sweep",
|
||||
MinInterval: dailyInterval,
|
||||
Run: func(ctx context.Context) (Result, error) {
|
||||
var result Result
|
||||
|
||||
cutoff := time.Now().Add(-browsedArtRetention)
|
||||
|
||||
// Collect the directories to remove before deleting rows, so
|
||||
// a failure partway leaves rows pointing at real files
|
||||
// rather than the reverse.
|
||||
stale, err := staleArtistMBIDs(db, cutoff)
|
||||
if err != nil {
|
||||
return Result{}, err
|
||||
}
|
||||
|
||||
for _, mbid := range stale {
|
||||
if ctx.Err() != nil {
|
||||
return result, nil
|
||||
}
|
||||
|
||||
dir := filepath.Join(artistImagesDir, mbid)
|
||||
|
||||
freed, files := dirSize(dir)
|
||||
|
||||
if err := os.RemoveAll(dir); err != nil && !os.IsNotExist(err) {
|
||||
continue
|
||||
}
|
||||
|
||||
result.FilesDeleted += files
|
||||
result.BytesFreed += freed
|
||||
}
|
||||
|
||||
if len(stale) > 0 {
|
||||
rows, delErr := deleteArtistImageRows(db, stale)
|
||||
if delErr != nil {
|
||||
return result, delErr
|
||||
}
|
||||
|
||||
result.RowsDeleted += rows
|
||||
}
|
||||
|
||||
return result, nil
|
||||
},
|
||||
}
|
||||
}
|
||||
|
||||
// staleArtistMBIDs returns artist MBIDs whose cached artwork may be
|
||||
// evicted: fetched before the cutoff and not an artist in the library.
|
||||
func staleArtistMBIDs(
|
||||
db *database.DB,
|
||||
cutoff time.Time,
|
||||
) ([]string, error) {
|
||||
rows, err := db.QueryContext(
|
||||
`SELECT DISTINCT artist_mbid FROM artist_images
|
||||
WHERE created_at < ?
|
||||
AND artist_mbid NOT IN (
|
||||
SELECT mbid FROM artists
|
||||
WHERE mbid IS NOT NULL AND mbid != ''
|
||||
)`,
|
||||
cutoff,
|
||||
)
|
||||
if err != nil {
|
||||
return nil, fmt.Errorf("query stale artist images: %w", err)
|
||||
}
|
||||
|
||||
defer func() { _ = rows.Close() }()
|
||||
|
||||
var mbids []string
|
||||
|
||||
for rows.Next() {
|
||||
var mbid string
|
||||
|
||||
if err := rows.Scan(&mbid); err != nil {
|
||||
return nil, fmt.Errorf("scan artist mbid: %w", err)
|
||||
}
|
||||
|
||||
if mbid != "" {
|
||||
mbids = append(mbids, mbid)
|
||||
}
|
||||
}
|
||||
|
||||
if err := rows.Err(); err != nil {
|
||||
return nil, fmt.Errorf("iterate stale artist images: %w", err)
|
||||
}
|
||||
|
||||
return mbids, nil
|
||||
}
|
||||
|
||||
// deleteArtistImageRows removes the rows for the given artist MBIDs.
|
||||
func deleteArtistImageRows(
|
||||
db *database.DB,
|
||||
mbids []string,
|
||||
) (int64, error) {
|
||||
var total int64
|
||||
|
||||
for _, mbid := range mbids {
|
||||
res, err := db.ExecContext(
|
||||
"DELETE FROM artist_images WHERE artist_mbid = ?", mbid,
|
||||
)
|
||||
if err != nil {
|
||||
return total, fmt.Errorf(
|
||||
"delete artist_images rows for %s: %w", mbid, err,
|
||||
)
|
||||
}
|
||||
|
||||
n, _ := res.RowsAffected()
|
||||
total += n
|
||||
}
|
||||
|
||||
return total, nil
|
||||
}
|
||||
|
||||
// ExpiredProxyCacheJob evicts Explore cover-art thumbnails that have not
|
||||
// been rewritten within the retention window.
|
||||
//
|
||||
// This cache has no database table at all — it is keyed by release-group
|
||||
// MBID on the filesystem — so age is the only signal available.
|
||||
func ExpiredProxyCacheJob(proxyCacheDir string) Job {
|
||||
return Job{
|
||||
Name: "cover-art-proxy-sweep",
|
||||
MinInterval: dailyInterval,
|
||||
Run: func(ctx context.Context) (Result, error) {
|
||||
cutoff := time.Now().Add(-proxyCacheRetention)
|
||||
|
||||
return sweepDirFunc(ctx, proxyCacheDir,
|
||||
func(_ string, info os.FileInfo) bool {
|
||||
return info.ModTime().Before(cutoff)
|
||||
},
|
||||
)
|
||||
},
|
||||
}
|
||||
}
|
||||
|
||||
// sweepDir removes every file in dir for which shouldDelete reports true.
|
||||
func sweepDir(
|
||||
ctx context.Context,
|
||||
dir string,
|
||||
shouldDelete func(name string) bool,
|
||||
) (Result, error) {
|
||||
return sweepDirFunc(ctx, dir, func(name string, _ os.FileInfo) bool {
|
||||
return shouldDelete(name)
|
||||
})
|
||||
}
|
||||
|
||||
// sweepDirFunc removes files from a flat directory based on a predicate
|
||||
// over the name and stat info. Subdirectories are left alone; sweeps
|
||||
// that own directory trees handle them explicitly.
|
||||
func sweepDirFunc(
|
||||
ctx context.Context,
|
||||
dir string,
|
||||
shouldDelete func(name string, info os.FileInfo) bool,
|
||||
) (Result, error) {
|
||||
entries, err := os.ReadDir(dir)
|
||||
if err != nil {
|
||||
if os.IsNotExist(err) {
|
||||
return Result{}, nil
|
||||
}
|
||||
|
||||
return Result{}, fmt.Errorf("read %s: %w", dir, err)
|
||||
}
|
||||
|
||||
var result Result
|
||||
|
||||
for _, entry := range entries {
|
||||
if ctx.Err() != nil {
|
||||
return result, nil
|
||||
}
|
||||
|
||||
if entry.IsDir() {
|
||||
continue
|
||||
}
|
||||
|
||||
info, infoErr := entry.Info()
|
||||
if infoErr != nil {
|
||||
continue
|
||||
}
|
||||
|
||||
if !shouldDelete(entry.Name(), info) {
|
||||
continue
|
||||
}
|
||||
|
||||
if err := os.Remove(filepath.Join(dir, entry.Name())); err != nil {
|
||||
continue
|
||||
}
|
||||
|
||||
result.FilesDeleted++
|
||||
result.BytesFreed += info.Size()
|
||||
}
|
||||
|
||||
return result, nil
|
||||
}
|
||||
|
||||
// dirSize totals the files in a directory tree.
|
||||
func dirSize(dir string) (bytes, files int64) {
|
||||
_ = filepath.WalkDir(dir, func(_ string, d os.DirEntry, err error) error {
|
||||
if err != nil || d.IsDir() {
|
||||
return nil //nolint:nilerr // best-effort accounting
|
||||
}
|
||||
|
||||
info, infoErr := d.Info()
|
||||
if infoErr != nil {
|
||||
return nil
|
||||
}
|
||||
|
||||
bytes += info.Size()
|
||||
files++
|
||||
|
||||
return nil
|
||||
})
|
||||
|
||||
return bytes, files
|
||||
}
|
||||
Reference in New Issue
Block a user