feat: data lifecycle rewrite, download clients, wanted list, and central catalog index
Build & publish Arch package / arch-package (push) Successful in 2m12s
Search index maintenance / maintain-index (push) Successful in 2h22m28s

Ships the fresh-start schema cleanup: rebuilt explore catalog index
pipeline (dump import, artifact fetch/build, incremental listen-count
refresh), a new download subsystem (Lidarr/Prowlarr/qBittorrent/SABnzbd/
slskd/yt-dlp providers, staging, reconciliation, wanted list), and the
supporting schema/query/store changes across backend and frontend.

Also includes two smaller follow-ups: bump the central index's
rebuild-after cadence from 90 to 180 days, and remove the Explore
"library only" online/offline toggle entirely (frontend-only, no
backend counterpart) rather than carry unused UI/state.

Co-Authored-By: Claude Sonnet 5 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_01Y2Agd9af5hE7qzti2ackiS
This commit is contained in:
2026-08-06 17:12:01 -04:00
co-authored by Claude Sonnet 5
parent d0d86f85d5
commit e190fd75b9
165 changed files with 31088 additions and 5192 deletions
+214
View File
@@ -0,0 +1,214 @@
// Package maintenance runs the janitorial work that keeps persisted
// data from accumulating without bound.
//
// It exists because cleanup used to have no owner. Functions that
// deleted expired rows were written and then never called; files written
// by one package had no counterpart anywhere that removed them. A
// registry makes the set of janitorial jobs a single visible list, so a
// new cache that forgets to register is obvious in review rather than
// discovered years later as unbounded growth.
//
// Policies come from the classification in backend/datamap:
//
// - Derived data is swept against a live set computed from the data it
// was derived from. Anything not in the live set is garbage.
// - Cache data is evicted by age, because it has no owner to be
// compared against and is merely expensive — not impossible — to
// re-fetch.
//
// Sweeps are idempotent and safe to interrupt: each deletes only what it
// has positively identified as unreferenced, so a partial run simply
// leaves work for the next one.
package maintenance
import (
"context"
"log/slog"
"sync"
"time"
)
// Result reports what a single job reclaimed.
type Result struct {
// RowsDeleted counts database rows removed.
RowsDeleted int64
// FilesDeleted counts files removed from disk.
FilesDeleted int64
// BytesFreed is the size of those files.
BytesFreed int64
}
// empty reports whether the job found nothing to do, so quiet runs can
// be logged at a lower level.
func (r Result) empty() bool {
return r.RowsDeleted == 0 && r.FilesDeleted == 0
}
// Job is one unit of janitorial work.
type Job struct {
// Name identifies the job in logs and in the run record.
Name string
// MinInterval is the minimum time between runs. A job is skipped if
// it ran more recently than this, so hooking the runner to a
// frequently-firing trigger stays cheap.
MinInterval time.Duration
// Run performs the work. It must be idempotent and must respect
// context cancellation.
Run func(ctx context.Context) (Result, error)
}
// Runner holds the registered jobs and enforces their intervals.
type Runner struct {
mu sync.Mutex
jobs []Job
lastRun map[string]time.Time
logger *slog.Logger
}
// NewRunner returns an empty runner.
func NewRunner(logger *slog.Logger) *Runner {
return &Runner{
lastRun: make(map[string]time.Time),
logger: logger,
}
}
// Register adds a job. Registering a name twice replaces the earlier
// job, so wiring code can be re-run without accumulating duplicates.
func (r *Runner) Register(job Job) {
r.mu.Lock()
defer r.mu.Unlock()
for i, existing := range r.jobs {
if existing.Name == job.Name {
r.jobs[i] = job
return
}
}
r.jobs = append(r.jobs, job)
}
// JobNames returns the registered job names, for tests and diagnostics.
func (r *Runner) JobNames() []string {
r.mu.Lock()
defer r.mu.Unlock()
names := make([]string, 0, len(r.jobs))
for _, j := range r.jobs {
names = append(names, j.Name)
}
return names
}
// due reports whether a job's interval has elapsed.
func (r *Runner) due(job Job, now time.Time) bool {
r.mu.Lock()
defer r.mu.Unlock()
last, ran := r.lastRun[job.Name]
if !ran {
return true
}
return now.Sub(last) >= job.MinInterval
}
func (r *Runner) markRun(name string, at time.Time) {
r.mu.Lock()
defer r.mu.Unlock()
r.lastRun[name] = at
}
// snapshot copies the job list so a run does not hold the lock while
// executing jobs.
func (r *Runner) snapshot() []Job {
r.mu.Lock()
defer r.mu.Unlock()
out := make([]Job, len(r.jobs))
copy(out, r.jobs)
return out
}
// RunDue runs every job whose interval has elapsed. A job that fails is
// logged and does not prevent the others from running; janitorial work
// is best-effort by nature and the next run will retry.
func (r *Runner) RunDue(ctx context.Context) Result {
var total Result
for _, job := range r.snapshot() {
if ctx.Err() != nil {
r.logger.Info("maintenance cancelled", "after", job.Name)
break
}
now := time.Now()
if !r.due(job, now) {
continue
}
start := time.Now()
result, err := job.Run(ctx)
r.markRun(job.Name, now)
if err != nil {
r.logger.Warn("maintenance job failed",
"job", job.Name, "err", err,
"duration", time.Since(start),
)
continue
}
total.RowsDeleted += result.RowsDeleted
total.FilesDeleted += result.FilesDeleted
total.BytesFreed += result.BytesFreed
if result.empty() {
r.logger.Debug("maintenance job found nothing",
"job", job.Name, "duration", time.Since(start),
)
continue
}
r.logger.Info("maintenance job reclaimed",
"job", job.Name,
"rows", result.RowsDeleted,
"files", result.FilesDeleted,
"bytes", result.BytesFreed,
"duration", time.Since(start),
)
}
return total
}
// Start runs the due jobs immediately and then on every tick until the
// context is cancelled. It returns straight away; the loop runs in its
// own goroutine.
func (r *Runner) Start(ctx context.Context, tick time.Duration) {
go func() {
r.RunDue(ctx)
ticker := time.NewTicker(tick)
defer ticker.Stop()
for {
select {
case <-ctx.Done():
return
case <-ticker.C:
r.RunDue(ctx)
}
}
}()
}
+434
View File
@@ -0,0 +1,434 @@
package maintenance
import (
"context"
"errors"
"log/slog"
"os"
"path/filepath"
"slices"
"testing"
"time"
"yellowjacket/backend/database"
)
// errTestJobFailed stands in for a job returning an error.
var errTestJobFailed = errors.New("job failed")
func testRunner() *Runner {
return NewRunner(slog.Default())
}
func TestRunnerRunsRegisteredJobs(t *testing.T) {
t.Parallel()
r := testRunner()
var ran int
r.Register(Job{
Name: "counter",
Run: func(_ context.Context) (Result, error) {
ran++
return Result{RowsDeleted: 3}, nil
},
})
total := r.RunDue(context.Background())
if ran != 1 {
t.Errorf("job ran %d times, want 1", ran)
}
if total.RowsDeleted != 3 {
t.Errorf("RowsDeleted = %d, want 3", total.RowsDeleted)
}
}
// A job must not run again before its interval has elapsed, so hooking
// the runner to a frequent trigger stays cheap.
func TestRunnerRespectsMinInterval(t *testing.T) {
t.Parallel()
r := testRunner()
var ran int
r.Register(Job{
Name: "throttled",
MinInterval: time.Hour,
Run: func(_ context.Context) (Result, error) {
ran++
return Result{}, nil
},
})
r.RunDue(context.Background())
r.RunDue(context.Background())
r.RunDue(context.Background())
if ran != 1 {
t.Errorf("job ran %d times despite 1h interval, want 1", ran)
}
}
// One failing job must not prevent the others from running — janitorial
// work is best-effort and the next run retries.
func TestRunnerContinuesAfterFailure(t *testing.T) {
t.Parallel()
r := testRunner()
var secondRan bool
r.Register(Job{
Name: "failing",
Run: func(_ context.Context) (Result, error) {
return Result{}, errTestJobFailed
},
})
r.Register(Job{
Name: "healthy",
Run: func(_ context.Context) (Result, error) {
secondRan = true
return Result{}, nil
},
})
r.RunDue(context.Background())
if !secondRan {
t.Error("second job did not run after the first failed")
}
}
// Registering the same name twice replaces the job rather than
// accumulating duplicates, so wiring code is safe to re-run.
func TestRunnerRegisterReplaces(t *testing.T) {
t.Parallel()
r := testRunner()
noop := func(_ context.Context) (Result, error) { return Result{}, nil }
r.Register(Job{Name: "dup", Run: noop})
r.Register(Job{Name: "dup", Run: noop})
if names := r.JobNames(); len(names) != 1 {
t.Errorf("JobNames() = %v, want one entry", names)
}
}
func TestRunnerStopsOnCancelledContext(t *testing.T) {
t.Parallel()
r := testRunner()
ctx, cancel := context.WithCancel(context.Background())
cancel()
var ran bool
r.Register(Job{
Name: "should-not-run",
Run: func(_ context.Context) (Result, error) {
ran = true
return Result{}, nil
},
})
r.RunDue(ctx)
if ran {
t.Error("job ran despite cancelled context")
}
}
// ---------------------------------------------------------------------------
// Sweeps
// ---------------------------------------------------------------------------
func TestExpiredHTTPCacheJob(t *testing.T) {
t.Parallel()
db := database.NewTestDB(t)
if _, err := db.ExecContext(
`INSERT INTO http_cache (url_key, response, expires_at)
VALUES ('stale', '{}', datetime('now', '-1 day')),
('fresh', '{}', datetime('now', '+1 day'))`,
); err != nil {
t.Fatalf("seed http_cache: %v", err)
}
result, err := ExpiredHTTPCacheJob(db).Run(context.Background())
if err != nil {
t.Fatalf("run job: %v", err)
}
if result.RowsDeleted != 1 {
t.Errorf("RowsDeleted = %d, want 1", result.RowsDeleted)
}
var remaining string
rows, err := db.QueryContext("SELECT url_key FROM http_cache")
if err != nil {
t.Fatalf("query http_cache: %v", err)
}
defer func() { _ = rows.Close() }()
if rows.Next() {
_ = rows.Scan(&remaining)
}
if remaining != "fresh" {
t.Errorf("remaining row = %q, want the unexpired one", remaining)
}
}
// The covers sweep must delete files no cover_art row references while
// keeping the referenced original and every derived variant.
func TestOrphanedCoverFilesJob(t *testing.T) {
t.Parallel()
db := database.NewTestDB(t)
dir := t.TempDir()
// Stand-in for library.CoverArtFileSet.
expand := func(original string) []string {
base := filepath.Base(original)
base = base[:len(base)-len(filepath.Ext(base))]
return []string{
original,
filepath.Join(filepath.Dir(original), base+"_sm.jpg"),
filepath.Join(filepath.Dir(original), base+"_md.jpg"),
}
}
keep := []string{"live.jpg", "live_sm.jpg", "live_md.jpg"}
drop := []string{"orphan.jpg", "orphan_sm.jpg", "stray_md.jpg"}
for _, name := range slices.Concat(keep, drop) {
if err := os.WriteFile(
filepath.Join(dir, name), []byte("img"), 0o600,
); err != nil {
t.Fatalf("write %s: %v", name, err)
}
}
if _, err := db.ExecContext(
`INSERT INTO cover_art (is_embedded, file_path, mime_type)
VALUES (0, ?, 'image/jpeg')`,
filepath.Join(dir, "live.jpg"),
); err != nil {
t.Fatalf("seed cover_art: %v", err)
}
result, err := OrphanedCoverFilesJob(db, dir, expand).
Run(context.Background())
if err != nil {
t.Fatalf("run job: %v", err)
}
if result.FilesDeleted != int64(len(drop)) {
t.Errorf("FilesDeleted = %d, want %d", result.FilesDeleted, len(drop))
}
for _, name := range keep {
if _, err := os.Stat(filepath.Join(dir, name)); err != nil {
t.Errorf("referenced file %s was deleted", name)
}
}
for _, name := range drop {
if _, err := os.Stat(filepath.Join(dir, name)); !os.IsNotExist(err) {
t.Errorf("orphan %s survived the sweep", name)
}
}
}
// An empty live set means the query saw nothing, not that every cover is
// garbage. The sweep must refuse to empty the directory in that case.
func TestOrphanedCoverFilesJob_EmptyLiveSetIsNoOp(t *testing.T) {
t.Parallel()
db := database.NewTestDB(t)
dir := t.TempDir()
path := filepath.Join(dir, "something.jpg")
if err := os.WriteFile(path, []byte("img"), 0o600); err != nil {
t.Fatalf("write file: %v", err)
}
expand := func(p string) []string { return []string{p} }
result, err := OrphanedCoverFilesJob(db, dir, expand).
Run(context.Background())
if err != nil {
t.Fatalf("run job: %v", err)
}
if result.FilesDeleted != 0 {
t.Errorf("FilesDeleted = %d, want 0", result.FilesDeleted)
}
if _, err := os.Stat(path); err != nil {
t.Error("sweep emptied the directory on an empty live set")
}
}
// Artwork for an artist in the library is kept regardless of age;
// artwork for a browsed artist ages out.
func TestOrphanedArtistImagesJob(t *testing.T) {
t.Parallel()
db := database.NewTestDB(t)
dir := t.TempDir()
const (
ownedMBID = "11111111-1111-1111-1111-111111111111"
browsedMBID = "22222222-2222-2222-2222-222222222222"
recentMBID = "33333333-3333-3333-3333-333333333333"
)
for _, mbid := range []string{ownedMBID, browsedMBID, recentMBID} {
artistDir := filepath.Join(dir, mbid)
if err := os.MkdirAll(artistDir, 0o755); err != nil {
t.Fatalf("mkdir %s: %v", mbid, err)
}
if err := os.WriteFile(
filepath.Join(artistDir, "primary.jpg"), []byte("img"), 0o600,
); err != nil {
t.Fatalf("write image: %v", err)
}
}
// The owned artist is in the library.
if _, err := db.ExecContext(
`INSERT INTO artists (name, mbid) VALUES ('Owned', ?)`, ownedMBID,
); err != nil {
t.Fatalf("seed artists: %v", err)
}
old := time.Now().Add(-200 * 24 * time.Hour)
for _, tc := range []struct {
mbid string
created time.Time
}{
{ownedMBID, old},
{browsedMBID, old},
{recentMBID, time.Now()},
} {
if _, err := db.ExecContext(
`INSERT INTO artist_images
(artist_mbid, source, source_url, file_path, created_at)
VALUES (?, 'test', 'http://x', ?, ?)`,
tc.mbid,
filepath.Join(dir, tc.mbid, "primary.jpg"),
tc.created,
); err != nil {
t.Fatalf("seed artist_images for %s: %v", tc.mbid, err)
}
}
if _, err := OrphanedArtistImagesJob(db, dir).
Run(context.Background()); err != nil {
t.Fatalf("run job: %v", err)
}
if _, err := os.Stat(filepath.Join(dir, ownedMBID)); err != nil {
t.Error("artwork for a library artist was evicted")
}
if _, err := os.Stat(filepath.Join(dir, recentMBID)); err != nil {
t.Error("recently fetched artwork was evicted")
}
if _, err := os.Stat(filepath.Join(dir, browsedMBID)); !os.IsNotExist(err) {
t.Error("stale browsed artwork survived the sweep")
}
// The rows must go with the files.
rows, err := db.QueryContext(
"SELECT COUNT(*) FROM artist_images WHERE artist_mbid = ?",
browsedMBID,
)
if err != nil {
t.Fatalf("count rows: %v", err)
}
defer func() { _ = rows.Close() }()
var n int
if rows.Next() {
_ = rows.Scan(&n)
}
if n != 0 {
t.Errorf("artist_images rows for evicted artist = %d, want 0", n)
}
}
func TestExpiredProxyCacheJob(t *testing.T) {
t.Parallel()
dir := t.TempDir()
stale := filepath.Join(dir, "stale.jpg")
fresh := filepath.Join(dir, "fresh.jpg")
for _, p := range []string{stale, fresh} {
if err := os.WriteFile(p, []byte("img"), 0o600); err != nil {
t.Fatalf("write %s: %v", p, err)
}
}
old := time.Now().Add(-60 * 24 * time.Hour)
if err := os.Chtimes(stale, old, old); err != nil {
t.Fatalf("chtimes: %v", err)
}
result, err := ExpiredProxyCacheJob(dir).Run(context.Background())
if err != nil {
t.Fatalf("run job: %v", err)
}
if result.FilesDeleted != 1 {
t.Errorf("FilesDeleted = %d, want 1", result.FilesDeleted)
}
if _, err := os.Stat(fresh); err != nil {
t.Error("recently written thumbnail was evicted")
}
if _, err := os.Stat(stale); !os.IsNotExist(err) {
t.Error("stale thumbnail survived the sweep")
}
}
// A missing directory is normal on a fresh install and must not error.
func TestSweepMissingDirectory(t *testing.T) {
t.Parallel()
result, err := ExpiredProxyCacheJob(
filepath.Join(t.TempDir(), "does-not-exist"),
).Run(context.Background())
if err != nil {
t.Fatalf("missing directory returned an error: %v", err)
}
if result.FilesDeleted != 0 {
t.Errorf("FilesDeleted = %d, want 0", result.FilesDeleted)
}
}
+356
View File
@@ -0,0 +1,356 @@
package maintenance
import (
"context"
"fmt"
"os"
"path/filepath"
"time"
"yellowjacket/backend/database"
)
// Retention windows for cache data. Cached artwork for artists the user
// actually owns is kept indefinitely; art fetched while browsing Explore
// is transient and ages out.
const (
// browsedArtRetention is how long artwork for a non-library artist
// survives after it was fetched.
browsedArtRetention = 90 * 24 * time.Hour
// proxyCacheRetention is how long an Explore cover-art thumbnail
// survives after it was last written.
proxyCacheRetention = 30 * 24 * time.Hour
)
// Default intervals. These are minimums, not schedules — the runner
// skips a job that ran more recently.
const (
frequentInterval = 6 * time.Hour
dailyInterval = 24 * time.Hour
)
// ExpiredHTTPCacheJob deletes HTTP cache rows past their TTL.
//
// Reads already filter on expires_at, so expired rows are inert — but
// nothing was deleting them, so the table grew without bound for the
// life of the install.
func ExpiredHTTPCacheJob(db *database.DB) Job {
return Job{
Name: "http-cache-evict",
MinInterval: frequentInterval,
Run: func(_ context.Context) (Result, error) {
res, err := db.ExecContext(
"DELETE FROM http_cache WHERE expires_at < datetime('now')",
)
if err != nil {
return Result{}, fmt.Errorf(
"delete expired http_cache rows: %w", err,
)
}
rows, _ := res.RowsAffected()
return Result{RowsDeleted: rows}, nil
},
}
}
// OrphanedCoverFilesJob removes files from the covers directory that no
// cover_art row references.
//
// Cover art is derived data, so the live set is authoritative: every
// file that is not the original named by a cover_art row, or one of that
// original's derived size variants, is garbage. This reclaims art left
// behind by earlier versions that deleted only the original and left its
// thumbnails.
func OrphanedCoverFilesJob(
db *database.DB,
coversDir string,
expandVariants func(originalPath string) []string,
) Job {
return Job{
Name: "covers-sweep",
MinInterval: dailyInterval,
Run: func(ctx context.Context) (Result, error) {
live, err := liveCoverFiles(db, coversDir, expandVariants)
if err != nil {
return Result{}, err
}
// A covers directory with no live entries almost certainly
// means the query failed to see the real table rather than
// that every cover is garbage. Refuse to empty the
// directory on that basis.
if len(live) == 0 {
return Result{}, nil
}
return sweepDir(ctx, coversDir, func(name string) bool {
return !live[name]
})
},
}
}
// liveCoverFiles returns the basenames of every file the covers
// directory is supposed to contain.
func liveCoverFiles(
db *database.DB,
coversDir string,
expandVariants func(string) []string,
) (map[string]bool, error) {
rows, err := db.QueryContext("SELECT file_path FROM cover_art")
if err != nil {
return nil, fmt.Errorf("read cover_art paths: %w", err)
}
defer func() { _ = rows.Close() }()
live := make(map[string]bool)
for rows.Next() {
var path string
if err := rows.Scan(&path); err != nil {
return nil, fmt.Errorf("scan cover_art path: %w", err)
}
// Rows may store an absolute path from a previous install
// location, so compare by basename within the covers directory.
original := filepath.Join(coversDir, filepath.Base(path))
for _, variant := range expandVariants(original) {
live[filepath.Base(variant)] = true
}
}
if err := rows.Err(); err != nil {
return nil, fmt.Errorf("iterate cover_art paths: %w", err)
}
return live, nil
}
// OrphanedArtistImagesJob evicts cached artist artwork.
//
// Artist images are cache data with no owner to compare against — they
// are fetched for any artist the user browses in Explore, most of whom
// are not in the library. The policy is therefore twofold: artwork for
// an artist the user owns is kept indefinitely, and everything else ages
// out. Rows whose file has vanished are dropped so the table matches
// what is actually on disk.
func OrphanedArtistImagesJob(db *database.DB, artistImagesDir string) Job {
return Job{
Name: "artist-images-sweep",
MinInterval: dailyInterval,
Run: func(ctx context.Context) (Result, error) {
var result Result
cutoff := time.Now().Add(-browsedArtRetention)
// Collect the directories to remove before deleting rows, so
// a failure partway leaves rows pointing at real files
// rather than the reverse.
stale, err := staleArtistMBIDs(db, cutoff)
if err != nil {
return Result{}, err
}
for _, mbid := range stale {
if ctx.Err() != nil {
return result, nil
}
dir := filepath.Join(artistImagesDir, mbid)
freed, files := dirSize(dir)
if err := os.RemoveAll(dir); err != nil && !os.IsNotExist(err) {
continue
}
result.FilesDeleted += files
result.BytesFreed += freed
}
if len(stale) > 0 {
rows, delErr := deleteArtistImageRows(db, stale)
if delErr != nil {
return result, delErr
}
result.RowsDeleted += rows
}
return result, nil
},
}
}
// staleArtistMBIDs returns artist MBIDs whose cached artwork may be
// evicted: fetched before the cutoff and not an artist in the library.
func staleArtistMBIDs(
db *database.DB,
cutoff time.Time,
) ([]string, error) {
rows, err := db.QueryContext(
`SELECT DISTINCT artist_mbid FROM artist_images
WHERE created_at < ?
AND artist_mbid NOT IN (
SELECT mbid FROM artists
WHERE mbid IS NOT NULL AND mbid != ''
)`,
cutoff,
)
if err != nil {
return nil, fmt.Errorf("query stale artist images: %w", err)
}
defer func() { _ = rows.Close() }()
var mbids []string
for rows.Next() {
var mbid string
if err := rows.Scan(&mbid); err != nil {
return nil, fmt.Errorf("scan artist mbid: %w", err)
}
if mbid != "" {
mbids = append(mbids, mbid)
}
}
if err := rows.Err(); err != nil {
return nil, fmt.Errorf("iterate stale artist images: %w", err)
}
return mbids, nil
}
// deleteArtistImageRows removes the rows for the given artist MBIDs.
func deleteArtistImageRows(
db *database.DB,
mbids []string,
) (int64, error) {
var total int64
for _, mbid := range mbids {
res, err := db.ExecContext(
"DELETE FROM artist_images WHERE artist_mbid = ?", mbid,
)
if err != nil {
return total, fmt.Errorf(
"delete artist_images rows for %s: %w", mbid, err,
)
}
n, _ := res.RowsAffected()
total += n
}
return total, nil
}
// ExpiredProxyCacheJob evicts Explore cover-art thumbnails that have not
// been rewritten within the retention window.
//
// This cache has no database table at all — it is keyed by release-group
// MBID on the filesystem — so age is the only signal available.
func ExpiredProxyCacheJob(proxyCacheDir string) Job {
return Job{
Name: "cover-art-proxy-sweep",
MinInterval: dailyInterval,
Run: func(ctx context.Context) (Result, error) {
cutoff := time.Now().Add(-proxyCacheRetention)
return sweepDirFunc(ctx, proxyCacheDir,
func(_ string, info os.FileInfo) bool {
return info.ModTime().Before(cutoff)
},
)
},
}
}
// sweepDir removes every file in dir for which shouldDelete reports true.
func sweepDir(
ctx context.Context,
dir string,
shouldDelete func(name string) bool,
) (Result, error) {
return sweepDirFunc(ctx, dir, func(name string, _ os.FileInfo) bool {
return shouldDelete(name)
})
}
// sweepDirFunc removes files from a flat directory based on a predicate
// over the name and stat info. Subdirectories are left alone; sweeps
// that own directory trees handle them explicitly.
func sweepDirFunc(
ctx context.Context,
dir string,
shouldDelete func(name string, info os.FileInfo) bool,
) (Result, error) {
entries, err := os.ReadDir(dir)
if err != nil {
if os.IsNotExist(err) {
return Result{}, nil
}
return Result{}, fmt.Errorf("read %s: %w", dir, err)
}
var result Result
for _, entry := range entries {
if ctx.Err() != nil {
return result, nil
}
if entry.IsDir() {
continue
}
info, infoErr := entry.Info()
if infoErr != nil {
continue
}
if !shouldDelete(entry.Name(), info) {
continue
}
if err := os.Remove(filepath.Join(dir, entry.Name())); err != nil {
continue
}
result.FilesDeleted++
result.BytesFreed += info.Size()
}
return result, nil
}
// dirSize totals the files in a directory tree.
func dirSize(dir string) (bytes, files int64) {
_ = filepath.WalkDir(dir, func(_ string, d os.DirEntry, err error) error {
if err != nil || d.IsDir() {
return nil //nolint:nilerr // best-effort accounting
}
info, infoErr := d.Info()
if infoErr != nil {
return nil
}
bytes += info.Size()
files++
return nil
})
return bytes, files
}