The walk's predicates partition the artifact's key space, so a merge that ends with fewer rows than the artifact declares does not mean the artifact was smaller than it said — it means a predicate filtered rows out, and the catalog is quietly partial while reporting complete. Equality rather than a lower bound: RowsAffected counts an upsert that changes nothing, and a row already merged locally is counted again here. One reachable case, so this is not merely a tripwire. A row whose mbid is empty is excluded by `mbid > ?` in both encodings, so an artifact carrying one imports as a success with a row missing — which is the shape #258 had, one cause over. The test covers exactly that artifact. Refs #258
939 lines
28 KiB
Go
939 lines
28 KiB
Go
package explore
|
|
|
|
import (
|
|
"bytes"
|
|
"context"
|
|
"database/sql"
|
|
"encoding/hex"
|
|
"os"
|
|
"path/filepath"
|
|
"strings"
|
|
"testing"
|
|
|
|
"yellowjacket/backend/database"
|
|
)
|
|
|
|
// artifactRow is one catalog row written into a test artifact.
|
|
type artifactRow struct {
|
|
entityType string
|
|
mbid string
|
|
title string
|
|
artistName string
|
|
artistMBID string
|
|
popularity int
|
|
}
|
|
|
|
// writeTestArtifact builds an artifact file matching what cmd/indexexport
|
|
// produces: catalog columns only, no FTS, no triggers.
|
|
func writeTestArtifact(
|
|
t *testing.T, meta map[string]string, rows []artifactRow,
|
|
) string {
|
|
t.Helper()
|
|
|
|
path := filepath.Join(t.TempDir(), "core-index.db")
|
|
|
|
db, err := sql.Open("sqlite", "file:"+path)
|
|
if err != nil {
|
|
t.Fatalf("open artifact: %v", err)
|
|
}
|
|
|
|
defer func() { _ = db.Close() }()
|
|
|
|
for _, stmt := range []string{
|
|
`CREATE TABLE explore_index (
|
|
entity_type TEXT NOT NULL,
|
|
mbid TEXT NOT NULL,
|
|
title TEXT NOT NULL,
|
|
artist_name TEXT NOT NULL,
|
|
artist_mbid TEXT NOT NULL,
|
|
aliases TEXT NOT NULL DEFAULT '',
|
|
popularity INTEGER NOT NULL DEFAULT 0,
|
|
listener_count INTEGER NOT NULL DEFAULT 0,
|
|
duration INTEGER NOT NULL DEFAULT 0,
|
|
caa_release_mbid TEXT NOT NULL DEFAULT '',
|
|
release_name TEXT NOT NULL DEFAULT '',
|
|
primary_type TEXT NOT NULL DEFAULT '',
|
|
secondary_types TEXT NOT NULL DEFAULT '',
|
|
release_date TEXT NOT NULL DEFAULT '',
|
|
artist_type TEXT NOT NULL DEFAULT '',
|
|
country TEXT NOT NULL DEFAULT '',
|
|
disambiguation TEXT NOT NULL DEFAULT '',
|
|
sort_name TEXT NOT NULL DEFAULT '',
|
|
discog_fetched INTEGER NOT NULL DEFAULT 0,
|
|
PRIMARY KEY (mbid)
|
|
) WITHOUT ROWID`,
|
|
`CREATE TABLE artifact_meta (
|
|
key TEXT PRIMARY KEY,
|
|
value TEXT NOT NULL
|
|
)`,
|
|
} {
|
|
if _, err := db.Exec(stmt); err != nil {
|
|
t.Fatalf("create artifact schema: %v", err)
|
|
}
|
|
}
|
|
|
|
stampArtifactMeta(t, db, meta)
|
|
|
|
for _, r := range rows {
|
|
if _, err := db.Exec(`
|
|
INSERT INTO explore_index
|
|
(entity_type, mbid, title, artist_name, artist_mbid, popularity)
|
|
VALUES (?, ?, ?, ?, ?, ?)`,
|
|
r.entityType, r.mbid, r.title, r.artistName, r.artistMBID, r.popularity,
|
|
); err != nil {
|
|
t.Fatalf("insert artifact row: %v", err)
|
|
}
|
|
}
|
|
|
|
return path
|
|
}
|
|
|
|
// compactArtifactSchema is the artifact cmd/indexexport publishes: the
|
|
// catalog's ids as 16 raw bytes, its entity types as codes, and the
|
|
// per-release-group total_tracks the exporter added after the first
|
|
// artifact was shipped.
|
|
//
|
|
// It matters that a fixture carries this encoding and not the older
|
|
// text one, because SQLite does not coerce between TEXT and BLOB and
|
|
// every comparison the importer makes against an mbid is therefore
|
|
// encoding-sensitive. writeTestArtifact above is the *other* fixture:
|
|
// it still writes the text form, which is what the first published
|
|
// artifact carries and what the importer must keep reading.
|
|
var compactArtifactSchema = []string{
|
|
`CREATE TABLE explore_index (
|
|
entity_type INTEGER NOT NULL,
|
|
mbid BLOB NOT NULL,
|
|
title TEXT NOT NULL,
|
|
artist_name TEXT NOT NULL,
|
|
artist_mbid BLOB NOT NULL,
|
|
aliases TEXT NOT NULL DEFAULT '',
|
|
popularity INTEGER NOT NULL DEFAULT 0,
|
|
listener_count INTEGER NOT NULL DEFAULT 0,
|
|
duration INTEGER NOT NULL DEFAULT 0,
|
|
caa_release_mbid BLOB NOT NULL DEFAULT x'',
|
|
release_name TEXT NOT NULL DEFAULT '',
|
|
primary_type TEXT NOT NULL DEFAULT '',
|
|
secondary_types TEXT NOT NULL DEFAULT '',
|
|
release_date TEXT NOT NULL DEFAULT '',
|
|
total_tracks INTEGER NOT NULL DEFAULT 0,
|
|
artist_type TEXT NOT NULL DEFAULT '',
|
|
country TEXT NOT NULL DEFAULT '',
|
|
disambiguation TEXT NOT NULL DEFAULT '',
|
|
sort_name TEXT NOT NULL DEFAULT '',
|
|
discog_fetched INTEGER NOT NULL DEFAULT 0,
|
|
PRIMARY KEY (mbid)
|
|
) WITHOUT ROWID`,
|
|
`CREATE TABLE artifact_meta (
|
|
key TEXT PRIMARY KEY,
|
|
value TEXT NOT NULL
|
|
)`,
|
|
}
|
|
|
|
// writeCompactTestArtifact builds the artifact the exporter publishes
|
|
// today, in its own encoding, so the importer is exercised against what
|
|
// a client actually downloads rather than against what it was written
|
|
// for.
|
|
func writeCompactTestArtifact(
|
|
t *testing.T, meta map[string]string, rows []artifactRow,
|
|
) string {
|
|
t.Helper()
|
|
|
|
path := filepath.Join(t.TempDir(), "core-index.db")
|
|
|
|
db, err := sql.Open("sqlite", "file:"+path)
|
|
if err != nil {
|
|
t.Fatalf("open artifact: %v", err)
|
|
}
|
|
|
|
defer func() { _ = db.Close() }()
|
|
|
|
for _, stmt := range compactArtifactSchema {
|
|
if _, err := db.Exec(stmt); err != nil {
|
|
t.Fatalf("create artifact schema: %v", err)
|
|
}
|
|
}
|
|
|
|
stampArtifactMeta(t, db, meta)
|
|
|
|
for _, r := range rows {
|
|
if _, err := db.Exec(`
|
|
INSERT INTO explore_index
|
|
(entity_type, mbid, title, artist_name, artist_mbid, popularity)
|
|
VALUES (?, ?, ?, ?, ?, ?)`,
|
|
entityCode(r.entityType), mbidBytes(r.mbid), r.title,
|
|
r.artistName, mbidBytes(r.artistMBID), r.popularity,
|
|
); err != nil {
|
|
t.Fatalf("insert artifact row: %v", err)
|
|
}
|
|
}
|
|
|
|
return path
|
|
}
|
|
|
|
// stampArtifactMeta writes the artifact_meta rows a fixture declares.
|
|
func stampArtifactMeta(t *testing.T, db *sql.DB, meta map[string]string) {
|
|
t.Helper()
|
|
|
|
for k, v := range meta {
|
|
if _, err := db.Exec(
|
|
`INSERT INTO artifact_meta (key, value) VALUES (?, ?)`, k, v,
|
|
); err != nil {
|
|
t.Fatalf("stamp artifact meta: %v", err)
|
|
}
|
|
}
|
|
}
|
|
|
|
// validMeta is the artifact_meta a well-formed artifact carries.
|
|
func validMeta() map[string]string {
|
|
return map[string]string{
|
|
"artifact_version": supportedArtifactVersion,
|
|
"built_at": "2026-07-17T03:53:43Z",
|
|
"listens_applied_series": "2593",
|
|
"source_rows": "2052168",
|
|
}
|
|
}
|
|
|
|
func TestImportCoreArtifactMergesCatalog(t *testing.T) {
|
|
db := database.NewTestDB(t)
|
|
si := NewSearchIndex(db, nil, nil, testLogger())
|
|
|
|
path := writeTestArtifact(t, validMeta(), []artifactRow{
|
|
{"artist", artA, "Artist A", "Artist A", artA, 5000},
|
|
{"artist", artB, "Artist B", "Artist B", artB, 4000},
|
|
{"release_group", rgA, "Album A", "Artist A", artA, 3000},
|
|
{"recording", recA, "Song A", "Artist A", artA, 2000},
|
|
})
|
|
|
|
if err := si.importCoreArtifact(context.Background(), path); err != nil {
|
|
t.Fatalf("importCoreArtifact: %v", err)
|
|
}
|
|
|
|
var got int
|
|
if err := db.QueryRowWriter(
|
|
"SELECT COUNT(*) FROM explore_index",
|
|
).Scan(&got); err != nil {
|
|
t.Fatalf("count rows: %v", err)
|
|
}
|
|
|
|
if got != 4 {
|
|
t.Errorf("merged %d rows, want 4", got)
|
|
}
|
|
|
|
// The merge must leave the index searchable: the FTS rebuild that
|
|
// closes the bulk-load window is the only thing populating it, since
|
|
// the triggers were dropped for the duration.
|
|
var hits int
|
|
if err := db.QueryRowWriter(
|
|
`SELECT COUNT(*) FROM explore_index_fts WHERE explore_index_fts MATCH ?`,
|
|
"Song",
|
|
).Scan(&hits); err != nil {
|
|
t.Fatalf("query fts: %v", err)
|
|
}
|
|
|
|
if hits == 0 {
|
|
t.Error("FTS index is empty after merge; search would return nothing")
|
|
}
|
|
}
|
|
|
|
func TestImportCoreArtifactStampsMeta(t *testing.T) {
|
|
db := database.NewTestDB(t)
|
|
si := NewSearchIndex(db, nil, nil, testLogger())
|
|
|
|
path := writeTestArtifact(t, validMeta(), []artifactRow{
|
|
{"artist", artA, "Artist A", "Artist A", artA, 5000},
|
|
})
|
|
|
|
if err := si.importCoreArtifact(context.Background(), path); err != nil {
|
|
t.Fatalf("importCoreArtifact: %v", err)
|
|
}
|
|
|
|
if !si.hasMeta(dumpImportDoneKey) {
|
|
t.Error("dump_import_done not stamped; a full dump import would run anyway")
|
|
}
|
|
|
|
// Without this the incremental refresh refuses to run and the
|
|
// shipped popularity never updates again.
|
|
if series, ok := si.metaInt(listensAppliedSeriesKey); !ok || series != 2593 {
|
|
t.Errorf("listens_applied_series = %d (ok=%v), want 2593", series, ok)
|
|
}
|
|
|
|
if !si.hasMeta(coreArtifactVersionKey) {
|
|
t.Error("core_artifact_version not stamped")
|
|
}
|
|
}
|
|
|
|
// A merge must never downgrade what the index already holds, because a
|
|
// user's own library rows and any lazily-fetched detail predate it.
|
|
func TestImportCoreArtifactPreservesLocalData(t *testing.T) {
|
|
db := database.NewTestDB(t)
|
|
si := NewSearchIndex(db, nil, nil, testLogger())
|
|
|
|
si.upsertBatch([]SearchIndexResult{{
|
|
EntityType: "recording",
|
|
MBID: recA,
|
|
Title: "Song A",
|
|
ArtistName: "Artist A",
|
|
ArtistMBID: artA,
|
|
Popularity: 9999,
|
|
Duration: 210000,
|
|
InLibrary: true,
|
|
DiscogFetched: true,
|
|
}})
|
|
|
|
path := writeTestArtifact(t, validMeta(), []artifactRow{
|
|
// Lower popularity and no duration: both must lose.
|
|
{"recording", recA, "Song A", "Artist A", artA, 10},
|
|
})
|
|
|
|
if err := si.importCoreArtifact(context.Background(), path); err != nil {
|
|
t.Fatalf("importCoreArtifact: %v", err)
|
|
}
|
|
|
|
var popularity, duration, inLibrary, discogFetched int
|
|
|
|
if err := db.QueryRowWriter(`
|
|
SELECT popularity, duration, in_library, discog_fetched
|
|
FROM explore_index WHERE mbid = ?`, dbMBID(recA),
|
|
).Scan(&popularity, &duration, &inLibrary, &discogFetched); err != nil {
|
|
t.Fatalf("read merged row: %v", err)
|
|
}
|
|
|
|
if popularity != 9999 {
|
|
t.Errorf("popularity = %d, want 9999 (higher must win)", popularity)
|
|
}
|
|
|
|
if duration != 210000 {
|
|
t.Errorf("duration = %d, want 210000 (artifact carries none)", duration)
|
|
}
|
|
|
|
if inLibrary != 1 {
|
|
t.Error("in_library was cleared; the artifact must not touch personal columns")
|
|
}
|
|
|
|
if discogFetched != 1 {
|
|
t.Error("discog_fetched was cleared by the merge")
|
|
}
|
|
}
|
|
|
|
// The batch walk is the part most likely to drop or duplicate rows, so
|
|
// it is exercised across many batch boundaries rather than one.
|
|
func TestImportCoreArtifactBatchWalkCoversAllRows(t *testing.T) {
|
|
db := database.NewTestDB(t)
|
|
si := NewSearchIndex(db, nil, nil, testLogger())
|
|
|
|
// Shrink the batch rather than growing the fixture: what matters is
|
|
// crossing several boundaries, including a short final one.
|
|
original := artifactMergeBatch
|
|
artifactMergeBatch = 100
|
|
|
|
t.Cleanup(func() { artifactMergeBatch = original })
|
|
|
|
const total = 337
|
|
|
|
rows := make([]artifactRow, 0, total)
|
|
for i := range total {
|
|
rows = append(rows, artifactRow{
|
|
entityType: "recording",
|
|
mbid: syntheticMBID(i),
|
|
title: "Song",
|
|
artistName: "Artist",
|
|
artistMBID: artA,
|
|
popularity: i,
|
|
})
|
|
}
|
|
|
|
path := writeTestArtifact(t, validMeta(), rows)
|
|
|
|
if err := si.importCoreArtifact(context.Background(), path); err != nil {
|
|
t.Fatalf("importCoreArtifact: %v", err)
|
|
}
|
|
|
|
var got int
|
|
if err := db.QueryRowWriter(
|
|
"SELECT COUNT(*) FROM explore_index",
|
|
).Scan(&got); err != nil {
|
|
t.Fatalf("count rows: %v", err)
|
|
}
|
|
|
|
if got != total {
|
|
t.Errorf("merged %d rows, want %d", got, total)
|
|
}
|
|
}
|
|
|
|
// TestImportCoreArtifactBatchWalkCoversAllRowsCompact is the batch walk
|
|
// on the encoding the exporter actually publishes.
|
|
//
|
|
// The walk positions itself by comparing the artifact's own mbid column
|
|
// against the last id it reached, and that column holds 16 raw bytes.
|
|
// SQLite does not coerce between TEXT and BLOB, and a blob sorts after
|
|
// every text value, so a cursor bound as text is a predicate that either
|
|
// matches every row or none: `mbid > ?` with an empty text key is true
|
|
// of the whole table, so
|
|
// the 100th row is always the 100th row and the bound never advances,
|
|
// while `mbid <= <text>` is false of the whole table, so no batch ever
|
|
// merges. The result is not a wrong import but an unbounded loop that
|
|
// merges nothing and never fails.
|
|
//
|
|
// Both encodings are covered on purpose. The walk was only ever tested
|
|
// against the text fixture above, which is why it shipped broken on the
|
|
// one the clients download.
|
|
func TestImportCoreArtifactBatchWalkCoversAllRowsCompact(t *testing.T) {
|
|
db := database.NewTestDB(t)
|
|
si := NewSearchIndex(db, nil, nil, testLogger())
|
|
|
|
original := artifactMergeBatch
|
|
artifactMergeBatch = 100
|
|
|
|
t.Cleanup(func() { artifactMergeBatch = original })
|
|
|
|
const total = 337
|
|
|
|
rows := make([]artifactRow, 0, total)
|
|
for i := range total {
|
|
rows = append(rows, artifactRow{
|
|
entityType: EntityRecording,
|
|
mbid: syntheticMBID(i),
|
|
title: "Song",
|
|
artistName: "Artist",
|
|
artistMBID: artA,
|
|
popularity: i,
|
|
})
|
|
}
|
|
|
|
path := writeCompactTestArtifact(t, validMeta(), rows)
|
|
|
|
if err := si.importCoreArtifact(context.Background(), path); err != nil {
|
|
t.Fatalf("importCoreArtifact: %v", err)
|
|
}
|
|
|
|
var got, top int
|
|
|
|
if err := db.QueryRowWriter(
|
|
"SELECT COUNT(*), MAX(popularity) FROM explore_index",
|
|
).Scan(&got, &top); err != nil {
|
|
t.Fatalf("count rows: %v", err)
|
|
}
|
|
|
|
if got != total {
|
|
t.Errorf("merged %d rows, want %d", got, total)
|
|
}
|
|
|
|
// A count alone would pass if the walk re-merged the same first
|
|
// batch forever, so the far end of the artifact is checked too.
|
|
if top != total-1 {
|
|
t.Errorf("highest popularity = %d, want %d", top, total-1)
|
|
}
|
|
}
|
|
|
|
func TestImportCoreArtifactRejectsBadArtifacts(t *testing.T) {
|
|
tests := []struct {
|
|
name string
|
|
meta map[string]string
|
|
rows []artifactRow
|
|
want error
|
|
}{
|
|
{
|
|
name: "version mismatch",
|
|
meta: map[string]string{"artifact_version": "99"},
|
|
rows: []artifactRow{{"artist", artA, "A", "A", artA, 1}},
|
|
want: ErrArtifactVersion,
|
|
},
|
|
{
|
|
name: "no version",
|
|
meta: map[string]string{},
|
|
rows: []artifactRow{{"artist", artA, "A", "A", artA, 1}},
|
|
want: ErrArtifactVersion,
|
|
},
|
|
{
|
|
name: "empty catalog",
|
|
meta: validMeta(),
|
|
rows: nil,
|
|
want: ErrArtifactUnusable,
|
|
},
|
|
}
|
|
|
|
for _, tt := range tests {
|
|
t.Run(tt.name, func(t *testing.T) {
|
|
db := database.NewTestDB(t)
|
|
si := NewSearchIndex(db, nil, nil, testLogger())
|
|
|
|
path := writeTestArtifact(t, tt.meta, tt.rows)
|
|
|
|
err := si.importCoreArtifact(context.Background(), path)
|
|
if err == nil {
|
|
t.Fatal("expected rejection, got nil")
|
|
}
|
|
|
|
if !strings.Contains(err.Error(), tt.want.Error()) {
|
|
t.Errorf("error = %v, want it to wrap %v", err, tt.want)
|
|
}
|
|
|
|
// A rejected artifact must not leave the index claiming it
|
|
// has a catalog, or the real build would never run.
|
|
if si.hasMeta(dumpImportDoneKey) {
|
|
t.Error("rejected artifact still stamped dump_import_done")
|
|
}
|
|
})
|
|
}
|
|
}
|
|
|
|
func TestInspectArtifactRejectsNonArtifactFile(t *testing.T) {
|
|
path := filepath.Join(t.TempDir(), "garbage.db")
|
|
|
|
if err := os.WriteFile(path, []byte("not a database"), 0o644); err != nil {
|
|
t.Fatalf("write file: %v", err)
|
|
}
|
|
|
|
if _, err := inspectArtifact(path); err == nil {
|
|
t.Error("expected rejection of a non-artifact file")
|
|
}
|
|
}
|
|
|
|
// syntheticMBID produces distinct, well-formed MBIDs for bulk fixtures.
|
|
func syntheticMBID(i int) string {
|
|
const hex = "0123456789abcdef"
|
|
|
|
buf := []byte("00000000-0000-0000-0000-000000000000")
|
|
|
|
for pos := len(buf) - 1; pos >= 0 && i > 0; pos-- {
|
|
if buf[pos] == '-' {
|
|
continue
|
|
}
|
|
|
|
buf[pos] = hex[i%16]
|
|
i /= 16
|
|
}
|
|
|
|
return string(buf)
|
|
}
|
|
|
|
// resolveArtistName falls back to the MBID when no name is available.
|
|
// That value must never reach the index: the upsert no longer defends
|
|
// against it, so the writer is the only thing standing in the way.
|
|
func TestAddFromCacheNeverStoresMBIDAsName(t *testing.T) {
|
|
db := database.NewTestDB(t)
|
|
si := NewSearchIndex(db, nil, nil, testLogger())
|
|
|
|
si.AddFromCache(artA, artA, []MBReleaseGroup{
|
|
{MBID: rgA, Title: "Album A", PrimaryType: "Album"},
|
|
})
|
|
|
|
var artistRows int
|
|
if err := db.QueryRowWriter(
|
|
`SELECT COUNT(*) FROM explore_index
|
|
WHERE entity_type = 'artist' AND title = mbid`,
|
|
).Scan(&artistRows); err != nil {
|
|
t.Fatalf("count artist rows: %v", err)
|
|
}
|
|
|
|
if artistRows != 0 {
|
|
t.Errorf("%d artist rows stored the MBID as their title", artistRows)
|
|
}
|
|
|
|
var artistName string
|
|
if err := db.QueryRowWriter(
|
|
`SELECT artist_name FROM explore_index WHERE mbid = ?`, dbMBID(rgA),
|
|
).Scan(&artistName); err != nil {
|
|
t.Fatalf("read release group: %v", err)
|
|
}
|
|
|
|
if artistName == artA {
|
|
t.Error("release group stored the artist MBID as its artist_name")
|
|
}
|
|
|
|
// A real name arriving later must still win over the empty one.
|
|
si.AddFromCache("Real Artist", artA, []MBReleaseGroup{
|
|
{MBID: rgA, Title: "Album A", PrimaryType: "Album"},
|
|
})
|
|
|
|
if err := db.QueryRowWriter(
|
|
`SELECT artist_name FROM explore_index WHERE mbid = ?`, dbMBID(rgA),
|
|
).Scan(&artistName); err != nil {
|
|
t.Fatalf("re-read release group: %v", err)
|
|
}
|
|
|
|
if artistName != "Real Artist" {
|
|
t.Errorf("artist_name = %q, want it filled in once known", artistName)
|
|
}
|
|
}
|
|
|
|
// The importer names the artifact's columns independently of the
|
|
// exporter that writes them. If the two lists drift, the merge either
|
|
// fails outright or silently shifts values into the wrong columns, so
|
|
// they are compared directly against cmd/indexexport's source.
|
|
func TestArtifactColumnsMatchExporter(t *testing.T) {
|
|
src, err := os.ReadFile("../../cmd/indexexport/main.go")
|
|
if err != nil {
|
|
t.Fatalf("read exporter: %v", err)
|
|
}
|
|
|
|
const marker = "const catalogColumns = `"
|
|
|
|
i := strings.Index(string(src), marker)
|
|
if i < 0 {
|
|
t.Fatalf("catalogColumns not found in cmd/indexexport/main.go")
|
|
}
|
|
|
|
rest := string(src)[i+len(marker):]
|
|
|
|
j := strings.Index(rest, "`")
|
|
if j < 0 {
|
|
t.Fatal("unterminated catalogColumns literal")
|
|
}
|
|
|
|
normalise := func(s string) string {
|
|
out := make([]string, 0, 32)
|
|
for _, f := range strings.Split(s, ",") {
|
|
out = append(out, strings.Join(strings.Fields(f), ""))
|
|
}
|
|
|
|
return strings.Join(out, ",")
|
|
}
|
|
|
|
exporter := normalise(rest[:j])
|
|
importer := normalise(artifactCatalogColumns)
|
|
|
|
if exporter != importer {
|
|
t.Errorf("column lists have drifted:\n exporter: %s\n importer: %s",
|
|
exporter, importer)
|
|
}
|
|
}
|
|
|
|
// TestImportCoreArtifactAcceptsBothEncodings is the compatibility half
|
|
// of the storage change.
|
|
//
|
|
// The catalog stores an MBID as 16 raw bytes and an entity type as a
|
|
// code, which took the table and its indexes from 677 MB to 389 MB. A
|
|
// published artifact carries whichever form the exporter that built it
|
|
// used, and there is one already out there in the older text form — so
|
|
// the importer decides by asking the artifact, not by trusting a
|
|
// version number, and both must land identically.
|
|
func TestImportCoreArtifactAcceptsBothEncodings(t *testing.T) {
|
|
compact := writeCompactTestArtifact(t, validMeta(), []artifactRow{
|
|
{EntityArtist, artA, "Artist A", "Artist A", artA, 5000},
|
|
})
|
|
|
|
live := database.NewTestDB(t)
|
|
si := NewSearchIndex(live, nil, nil, testLogger())
|
|
|
|
if err := si.importCoreArtifact(context.Background(), compact); err != nil {
|
|
t.Fatalf("importCoreArtifact (compact): %v", err)
|
|
}
|
|
|
|
got := si.LookupArtistByMBID(artA)
|
|
if got == nil {
|
|
t.Fatal("artist from a compact artifact was not imported")
|
|
}
|
|
|
|
if got.Title != "Artist A" || got.MBID != artA {
|
|
t.Errorf("imported %+v, want Artist A / %s", got, artA)
|
|
}
|
|
}
|
|
|
|
// TestImportCoreArtifactReadsTotalsWhenPresent covers both halves of the
|
|
// denominator's arrival: an artifact that carries total_tracks imports
|
|
// it, and one built before the column existed still imports at all.
|
|
//
|
|
// The second half is the one worth a test. Adding a column to the
|
|
// importer's SELECT list is how you break every artifact already
|
|
// published - "no such column: total_tracks", on a file nobody can
|
|
// re-cut retroactively - so the importer asks the artifact what it has,
|
|
// the same way it asks which encoding it uses.
|
|
func TestImportCoreArtifactReadsTotalsWhenPresent(t *testing.T) {
|
|
path := writeTestArtifact(t, validMeta(), []artifactRow{
|
|
{"release_group", rgA, "Big Album", "Solo Star", artA, 5000},
|
|
})
|
|
|
|
// The exporter writes the column; writeTestArtifact builds the older
|
|
// shape, so add it here rather than changing every other test's
|
|
// fixture to carry a value they do not use.
|
|
artifact, err := sql.Open("sqlite", "file:"+path)
|
|
if err != nil {
|
|
t.Fatalf("reopen artifact: %v", err)
|
|
}
|
|
|
|
for _, stmt := range []string{
|
|
`ALTER TABLE explore_index ADD COLUMN total_tracks INTEGER NOT NULL DEFAULT 0`,
|
|
`UPDATE explore_index SET total_tracks = 12`,
|
|
} {
|
|
if _, err := artifact.Exec(stmt); err != nil {
|
|
t.Fatalf("add total_tracks: %v", err)
|
|
}
|
|
}
|
|
|
|
_ = artifact.Close()
|
|
|
|
live := database.NewTestDB(t)
|
|
si := NewSearchIndex(live, nil, nil, testLogger())
|
|
|
|
if err := si.importCoreArtifact(context.Background(), path); err != nil {
|
|
t.Fatalf("importCoreArtifact: %v", err)
|
|
}
|
|
|
|
rg := si.LookupReleaseGroupByMBID(rgA)
|
|
if rg == nil {
|
|
t.Fatal("release group was not imported")
|
|
}
|
|
|
|
if rg.TotalTracks != 12 {
|
|
t.Errorf("TotalTracks = %d, want 12", rg.TotalTracks)
|
|
}
|
|
|
|
// And the shape that predates the column: the same import, from an
|
|
// artifact that has no total_tracks at all.
|
|
older := writeTestArtifact(t, validMeta(), []artifactRow{
|
|
{"release_group", rgB, "Duet Album", "Solo Star", artA, 4000},
|
|
})
|
|
|
|
live2 := database.NewTestDB(t)
|
|
si2 := NewSearchIndex(live2, nil, nil, testLogger())
|
|
|
|
if err := si2.importCoreArtifact(context.Background(), older); err != nil {
|
|
t.Fatalf("importCoreArtifact (no total_tracks column): %v", err)
|
|
}
|
|
|
|
old := si2.LookupReleaseGroupByMBID(rgB)
|
|
if old == nil {
|
|
t.Fatal("release group from a column-less artifact was not imported")
|
|
}
|
|
|
|
if old.TotalTracks != 0 {
|
|
t.Errorf("TotalTracks = %d, want 0 (the catalog does not say)", old.TotalTracks)
|
|
}
|
|
}
|
|
|
|
// addArtifactCredits gives an artifact file the credit tables the
|
|
// exporter now writes, so the import path can be exercised against one
|
|
// that has them.
|
|
func addArtifactCredits(t *testing.T, path string) {
|
|
t.Helper()
|
|
|
|
db, err := sql.Open("sqlite", "file:"+path)
|
|
if err != nil {
|
|
t.Fatalf("open artifact: %v", err)
|
|
}
|
|
|
|
defer func() { _ = db.Close() }()
|
|
|
|
for _, stmt := range []string{
|
|
`CREATE TABLE artist_credit_part (
|
|
credit_id INTEGER NOT NULL,
|
|
position INTEGER NOT NULL,
|
|
artist_mbid BLOB NOT NULL,
|
|
credited_name TEXT NOT NULL,
|
|
join_phrase TEXT NOT NULL DEFAULT '',
|
|
PRIMARY KEY (credit_id, position)
|
|
) WITHOUT ROWID`,
|
|
`CREATE TABLE artist_credit_ref (
|
|
mbid BLOB NOT NULL PRIMARY KEY,
|
|
credit_id INTEGER NOT NULL
|
|
) WITHOUT ROWID`,
|
|
} {
|
|
if _, err := db.Exec(stmt); err != nil {
|
|
t.Fatalf("create credit tables: %v", err)
|
|
}
|
|
}
|
|
|
|
// The packed form the catalog stores. uuid16/parseUUID live behind
|
|
// the indexbuild tag, so this file decodes for itself.
|
|
pack := func(mbid string) []byte {
|
|
raw, err := hex.DecodeString(strings.ReplaceAll(mbid, "-", ""))
|
|
if err != nil || len(raw) != 16 {
|
|
t.Fatalf("fixture MBID %q is not a UUID: %v", mbid, err)
|
|
}
|
|
|
|
return raw
|
|
}
|
|
|
|
a, b, rec := pack(artA), pack(artB), pack(recA)
|
|
|
|
for _, part := range [][]any{
|
|
{7, 0, a, "Artist A", " feat. "},
|
|
{7, 1, b, "Artist B", ""},
|
|
} {
|
|
if _, err := db.Exec(`INSERT INTO artist_credit_part
|
|
(credit_id, position, artist_mbid, credited_name, join_phrase)
|
|
VALUES (?, ?, ?, ?, ?)`, part...); err != nil {
|
|
t.Fatalf("insert part: %v", err)
|
|
}
|
|
}
|
|
|
|
if _, err := db.Exec(
|
|
"INSERT INTO artist_credit_ref (mbid, credit_id) VALUES (?, ?)", rec, 7,
|
|
); err != nil {
|
|
t.Fatalf("insert ref: %v", err)
|
|
}
|
|
}
|
|
|
|
// TestImportCoreArtifactMergesCredits is the positive half of the
|
|
// compatibility pair: an artifact that carries credits delivers them,
|
|
// rendering back to the credit string they decompose.
|
|
func TestImportCoreArtifactMergesCredits(t *testing.T) {
|
|
db := database.NewTestDB(t)
|
|
si := NewSearchIndex(db, nil, nil, testLogger())
|
|
|
|
path := writeTestArtifact(t, validMeta(), []artifactRow{
|
|
{"recording", recA, "Song A", "Artist A feat. Artist B", artA, 2000},
|
|
})
|
|
|
|
addArtifactCredits(t, path)
|
|
|
|
if err := si.importCoreArtifact(context.Background(), path); err != nil {
|
|
t.Fatalf("importCoreArtifact: %v", err)
|
|
}
|
|
|
|
rows, err := db.QueryContext(
|
|
`SELECT p.credited_name, p.join_phrase
|
|
FROM artist_credit_ref r
|
|
JOIN artist_credit_part p ON p.credit_id = r.credit_id
|
|
ORDER BY p.position`,
|
|
)
|
|
if err != nil {
|
|
t.Fatalf("query credits: %v", err)
|
|
}
|
|
|
|
defer func() { _ = rows.Close() }()
|
|
|
|
var rendered strings.Builder
|
|
|
|
for rows.Next() {
|
|
var name, join string
|
|
|
|
if err := rows.Scan(&name, &join); err != nil {
|
|
t.Fatalf("scan: %v", err)
|
|
}
|
|
|
|
rendered.WriteString(name)
|
|
rendered.WriteString(join)
|
|
}
|
|
|
|
if got := rendered.String(); got != "Artist A feat. Artist B" {
|
|
t.Errorf("rendered credit = %q, want %q", got, "Artist A feat. Artist B")
|
|
}
|
|
}
|
|
|
|
// TestImportCoreArtifactWithoutCredits is the regression that matters
|
|
// most here: an artifact published before credits existed cannot be
|
|
// re-cut retroactively, so it must import as a catalog that declines to
|
|
// answer rather than failing outright. writeTestArtifact deliberately
|
|
// builds one without the tables.
|
|
func TestImportCoreArtifactWithoutCredits(t *testing.T) {
|
|
db := database.NewTestDB(t)
|
|
si := NewSearchIndex(db, nil, nil, testLogger())
|
|
|
|
path := writeTestArtifact(t, validMeta(), []artifactRow{
|
|
{"recording", recA, "Song A", "Artist A", artA, 2000},
|
|
})
|
|
|
|
if err := si.importCoreArtifact(context.Background(), path); err != nil {
|
|
t.Fatalf("an artifact without credit tables must still import: %v", err)
|
|
}
|
|
|
|
var rows int
|
|
if err := db.QueryRowWriter(
|
|
"SELECT COUNT(*) FROM explore_index",
|
|
).Scan(&rows); err != nil {
|
|
t.Fatalf("count: %v", err)
|
|
}
|
|
|
|
if rows != 1 {
|
|
t.Errorf("catalog rows = %d, want 1", rows)
|
|
}
|
|
|
|
var refs int
|
|
if err := db.QueryRowWriter(
|
|
"SELECT COUNT(*) FROM artist_credit_ref",
|
|
).Scan(&refs); err != nil {
|
|
t.Fatalf("count refs: %v", err)
|
|
}
|
|
|
|
if refs != 0 {
|
|
t.Errorf("credit refs = %d, want 0", refs)
|
|
}
|
|
}
|
|
|
|
// TestImportCoreArtifactRefusesAMergeThatLosesRows is the count guard's
|
|
// positive case.
|
|
//
|
|
// The walk's predicates partition the artifact's key space, so a merge
|
|
// that lands fewer rows than the artifact declares means a predicate
|
|
// dropped some — and the failure is a catalog that looks populated and
|
|
// is missing things nobody can name. An empty mbid is the reachable
|
|
// way to get there: `mbid > ?` is false of it in both encodings, so it
|
|
// is never selected, and nothing else in the import would notice.
|
|
func TestImportCoreArtifactRefusesAMergeThatLosesRows(t *testing.T) {
|
|
db := database.NewTestDB(t)
|
|
si := NewSearchIndex(db, nil, nil, testLogger())
|
|
|
|
path := writeCompactTestArtifact(t, validMeta(), []artifactRow{
|
|
{EntityArtist, artA, "Artist A", "Artist A", artA, 5000},
|
|
{EntityArtist, "", "Nameless", "Artist A", artA, 4000},
|
|
})
|
|
|
|
err := si.importCoreArtifact(context.Background(), path)
|
|
if err == nil {
|
|
t.Fatal("a merge that lost a row was reported as a complete import")
|
|
}
|
|
|
|
if !strings.Contains(err.Error(), "merged 1 of 2 rows") {
|
|
t.Errorf("error = %v, want it to name the shortfall", err)
|
|
}
|
|
|
|
// And the same rule as every other rejection: a failed merge must not
|
|
// leave the index claiming it has a catalog, or the real build would
|
|
// never run again.
|
|
if si.hasMeta(dumpImportDoneKey) {
|
|
t.Error("a failed import still stamped dump_import_done")
|
|
}
|
|
}
|
|
|
|
// TestArtifactKeyBindsInTheArtifactsOwnEncoding pins the one place the
|
|
// batch walk's comparison type is decided.
|
|
//
|
|
// Every wrong answer is silent, which is why it is worth pinning all
|
|
// four. SQLite does not coerce TEXT to BLOB and orders every blob after
|
|
// every text value, so a text key against a byte column makes
|
|
// `mbid > ?` true of the whole artifact - the cursor never advances and
|
|
// the walk spins forever without merging a row - while a byte key
|
|
// against a text column makes it false of the whole artifact, so every
|
|
// batch merges nothing and the import "succeeds" empty. An unset cursor
|
|
// is the same fault once more: database/sql converts a nil []byte to
|
|
// SQL NULL, and `mbid > NULL` matches no row at all.
|
|
func TestArtifactKeyBindsInTheArtifactsOwnEncoding(t *testing.T) {
|
|
raw := mbidBytes(artA)
|
|
|
|
for _, tt := range []struct {
|
|
name string
|
|
key artifactKey
|
|
want []byte
|
|
}{
|
|
{"unset", nil, []byte{}},
|
|
{"set", artifactKey(raw), raw},
|
|
} {
|
|
t.Run("bytes/"+tt.name, func(t *testing.T) {
|
|
got, ok := tt.key.bind(false).([]byte)
|
|
if !ok {
|
|
t.Fatalf("bind(false) = %T, want []byte", tt.key.bind(false))
|
|
}
|
|
|
|
if got == nil {
|
|
t.Fatal("bound to SQL NULL, which matches no row")
|
|
}
|
|
|
|
if !bytes.Equal(got, tt.want) {
|
|
t.Errorf("bind(false) = %x, want %x", got, tt.want)
|
|
}
|
|
})
|
|
}
|
|
|
|
// The dashed form is what an artifact published before the storage
|
|
// change carries, and it has to compare as text against text.
|
|
if got := artifactKey(nil).bind(true); got != "" {
|
|
t.Errorf("bind(true) on an unset cursor = %#v, want an empty string", got)
|
|
}
|
|
|
|
if got := artifactKey(artA).bind(true); got != artA {
|
|
t.Errorf("bind(true) = %#v, want %q", got, artA)
|
|
}
|
|
}
|