The prebuilt catalog never merged. `mergeArtifactRows` positions itself with `WHERE mbid > ? ORDER BY mbid LIMIT 1 OFFSET ?` against the attached artifact, and it bound that cursor as a Go `string` while `cmd/indexexport` publishes `explore_index.mbid` as 16 raw bytes — the storage change that took the table from 677 MB to 389 MB. SQLite does not coerce between TEXT and BLOB and orders every blob after every text value, so against a byte column the predicate was not wrong but unconditional: `mbid > <text>` matched the whole artifact, so the bound the walk looked up was the same row every time and the cursor never advanced, and `mbid <= <text>` matched nothing, so no batch merged. No error, no rows, no state change — a fresh install sat at "0 of 1,077,893 rows" burning a core indefinitely, which is what it did here for a day, while Explore showed only the rows the library scan and the lazy artist enrichment had produced and popularity for none of the catalog. The cursor is now an `artifactKey`, typed to the encoding `artifactStoresText` reports for the file it is attached to, so the comparison is made in the same type as the column it is made against. Two things guard the class rather than the instance: a nil key binds as an empty value instead of SQL NULL, because `mbid > NULL` agrees with nothing and would import nothing just as silently; and the walk returns an error when its bound does not strictly advance, because the failure here is silence and the next one should be a failed job with a reason. It was never caught because the fixture that guards the walk writes the old text encoding, and the only compact fixture is a single row — below `artifactMergeBatch`, so the bound query never ran at all. The walk is now covered on both encodings, across several batch boundaries. Closes #258
904 lines
26 KiB
Go
904 lines
26 KiB
Go
package explore
|
|
|
|
import (
|
|
"bytes"
|
|
"context"
|
|
"database/sql"
|
|
"encoding/hex"
|
|
"os"
|
|
"path/filepath"
|
|
"strings"
|
|
"testing"
|
|
|
|
"yellowjacket/backend/database"
|
|
)
|
|
|
|
// artifactRow is one catalog row written into a test artifact.
|
|
type artifactRow struct {
|
|
entityType string
|
|
mbid string
|
|
title string
|
|
artistName string
|
|
artistMBID string
|
|
popularity int
|
|
}
|
|
|
|
// writeTestArtifact builds an artifact file matching what cmd/indexexport
|
|
// produces: catalog columns only, no FTS, no triggers.
|
|
func writeTestArtifact(
|
|
t *testing.T, meta map[string]string, rows []artifactRow,
|
|
) string {
|
|
t.Helper()
|
|
|
|
path := filepath.Join(t.TempDir(), "core-index.db")
|
|
|
|
db, err := sql.Open("sqlite", "file:"+path)
|
|
if err != nil {
|
|
t.Fatalf("open artifact: %v", err)
|
|
}
|
|
|
|
defer func() { _ = db.Close() }()
|
|
|
|
for _, stmt := range []string{
|
|
`CREATE TABLE explore_index (
|
|
entity_type TEXT NOT NULL,
|
|
mbid TEXT NOT NULL,
|
|
title TEXT NOT NULL,
|
|
artist_name TEXT NOT NULL,
|
|
artist_mbid TEXT NOT NULL,
|
|
aliases TEXT NOT NULL DEFAULT '',
|
|
popularity INTEGER NOT NULL DEFAULT 0,
|
|
listener_count INTEGER NOT NULL DEFAULT 0,
|
|
duration INTEGER NOT NULL DEFAULT 0,
|
|
caa_release_mbid TEXT NOT NULL DEFAULT '',
|
|
release_name TEXT NOT NULL DEFAULT '',
|
|
primary_type TEXT NOT NULL DEFAULT '',
|
|
secondary_types TEXT NOT NULL DEFAULT '',
|
|
release_date TEXT NOT NULL DEFAULT '',
|
|
artist_type TEXT NOT NULL DEFAULT '',
|
|
country TEXT NOT NULL DEFAULT '',
|
|
disambiguation TEXT NOT NULL DEFAULT '',
|
|
sort_name TEXT NOT NULL DEFAULT '',
|
|
discog_fetched INTEGER NOT NULL DEFAULT 0,
|
|
PRIMARY KEY (mbid)
|
|
) WITHOUT ROWID`,
|
|
`CREATE TABLE artifact_meta (
|
|
key TEXT PRIMARY KEY,
|
|
value TEXT NOT NULL
|
|
)`,
|
|
} {
|
|
if _, err := db.Exec(stmt); err != nil {
|
|
t.Fatalf("create artifact schema: %v", err)
|
|
}
|
|
}
|
|
|
|
stampArtifactMeta(t, db, meta)
|
|
|
|
for _, r := range rows {
|
|
if _, err := db.Exec(`
|
|
INSERT INTO explore_index
|
|
(entity_type, mbid, title, artist_name, artist_mbid, popularity)
|
|
VALUES (?, ?, ?, ?, ?, ?)`,
|
|
r.entityType, r.mbid, r.title, r.artistName, r.artistMBID, r.popularity,
|
|
); err != nil {
|
|
t.Fatalf("insert artifact row: %v", err)
|
|
}
|
|
}
|
|
|
|
return path
|
|
}
|
|
|
|
// compactArtifactSchema is the artifact cmd/indexexport publishes: the
|
|
// catalog's ids as 16 raw bytes, its entity types as codes, and the
|
|
// per-release-group total_tracks the exporter added after the first
|
|
// artifact was shipped.
|
|
//
|
|
// It matters that a fixture carries this encoding and not the older
|
|
// text one, because SQLite does not coerce between TEXT and BLOB and
|
|
// every comparison the importer makes against an mbid is therefore
|
|
// encoding-sensitive. writeTestArtifact above is the *other* fixture:
|
|
// it still writes the text form, which is what the first published
|
|
// artifact carries and what the importer must keep reading.
|
|
var compactArtifactSchema = []string{
|
|
`CREATE TABLE explore_index (
|
|
entity_type INTEGER NOT NULL,
|
|
mbid BLOB NOT NULL,
|
|
title TEXT NOT NULL,
|
|
artist_name TEXT NOT NULL,
|
|
artist_mbid BLOB NOT NULL,
|
|
aliases TEXT NOT NULL DEFAULT '',
|
|
popularity INTEGER NOT NULL DEFAULT 0,
|
|
listener_count INTEGER NOT NULL DEFAULT 0,
|
|
duration INTEGER NOT NULL DEFAULT 0,
|
|
caa_release_mbid BLOB NOT NULL DEFAULT x'',
|
|
release_name TEXT NOT NULL DEFAULT '',
|
|
primary_type TEXT NOT NULL DEFAULT '',
|
|
secondary_types TEXT NOT NULL DEFAULT '',
|
|
release_date TEXT NOT NULL DEFAULT '',
|
|
total_tracks INTEGER NOT NULL DEFAULT 0,
|
|
artist_type TEXT NOT NULL DEFAULT '',
|
|
country TEXT NOT NULL DEFAULT '',
|
|
disambiguation TEXT NOT NULL DEFAULT '',
|
|
sort_name TEXT NOT NULL DEFAULT '',
|
|
discog_fetched INTEGER NOT NULL DEFAULT 0,
|
|
PRIMARY KEY (mbid)
|
|
) WITHOUT ROWID`,
|
|
`CREATE TABLE artifact_meta (
|
|
key TEXT PRIMARY KEY,
|
|
value TEXT NOT NULL
|
|
)`,
|
|
}
|
|
|
|
// writeCompactTestArtifact builds the artifact the exporter publishes
|
|
// today, in its own encoding, so the importer is exercised against what
|
|
// a client actually downloads rather than against what it was written
|
|
// for.
|
|
func writeCompactTestArtifact(
|
|
t *testing.T, meta map[string]string, rows []artifactRow,
|
|
) string {
|
|
t.Helper()
|
|
|
|
path := filepath.Join(t.TempDir(), "core-index.db")
|
|
|
|
db, err := sql.Open("sqlite", "file:"+path)
|
|
if err != nil {
|
|
t.Fatalf("open artifact: %v", err)
|
|
}
|
|
|
|
defer func() { _ = db.Close() }()
|
|
|
|
for _, stmt := range compactArtifactSchema {
|
|
if _, err := db.Exec(stmt); err != nil {
|
|
t.Fatalf("create artifact schema: %v", err)
|
|
}
|
|
}
|
|
|
|
stampArtifactMeta(t, db, meta)
|
|
|
|
for _, r := range rows {
|
|
if _, err := db.Exec(`
|
|
INSERT INTO explore_index
|
|
(entity_type, mbid, title, artist_name, artist_mbid, popularity)
|
|
VALUES (?, ?, ?, ?, ?, ?)`,
|
|
entityCode(r.entityType), mbidBytes(r.mbid), r.title,
|
|
r.artistName, mbidBytes(r.artistMBID), r.popularity,
|
|
); err != nil {
|
|
t.Fatalf("insert artifact row: %v", err)
|
|
}
|
|
}
|
|
|
|
return path
|
|
}
|
|
|
|
// stampArtifactMeta writes the artifact_meta rows a fixture declares.
|
|
func stampArtifactMeta(t *testing.T, db *sql.DB, meta map[string]string) {
|
|
t.Helper()
|
|
|
|
for k, v := range meta {
|
|
if _, err := db.Exec(
|
|
`INSERT INTO artifact_meta (key, value) VALUES (?, ?)`, k, v,
|
|
); err != nil {
|
|
t.Fatalf("stamp artifact meta: %v", err)
|
|
}
|
|
}
|
|
}
|
|
|
|
// validMeta is the artifact_meta a well-formed artifact carries.
|
|
func validMeta() map[string]string {
|
|
return map[string]string{
|
|
"artifact_version": supportedArtifactVersion,
|
|
"built_at": "2026-07-17T03:53:43Z",
|
|
"listens_applied_series": "2593",
|
|
"source_rows": "2052168",
|
|
}
|
|
}
|
|
|
|
func TestImportCoreArtifactMergesCatalog(t *testing.T) {
|
|
db := database.NewTestDB(t)
|
|
si := NewSearchIndex(db, nil, nil, testLogger())
|
|
|
|
path := writeTestArtifact(t, validMeta(), []artifactRow{
|
|
{"artist", artA, "Artist A", "Artist A", artA, 5000},
|
|
{"artist", artB, "Artist B", "Artist B", artB, 4000},
|
|
{"release_group", rgA, "Album A", "Artist A", artA, 3000},
|
|
{"recording", recA, "Song A", "Artist A", artA, 2000},
|
|
})
|
|
|
|
if err := si.importCoreArtifact(context.Background(), path); err != nil {
|
|
t.Fatalf("importCoreArtifact: %v", err)
|
|
}
|
|
|
|
var got int
|
|
if err := db.QueryRowWriter(
|
|
"SELECT COUNT(*) FROM explore_index",
|
|
).Scan(&got); err != nil {
|
|
t.Fatalf("count rows: %v", err)
|
|
}
|
|
|
|
if got != 4 {
|
|
t.Errorf("merged %d rows, want 4", got)
|
|
}
|
|
|
|
// The merge must leave the index searchable: the FTS rebuild that
|
|
// closes the bulk-load window is the only thing populating it, since
|
|
// the triggers were dropped for the duration.
|
|
var hits int
|
|
if err := db.QueryRowWriter(
|
|
`SELECT COUNT(*) FROM explore_index_fts WHERE explore_index_fts MATCH ?`,
|
|
"Song",
|
|
).Scan(&hits); err != nil {
|
|
t.Fatalf("query fts: %v", err)
|
|
}
|
|
|
|
if hits == 0 {
|
|
t.Error("FTS index is empty after merge; search would return nothing")
|
|
}
|
|
}
|
|
|
|
func TestImportCoreArtifactStampsMeta(t *testing.T) {
|
|
db := database.NewTestDB(t)
|
|
si := NewSearchIndex(db, nil, nil, testLogger())
|
|
|
|
path := writeTestArtifact(t, validMeta(), []artifactRow{
|
|
{"artist", artA, "Artist A", "Artist A", artA, 5000},
|
|
})
|
|
|
|
if err := si.importCoreArtifact(context.Background(), path); err != nil {
|
|
t.Fatalf("importCoreArtifact: %v", err)
|
|
}
|
|
|
|
if !si.hasMeta(dumpImportDoneKey) {
|
|
t.Error("dump_import_done not stamped; a full dump import would run anyway")
|
|
}
|
|
|
|
// Without this the incremental refresh refuses to run and the
|
|
// shipped popularity never updates again.
|
|
if series, ok := si.metaInt(listensAppliedSeriesKey); !ok || series != 2593 {
|
|
t.Errorf("listens_applied_series = %d (ok=%v), want 2593", series, ok)
|
|
}
|
|
|
|
if !si.hasMeta(coreArtifactVersionKey) {
|
|
t.Error("core_artifact_version not stamped")
|
|
}
|
|
}
|
|
|
|
// A merge must never downgrade what the index already holds, because a
|
|
// user's own library rows and any lazily-fetched detail predate it.
|
|
func TestImportCoreArtifactPreservesLocalData(t *testing.T) {
|
|
db := database.NewTestDB(t)
|
|
si := NewSearchIndex(db, nil, nil, testLogger())
|
|
|
|
si.upsertBatch([]SearchIndexResult{{
|
|
EntityType: "recording",
|
|
MBID: recA,
|
|
Title: "Song A",
|
|
ArtistName: "Artist A",
|
|
ArtistMBID: artA,
|
|
Popularity: 9999,
|
|
Duration: 210000,
|
|
InLibrary: true,
|
|
DiscogFetched: true,
|
|
}})
|
|
|
|
path := writeTestArtifact(t, validMeta(), []artifactRow{
|
|
// Lower popularity and no duration: both must lose.
|
|
{"recording", recA, "Song A", "Artist A", artA, 10},
|
|
})
|
|
|
|
if err := si.importCoreArtifact(context.Background(), path); err != nil {
|
|
t.Fatalf("importCoreArtifact: %v", err)
|
|
}
|
|
|
|
var popularity, duration, inLibrary, discogFetched int
|
|
|
|
if err := db.QueryRowWriter(`
|
|
SELECT popularity, duration, in_library, discog_fetched
|
|
FROM explore_index WHERE mbid = ?`, dbMBID(recA),
|
|
).Scan(&popularity, &duration, &inLibrary, &discogFetched); err != nil {
|
|
t.Fatalf("read merged row: %v", err)
|
|
}
|
|
|
|
if popularity != 9999 {
|
|
t.Errorf("popularity = %d, want 9999 (higher must win)", popularity)
|
|
}
|
|
|
|
if duration != 210000 {
|
|
t.Errorf("duration = %d, want 210000 (artifact carries none)", duration)
|
|
}
|
|
|
|
if inLibrary != 1 {
|
|
t.Error("in_library was cleared; the artifact must not touch personal columns")
|
|
}
|
|
|
|
if discogFetched != 1 {
|
|
t.Error("discog_fetched was cleared by the merge")
|
|
}
|
|
}
|
|
|
|
// The batch walk is the part most likely to drop or duplicate rows, so
|
|
// it is exercised across many batch boundaries rather than one.
|
|
func TestImportCoreArtifactBatchWalkCoversAllRows(t *testing.T) {
|
|
db := database.NewTestDB(t)
|
|
si := NewSearchIndex(db, nil, nil, testLogger())
|
|
|
|
// Shrink the batch rather than growing the fixture: what matters is
|
|
// crossing several boundaries, including a short final one.
|
|
original := artifactMergeBatch
|
|
artifactMergeBatch = 100
|
|
|
|
t.Cleanup(func() { artifactMergeBatch = original })
|
|
|
|
const total = 337
|
|
|
|
rows := make([]artifactRow, 0, total)
|
|
for i := range total {
|
|
rows = append(rows, artifactRow{
|
|
entityType: "recording",
|
|
mbid: syntheticMBID(i),
|
|
title: "Song",
|
|
artistName: "Artist",
|
|
artistMBID: artA,
|
|
popularity: i,
|
|
})
|
|
}
|
|
|
|
path := writeTestArtifact(t, validMeta(), rows)
|
|
|
|
if err := si.importCoreArtifact(context.Background(), path); err != nil {
|
|
t.Fatalf("importCoreArtifact: %v", err)
|
|
}
|
|
|
|
var got int
|
|
if err := db.QueryRowWriter(
|
|
"SELECT COUNT(*) FROM explore_index",
|
|
).Scan(&got); err != nil {
|
|
t.Fatalf("count rows: %v", err)
|
|
}
|
|
|
|
if got != total {
|
|
t.Errorf("merged %d rows, want %d", got, total)
|
|
}
|
|
}
|
|
|
|
// TestImportCoreArtifactBatchWalkCoversAllRowsCompact is the batch walk
|
|
// on the encoding the exporter actually publishes.
|
|
//
|
|
// The walk positions itself by comparing the artifact's own mbid column
|
|
// against the last id it reached, and that column holds 16 raw bytes.
|
|
// SQLite does not coerce between TEXT and BLOB, and a blob sorts after
|
|
// every text value, so a cursor bound as text is a predicate that either
|
|
// matches every row or none: `mbid > ?` with an empty text key is true
|
|
// of the whole table, so
|
|
// the 100th row is always the 100th row and the bound never advances,
|
|
// while `mbid <= <text>` is false of the whole table, so no batch ever
|
|
// merges. The result is not a wrong import but an unbounded loop that
|
|
// merges nothing and never fails.
|
|
//
|
|
// Both encodings are covered on purpose. The walk was only ever tested
|
|
// against the text fixture above, which is why it shipped broken on the
|
|
// one the clients download.
|
|
func TestImportCoreArtifactBatchWalkCoversAllRowsCompact(t *testing.T) {
|
|
db := database.NewTestDB(t)
|
|
si := NewSearchIndex(db, nil, nil, testLogger())
|
|
|
|
original := artifactMergeBatch
|
|
artifactMergeBatch = 100
|
|
|
|
t.Cleanup(func() { artifactMergeBatch = original })
|
|
|
|
const total = 337
|
|
|
|
rows := make([]artifactRow, 0, total)
|
|
for i := range total {
|
|
rows = append(rows, artifactRow{
|
|
entityType: EntityRecording,
|
|
mbid: syntheticMBID(i),
|
|
title: "Song",
|
|
artistName: "Artist",
|
|
artistMBID: artA,
|
|
popularity: i,
|
|
})
|
|
}
|
|
|
|
path := writeCompactTestArtifact(t, validMeta(), rows)
|
|
|
|
if err := si.importCoreArtifact(context.Background(), path); err != nil {
|
|
t.Fatalf("importCoreArtifact: %v", err)
|
|
}
|
|
|
|
var got, top int
|
|
|
|
if err := db.QueryRowWriter(
|
|
"SELECT COUNT(*), MAX(popularity) FROM explore_index",
|
|
).Scan(&got, &top); err != nil {
|
|
t.Fatalf("count rows: %v", err)
|
|
}
|
|
|
|
if got != total {
|
|
t.Errorf("merged %d rows, want %d", got, total)
|
|
}
|
|
|
|
// A count alone would pass if the walk re-merged the same first
|
|
// batch forever, so the far end of the artifact is checked too.
|
|
if top != total-1 {
|
|
t.Errorf("highest popularity = %d, want %d", top, total-1)
|
|
}
|
|
}
|
|
|
|
func TestImportCoreArtifactRejectsBadArtifacts(t *testing.T) {
|
|
tests := []struct {
|
|
name string
|
|
meta map[string]string
|
|
rows []artifactRow
|
|
want error
|
|
}{
|
|
{
|
|
name: "version mismatch",
|
|
meta: map[string]string{"artifact_version": "99"},
|
|
rows: []artifactRow{{"artist", artA, "A", "A", artA, 1}},
|
|
want: ErrArtifactVersion,
|
|
},
|
|
{
|
|
name: "no version",
|
|
meta: map[string]string{},
|
|
rows: []artifactRow{{"artist", artA, "A", "A", artA, 1}},
|
|
want: ErrArtifactVersion,
|
|
},
|
|
{
|
|
name: "empty catalog",
|
|
meta: validMeta(),
|
|
rows: nil,
|
|
want: ErrArtifactUnusable,
|
|
},
|
|
}
|
|
|
|
for _, tt := range tests {
|
|
t.Run(tt.name, func(t *testing.T) {
|
|
db := database.NewTestDB(t)
|
|
si := NewSearchIndex(db, nil, nil, testLogger())
|
|
|
|
path := writeTestArtifact(t, tt.meta, tt.rows)
|
|
|
|
err := si.importCoreArtifact(context.Background(), path)
|
|
if err == nil {
|
|
t.Fatal("expected rejection, got nil")
|
|
}
|
|
|
|
if !strings.Contains(err.Error(), tt.want.Error()) {
|
|
t.Errorf("error = %v, want it to wrap %v", err, tt.want)
|
|
}
|
|
|
|
// A rejected artifact must not leave the index claiming it
|
|
// has a catalog, or the real build would never run.
|
|
if si.hasMeta(dumpImportDoneKey) {
|
|
t.Error("rejected artifact still stamped dump_import_done")
|
|
}
|
|
})
|
|
}
|
|
}
|
|
|
|
func TestInspectArtifactRejectsNonArtifactFile(t *testing.T) {
|
|
path := filepath.Join(t.TempDir(), "garbage.db")
|
|
|
|
if err := os.WriteFile(path, []byte("not a database"), 0o644); err != nil {
|
|
t.Fatalf("write file: %v", err)
|
|
}
|
|
|
|
if _, err := inspectArtifact(path); err == nil {
|
|
t.Error("expected rejection of a non-artifact file")
|
|
}
|
|
}
|
|
|
|
// syntheticMBID produces distinct, well-formed MBIDs for bulk fixtures.
|
|
func syntheticMBID(i int) string {
|
|
const hex = "0123456789abcdef"
|
|
|
|
buf := []byte("00000000-0000-0000-0000-000000000000")
|
|
|
|
for pos := len(buf) - 1; pos >= 0 && i > 0; pos-- {
|
|
if buf[pos] == '-' {
|
|
continue
|
|
}
|
|
|
|
buf[pos] = hex[i%16]
|
|
i /= 16
|
|
}
|
|
|
|
return string(buf)
|
|
}
|
|
|
|
// resolveArtistName falls back to the MBID when no name is available.
|
|
// That value must never reach the index: the upsert no longer defends
|
|
// against it, so the writer is the only thing standing in the way.
|
|
func TestAddFromCacheNeverStoresMBIDAsName(t *testing.T) {
|
|
db := database.NewTestDB(t)
|
|
si := NewSearchIndex(db, nil, nil, testLogger())
|
|
|
|
si.AddFromCache(artA, artA, []MBReleaseGroup{
|
|
{MBID: rgA, Title: "Album A", PrimaryType: "Album"},
|
|
})
|
|
|
|
var artistRows int
|
|
if err := db.QueryRowWriter(
|
|
`SELECT COUNT(*) FROM explore_index
|
|
WHERE entity_type = 'artist' AND title = mbid`,
|
|
).Scan(&artistRows); err != nil {
|
|
t.Fatalf("count artist rows: %v", err)
|
|
}
|
|
|
|
if artistRows != 0 {
|
|
t.Errorf("%d artist rows stored the MBID as their title", artistRows)
|
|
}
|
|
|
|
var artistName string
|
|
if err := db.QueryRowWriter(
|
|
`SELECT artist_name FROM explore_index WHERE mbid = ?`, dbMBID(rgA),
|
|
).Scan(&artistName); err != nil {
|
|
t.Fatalf("read release group: %v", err)
|
|
}
|
|
|
|
if artistName == artA {
|
|
t.Error("release group stored the artist MBID as its artist_name")
|
|
}
|
|
|
|
// A real name arriving later must still win over the empty one.
|
|
si.AddFromCache("Real Artist", artA, []MBReleaseGroup{
|
|
{MBID: rgA, Title: "Album A", PrimaryType: "Album"},
|
|
})
|
|
|
|
if err := db.QueryRowWriter(
|
|
`SELECT artist_name FROM explore_index WHERE mbid = ?`, dbMBID(rgA),
|
|
).Scan(&artistName); err != nil {
|
|
t.Fatalf("re-read release group: %v", err)
|
|
}
|
|
|
|
if artistName != "Real Artist" {
|
|
t.Errorf("artist_name = %q, want it filled in once known", artistName)
|
|
}
|
|
}
|
|
|
|
// The importer names the artifact's columns independently of the
|
|
// exporter that writes them. If the two lists drift, the merge either
|
|
// fails outright or silently shifts values into the wrong columns, so
|
|
// they are compared directly against cmd/indexexport's source.
|
|
func TestArtifactColumnsMatchExporter(t *testing.T) {
|
|
src, err := os.ReadFile("../../cmd/indexexport/main.go")
|
|
if err != nil {
|
|
t.Fatalf("read exporter: %v", err)
|
|
}
|
|
|
|
const marker = "const catalogColumns = `"
|
|
|
|
i := strings.Index(string(src), marker)
|
|
if i < 0 {
|
|
t.Fatalf("catalogColumns not found in cmd/indexexport/main.go")
|
|
}
|
|
|
|
rest := string(src)[i+len(marker):]
|
|
|
|
j := strings.Index(rest, "`")
|
|
if j < 0 {
|
|
t.Fatal("unterminated catalogColumns literal")
|
|
}
|
|
|
|
normalise := func(s string) string {
|
|
out := make([]string, 0, 32)
|
|
for _, f := range strings.Split(s, ",") {
|
|
out = append(out, strings.Join(strings.Fields(f), ""))
|
|
}
|
|
|
|
return strings.Join(out, ",")
|
|
}
|
|
|
|
exporter := normalise(rest[:j])
|
|
importer := normalise(artifactCatalogColumns)
|
|
|
|
if exporter != importer {
|
|
t.Errorf("column lists have drifted:\n exporter: %s\n importer: %s",
|
|
exporter, importer)
|
|
}
|
|
}
|
|
|
|
// TestImportCoreArtifactAcceptsBothEncodings is the compatibility half
|
|
// of the storage change.
|
|
//
|
|
// The catalog stores an MBID as 16 raw bytes and an entity type as a
|
|
// code, which took the table and its indexes from 677 MB to 389 MB. A
|
|
// published artifact carries whichever form the exporter that built it
|
|
// used, and there is one already out there in the older text form — so
|
|
// the importer decides by asking the artifact, not by trusting a
|
|
// version number, and both must land identically.
|
|
func TestImportCoreArtifactAcceptsBothEncodings(t *testing.T) {
|
|
compact := writeCompactTestArtifact(t, validMeta(), []artifactRow{
|
|
{EntityArtist, artA, "Artist A", "Artist A", artA, 5000},
|
|
})
|
|
|
|
live := database.NewTestDB(t)
|
|
si := NewSearchIndex(live, nil, nil, testLogger())
|
|
|
|
if err := si.importCoreArtifact(context.Background(), compact); err != nil {
|
|
t.Fatalf("importCoreArtifact (compact): %v", err)
|
|
}
|
|
|
|
got := si.LookupArtistByMBID(artA)
|
|
if got == nil {
|
|
t.Fatal("artist from a compact artifact was not imported")
|
|
}
|
|
|
|
if got.Title != "Artist A" || got.MBID != artA {
|
|
t.Errorf("imported %+v, want Artist A / %s", got, artA)
|
|
}
|
|
}
|
|
|
|
// TestImportCoreArtifactReadsTotalsWhenPresent covers both halves of the
|
|
// denominator's arrival: an artifact that carries total_tracks imports
|
|
// it, and one built before the column existed still imports at all.
|
|
//
|
|
// The second half is the one worth a test. Adding a column to the
|
|
// importer's SELECT list is how you break every artifact already
|
|
// published - "no such column: total_tracks", on a file nobody can
|
|
// re-cut retroactively - so the importer asks the artifact what it has,
|
|
// the same way it asks which encoding it uses.
|
|
func TestImportCoreArtifactReadsTotalsWhenPresent(t *testing.T) {
|
|
path := writeTestArtifact(t, validMeta(), []artifactRow{
|
|
{"release_group", rgA, "Big Album", "Solo Star", artA, 5000},
|
|
})
|
|
|
|
// The exporter writes the column; writeTestArtifact builds the older
|
|
// shape, so add it here rather than changing every other test's
|
|
// fixture to carry a value they do not use.
|
|
artifact, err := sql.Open("sqlite", "file:"+path)
|
|
if err != nil {
|
|
t.Fatalf("reopen artifact: %v", err)
|
|
}
|
|
|
|
for _, stmt := range []string{
|
|
`ALTER TABLE explore_index ADD COLUMN total_tracks INTEGER NOT NULL DEFAULT 0`,
|
|
`UPDATE explore_index SET total_tracks = 12`,
|
|
} {
|
|
if _, err := artifact.Exec(stmt); err != nil {
|
|
t.Fatalf("add total_tracks: %v", err)
|
|
}
|
|
}
|
|
|
|
_ = artifact.Close()
|
|
|
|
live := database.NewTestDB(t)
|
|
si := NewSearchIndex(live, nil, nil, testLogger())
|
|
|
|
if err := si.importCoreArtifact(context.Background(), path); err != nil {
|
|
t.Fatalf("importCoreArtifact: %v", err)
|
|
}
|
|
|
|
rg := si.LookupReleaseGroupByMBID(rgA)
|
|
if rg == nil {
|
|
t.Fatal("release group was not imported")
|
|
}
|
|
|
|
if rg.TotalTracks != 12 {
|
|
t.Errorf("TotalTracks = %d, want 12", rg.TotalTracks)
|
|
}
|
|
|
|
// And the shape that predates the column: the same import, from an
|
|
// artifact that has no total_tracks at all.
|
|
older := writeTestArtifact(t, validMeta(), []artifactRow{
|
|
{"release_group", rgB, "Duet Album", "Solo Star", artA, 4000},
|
|
})
|
|
|
|
live2 := database.NewTestDB(t)
|
|
si2 := NewSearchIndex(live2, nil, nil, testLogger())
|
|
|
|
if err := si2.importCoreArtifact(context.Background(), older); err != nil {
|
|
t.Fatalf("importCoreArtifact (no total_tracks column): %v", err)
|
|
}
|
|
|
|
old := si2.LookupReleaseGroupByMBID(rgB)
|
|
if old == nil {
|
|
t.Fatal("release group from a column-less artifact was not imported")
|
|
}
|
|
|
|
if old.TotalTracks != 0 {
|
|
t.Errorf("TotalTracks = %d, want 0 (the catalog does not say)", old.TotalTracks)
|
|
}
|
|
}
|
|
|
|
// addArtifactCredits gives an artifact file the credit tables the
|
|
// exporter now writes, so the import path can be exercised against one
|
|
// that has them.
|
|
func addArtifactCredits(t *testing.T, path string) {
|
|
t.Helper()
|
|
|
|
db, err := sql.Open("sqlite", "file:"+path)
|
|
if err != nil {
|
|
t.Fatalf("open artifact: %v", err)
|
|
}
|
|
|
|
defer func() { _ = db.Close() }()
|
|
|
|
for _, stmt := range []string{
|
|
`CREATE TABLE artist_credit_part (
|
|
credit_id INTEGER NOT NULL,
|
|
position INTEGER NOT NULL,
|
|
artist_mbid BLOB NOT NULL,
|
|
credited_name TEXT NOT NULL,
|
|
join_phrase TEXT NOT NULL DEFAULT '',
|
|
PRIMARY KEY (credit_id, position)
|
|
) WITHOUT ROWID`,
|
|
`CREATE TABLE artist_credit_ref (
|
|
mbid BLOB NOT NULL PRIMARY KEY,
|
|
credit_id INTEGER NOT NULL
|
|
) WITHOUT ROWID`,
|
|
} {
|
|
if _, err := db.Exec(stmt); err != nil {
|
|
t.Fatalf("create credit tables: %v", err)
|
|
}
|
|
}
|
|
|
|
// The packed form the catalog stores. uuid16/parseUUID live behind
|
|
// the indexbuild tag, so this file decodes for itself.
|
|
pack := func(mbid string) []byte {
|
|
raw, err := hex.DecodeString(strings.ReplaceAll(mbid, "-", ""))
|
|
if err != nil || len(raw) != 16 {
|
|
t.Fatalf("fixture MBID %q is not a UUID: %v", mbid, err)
|
|
}
|
|
|
|
return raw
|
|
}
|
|
|
|
a, b, rec := pack(artA), pack(artB), pack(recA)
|
|
|
|
for _, part := range [][]any{
|
|
{7, 0, a, "Artist A", " feat. "},
|
|
{7, 1, b, "Artist B", ""},
|
|
} {
|
|
if _, err := db.Exec(`INSERT INTO artist_credit_part
|
|
(credit_id, position, artist_mbid, credited_name, join_phrase)
|
|
VALUES (?, ?, ?, ?, ?)`, part...); err != nil {
|
|
t.Fatalf("insert part: %v", err)
|
|
}
|
|
}
|
|
|
|
if _, err := db.Exec(
|
|
"INSERT INTO artist_credit_ref (mbid, credit_id) VALUES (?, ?)", rec, 7,
|
|
); err != nil {
|
|
t.Fatalf("insert ref: %v", err)
|
|
}
|
|
}
|
|
|
|
// TestImportCoreArtifactMergesCredits is the positive half of the
|
|
// compatibility pair: an artifact that carries credits delivers them,
|
|
// rendering back to the credit string they decompose.
|
|
func TestImportCoreArtifactMergesCredits(t *testing.T) {
|
|
db := database.NewTestDB(t)
|
|
si := NewSearchIndex(db, nil, nil, testLogger())
|
|
|
|
path := writeTestArtifact(t, validMeta(), []artifactRow{
|
|
{"recording", recA, "Song A", "Artist A feat. Artist B", artA, 2000},
|
|
})
|
|
|
|
addArtifactCredits(t, path)
|
|
|
|
if err := si.importCoreArtifact(context.Background(), path); err != nil {
|
|
t.Fatalf("importCoreArtifact: %v", err)
|
|
}
|
|
|
|
rows, err := db.QueryContext(
|
|
`SELECT p.credited_name, p.join_phrase
|
|
FROM artist_credit_ref r
|
|
JOIN artist_credit_part p ON p.credit_id = r.credit_id
|
|
ORDER BY p.position`,
|
|
)
|
|
if err != nil {
|
|
t.Fatalf("query credits: %v", err)
|
|
}
|
|
|
|
defer func() { _ = rows.Close() }()
|
|
|
|
var rendered strings.Builder
|
|
|
|
for rows.Next() {
|
|
var name, join string
|
|
|
|
if err := rows.Scan(&name, &join); err != nil {
|
|
t.Fatalf("scan: %v", err)
|
|
}
|
|
|
|
rendered.WriteString(name)
|
|
rendered.WriteString(join)
|
|
}
|
|
|
|
if got := rendered.String(); got != "Artist A feat. Artist B" {
|
|
t.Errorf("rendered credit = %q, want %q", got, "Artist A feat. Artist B")
|
|
}
|
|
}
|
|
|
|
// TestImportCoreArtifactWithoutCredits is the regression that matters
|
|
// most here: an artifact published before credits existed cannot be
|
|
// re-cut retroactively, so it must import as a catalog that declines to
|
|
// answer rather than failing outright. writeTestArtifact deliberately
|
|
// builds one without the tables.
|
|
func TestImportCoreArtifactWithoutCredits(t *testing.T) {
|
|
db := database.NewTestDB(t)
|
|
si := NewSearchIndex(db, nil, nil, testLogger())
|
|
|
|
path := writeTestArtifact(t, validMeta(), []artifactRow{
|
|
{"recording", recA, "Song A", "Artist A", artA, 2000},
|
|
})
|
|
|
|
if err := si.importCoreArtifact(context.Background(), path); err != nil {
|
|
t.Fatalf("an artifact without credit tables must still import: %v", err)
|
|
}
|
|
|
|
var rows int
|
|
if err := db.QueryRowWriter(
|
|
"SELECT COUNT(*) FROM explore_index",
|
|
).Scan(&rows); err != nil {
|
|
t.Fatalf("count: %v", err)
|
|
}
|
|
|
|
if rows != 1 {
|
|
t.Errorf("catalog rows = %d, want 1", rows)
|
|
}
|
|
|
|
var refs int
|
|
if err := db.QueryRowWriter(
|
|
"SELECT COUNT(*) FROM artist_credit_ref",
|
|
).Scan(&refs); err != nil {
|
|
t.Fatalf("count refs: %v", err)
|
|
}
|
|
|
|
if refs != 0 {
|
|
t.Errorf("credit refs = %d, want 0", refs)
|
|
}
|
|
}
|
|
|
|
// TestArtifactKeyBindsInTheArtifactsOwnEncoding pins the one place the
|
|
// batch walk's comparison type is decided.
|
|
//
|
|
// Every wrong answer is silent, which is why it is worth pinning all
|
|
// four. SQLite does not coerce TEXT to BLOB and orders every blob after
|
|
// every text value, so a text key against a byte column makes
|
|
// `mbid > ?` true of the whole artifact - the cursor never advances and
|
|
// the walk spins forever without merging a row - while a byte key
|
|
// against a text column makes it false of the whole artifact, so every
|
|
// batch merges nothing and the import "succeeds" empty. An unset cursor
|
|
// is the same fault once more: database/sql converts a nil []byte to
|
|
// SQL NULL, and `mbid > NULL` matches no row at all.
|
|
func TestArtifactKeyBindsInTheArtifactsOwnEncoding(t *testing.T) {
|
|
raw := mbidBytes(artA)
|
|
|
|
for _, tt := range []struct {
|
|
name string
|
|
key artifactKey
|
|
want []byte
|
|
}{
|
|
{"unset", nil, []byte{}},
|
|
{"set", artifactKey(raw), raw},
|
|
} {
|
|
t.Run("bytes/"+tt.name, func(t *testing.T) {
|
|
got, ok := tt.key.bind(false).([]byte)
|
|
if !ok {
|
|
t.Fatalf("bind(false) = %T, want []byte", tt.key.bind(false))
|
|
}
|
|
|
|
if got == nil {
|
|
t.Fatal("bound to SQL NULL, which matches no row")
|
|
}
|
|
|
|
if !bytes.Equal(got, tt.want) {
|
|
t.Errorf("bind(false) = %x, want %x", got, tt.want)
|
|
}
|
|
})
|
|
}
|
|
|
|
// The dashed form is what an artifact published before the storage
|
|
// change carries, and it has to compare as text against text.
|
|
if got := artifactKey(nil).bind(true); got != "" {
|
|
t.Errorf("bind(true) on an unset cursor = %#v, want an empty string", got)
|
|
}
|
|
|
|
if got := artifactKey(artA).bind(true); got != artA {
|
|
t.Errorf("bind(true) = %#v, want %q", got, artA)
|
|
}
|
|
}
|