Compare commits

..
Author SHA1 Message Date
logan a113b7bd62 fix(riff): grow a chunk buffer with what arrives
CI / check (push) Skipped
CI / e2e (push) Skipped
CI / check (pull_request) Successful in 6m23s
CI / e2e (pull_request) Successful in 10m8s
Parse sized its buffer from the chunk header, which is four bytes read
off the file, so a truncated or malformed WAV declaring a 4 GB data
chunk in a 2 kB file got 4 GB from the allocator before the read
discovered there was nothing to put in it. The error was always right;
the allocation happened first.

io.CopyN into a bytes.Buffer is what ID3Chunk beside it has done since
#104, and needs nothing new: the reader stays an io.Reader and the
buffer grows with what actually arrives.

The regression test measures rather than asserts the error, because the
error is identical on a build that allocates the gigabyte. Measured on
the pre-fix build: 1,073,750,920 bytes of TotalAlloc for a 42-byte
container whose data chunk claimed 1 GiB.

Closes #216
2026-08-26 06:35:04 -04:00
3 changed files with 55 additions and 14 deletions
+6 -3
View File
@@ -63,12 +63,15 @@ func Parse(r io.Reader) ([]Chunk, error) {
return nil, err
}
data := make([]byte, size)
if _, err := io.ReadFull(r, data); err != nil {
// Copied rather than allocated up front, as ID3Chunk does: the
// size is four bytes off the file, so a truncated one is free to
// declare a chunk larger than the whole of itself.
var data bytes.Buffer
if _, err := io.CopyN(&data, r, int64(size)); err != nil {
return nil, fmt.Errorf("read chunk data for %q: %w", id, err)
}
chunks = append(chunks, Chunk{ID: id, Data: data})
chunks = append(chunks, Chunk{ID: id, Data: data.Bytes()})
// Odd-length chunks have a padding byte. Lenient: if the
// read fails (e.g. EOF), just break rather than error.
+43
View File
@@ -4,6 +4,7 @@ import (
"bytes"
"encoding/binary"
"errors"
"runtime"
"testing"
"yellowjacket/backend/riff"
@@ -209,3 +210,45 @@ func TestParse_ReadsEveryChunkInOrder(t *testing.T) {
t.Errorf("odd chunk data: got %q, want %q", chunks[1].Data, "INFOodd")
}
}
// A chunk size is four bytes read off the file, so a truncated or
// malformed WAV is free to declare a chunk larger than the whole of
// itself. Parse must grow with what arrives rather than with what was
// claimed.
//
// This measures the allocation instead of the error because the error
// is the same either way: a build sizing its buffer from the header
// reports the truncation correctly, having asked the allocator for a
// gigabyte on the way. Deliberately not parallel — TotalAlloc is
// process-wide, and a test paused beside another one is measuring it
// too.
func TestParse_DoesNotAllocateWhatAChunkClaims(t *testing.T) {
// Large enough that a header-sized buffer is unmistakable, in a
// container of a few dozen bytes.
const declared = 1 << 30
var raw bytes.Buffer
raw.WriteString("RIFF")
_ = binary.Write(&raw, binary.LittleEndian, uint32(declared+12))
raw.WriteString("WAVE")
raw.WriteString("data")
_ = binary.Write(&raw, binary.LittleEndian, uint32(declared))
raw.WriteString("and then the file ends")
var before, after runtime.MemStats
runtime.GC()
runtime.ReadMemStats(&before)
if _, err := riff.Parse(bytes.NewReader(raw.Bytes())); err == nil {
t.Fatal("Parse: got nil error for a chunk larger than the file holding it")
}
runtime.ReadMemStats(&after)
if grew := after.TotalAlloc - before.TotalAlloc; grew > 1<<20 {
t.Errorf("Parse allocated %d bytes reading a %d-byte file whose chunk header claimed %d",
grew, raw.Len(), declared)
}
}
+6 -11
View File
@@ -103,17 +103,12 @@ async function queueSixAndOpen(app: Page): Promise<void> {
*
* `explore-link` routes a track name to its *album's* page, so a
* track with no album renders a name that navigates nowhere — and
* the fixture library deliberately contains two,
* `unsorted/no-tags-at-all.mp3` and `unsorted/title-only.mp3`.
* (It contained four until #104: the two WAVs under `Field
* Recordings/Test Tones` had been tagged on disk all along and
* scan in with their album now, so they are ordinary tracks and
* not examples of this.) Which tracks arrive first is
* `audio_files.id` order, i.e. the order the **scan** inserted
* them, which depends on concurrency and directory traversal:
* locally the first eight all had albums and the spec passed twice
* over, and CI rebuilds its seed with a real scan and got a
* different eight.
* the fixture library deliberately contains two (`01 Tone A`,
* `02 Tone B`). Which tracks arrive first is `audio_files.id`
* order, i.e. the order the **scan** inserted them, which depends
* on concurrency and directory traversal: locally the first eight
* all had albums and the spec passed twice over, and CI rebuilds
* its seed with a real scan and got a different eight.
*
* Asking for what the test needs is the fix. It is not a
* narrowing: every assertion here wants an ordinary track, and