Skip to content

Commit b08eacd

Browse files
committed
test_device_buffer: make OwnedBufferFreesOnDestruction robust
A single 1 MB cudaMalloc is not reliably visible in cudaMemGetInfo (coarse granularity / context reservation varies by machine), so the alloc-visibility assert flaked on CI. Test the real invariant instead: no monotonic memory growth across 128 owned decompress/destroy cycles.
1 parent 62001a2 commit b08eacd

1 file changed

Lines changed: 16 additions & 6 deletions

File tree

‎tests/pipeline/test_device_buffer.cpp‎

Lines changed: 16 additions & 6 deletions
Original file line numberDiff line numberDiff line change
@@ -129,7 +129,7 @@ TEST(DeviceBuffer, CompressIntoMatchesBorrowingForm) {
129129

130130
// ── DB4 ──────────────────────────────────────────────────────────────────────
131131
TEST(DeviceBuffer, OwnedBufferFreesOnDestruction) {
132-
constexpr size_t N = 1 << 18; // 1 MB, large enough to see in cudaMemGetInfo
132+
constexpr size_t N = 1 << 18; // 1 MB per decompress output
133133
const size_t in_bytes = N * sizeof(float);
134134

135135
auto h_in = make_random_floats(N, 13);
@@ -142,16 +142,26 @@ TEST(DeviceBuffer, OwnedBufferFreesOnDestruction) {
142142

143143
BorrowedDeviceBuffer comp = p->compress(ConstDeviceSpan(d_in.void_ptr(), in_bytes), stream);
144144

145+
// "Frees on destruction" is tested as "no monotonic growth across many cycles":
146+
// each iteration allocates a fresh owned buffer and destroys it at scope exit.
147+
// A single cudaMalloc is not reliably visible in cudaMemGetInfo (coarse
148+
// granularity / context reservation vary by machine), but a destructor that
149+
// failed to free would accumulate kIters * in_bytes and show clearly.
150+
constexpr int kIters = 128; // a leak here would be ~128 MB, far above noise
145151
const size_t before = free_device_bytes();
146-
{
152+
for (int i = 0; i < kIters; ++i) {
147153
OwnedDeviceBuffer dec = p->decompressOwned(comp.cspan(), stream);
148154
ASSERT_NE(dec.data(), nullptr);
149155
EXPECT_EQ(dec.bytes(), in_bytes);
150-
EXPECT_LT(free_device_bytes(), before); // allocation is real
151-
}
152-
// Destructor must have freed it — no cudaFree by the caller.
156+
} // dec destructs each iteration — must free
153157
FZ_TEST_CUDA(cudaDeviceSynchronize());
154-
EXPECT_GE(free_device_bytes(), before - (in_bytes / 2));
158+
const size_t after = free_device_bytes();
159+
160+
// Allow a few buffers of slack for pool/context noise; catch a real per-iter leak.
161+
const size_t slack = 8 * in_bytes;
162+
EXPECT_GE(after + slack, before)
163+
<< "owned decompress leaked across " << kIters << " cycles: before=" << before
164+
<< " after=" << after;
155165
}
156166

157167
// ── DB5 / DB6 ────────────────────────────────────────────────────────────────

0 commit comments

Comments
 (0)