ModelMaskGPUTest: upload on the mask's stream
cudaMemcpy from pageable memory can return before the DMA lands, and the legacy stream it runs on does not order a non-blocking stream, so RemoveIslands could read a partly uploaded mask. Both uploads are now cudaMemcpyAsync on the test's stream. Co-Authored-By: Claude Opus 5.5 (1M context) <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_01D1G8gJVAy6gp1K5Dz3NE5C
This commit is contained in:
@@ -143,7 +143,7 @@ TEST_CASE("ModelMaskGPU_MatchesGemmi", "[ModelValidation][gpu]") {
|
||||
const std::vector<ModelMaskAtom> atoms = MaskAtoms(st);
|
||||
const std::vector<ModelMaskOp> ops = MaskOps(*sg);
|
||||
CudaDevicePtr<ModelMaskAtom> atoms_d(atoms.size());
|
||||
REQUIRE(cudaMemcpy(atoms_d, atoms.data(), atoms.size() * sizeof(ModelMaskAtom), cudaMemcpyHostToDevice) == cudaSuccess);
|
||||
REQUIRE(cudaMemcpyAsync(atoms_d, atoms.data(), atoms.size() * sizeof(ModelMaskAtom), cudaMemcpyHostToDevice, stream) == cudaSuccess);
|
||||
|
||||
for (double d_min : {6.0, 3.5}) {
|
||||
const gemmi::SolventMasker masker(gemmi::AtomicRadiiSet::Refmac);
|
||||
@@ -164,7 +164,7 @@ TEST_CASE("ModelMaskGPU_MatchesGemmi", "[ModelValidation][gpu]") {
|
||||
std::vector<float> out(n), first;
|
||||
|
||||
// The island step alone, on gemmi's pre-island mask: exact.
|
||||
REQUIRE(cudaMemcpy(mask_d, pre.data.data(), n * sizeof(float), cudaMemcpyHostToDevice) == cudaSuccess);
|
||||
REQUIRE(cudaMemcpyAsync(mask_d, pre.data.data(), n * sizeof(float), cudaMemcpyHostToDevice, stream) == cudaSuccess);
|
||||
mask.RemoveIslands(mask_d);
|
||||
REQUIRE(cudaMemcpyAsync(out.data(), mask_d, n * sizeof(float), cudaMemcpyDeviceToHost, stream) == cudaSuccess);
|
||||
REQUIRE(cudaStreamSynchronize(stream) == cudaSuccess);
|
||||
|
||||
Reference in New Issue
Block a user