fix vulkan synchronization bugs that affect unified memory platforms

This commit is contained in:
Harry Chen
2026-08-31 23:26:52 -04:00
parent 41f7fff5f2
commit 5201973dae
3 changed files with 22 additions and 7 deletions
+4 -7
View File
@@ -623,12 +623,9 @@ uint64_t Context::submit(VkCommandBuffer cb) {
bool Context::wait(uint64_t value) {
if (value == 0) return true;
// Poll the counter first (the timeline analog of vkGetFenceStatus,
// measurably cheaper than the blocking vkWaitSemaphores path on desktop
// drivers). The spin is bounded so a stuck device (device fault) ends up
// parked in the blocking wait instead of burning a core; CPU devices
// (llvmpipe) skip it entirely — the spinning host thread would compete
// with the driver's own worker threads.
// Spin on the counter before the blocking wait to save the park/wake
// round trip. It cannot replace that wait: reading the counter is a query,
// not the host domain operation that makes device writes visible.
if (_poll_waits) {
const auto deadline = std::chrono::steady_clock::now() +
std::chrono::milliseconds(100);
@@ -640,7 +637,7 @@ bool Context::wait(uint64_t value) {
set_error("vkGetSemaphoreCounterValue failed", r);
return false;
}
if (current >= value) return true;
if (current >= value) break;
std::this_thread::yield();
} while (std::chrono::steady_clock::now() < deadline);
}
+1
View File
@@ -61,6 +61,7 @@ uint64_t stream_flush(StreamImpl* s);
// Called after every dispatch/copy recorded into a batch; reproduces CUDA
// stream ordering (see README).
void stream_barrier(VkCommandBuffer cb);
void host_read_barrier(VkCommandBuffer cb);
// Flushes every live stream and returns the highest submitted value.
uint64_t flush_all_streams();
+17
View File
@@ -312,6 +312,20 @@ void stream_barrier(VkCommandBuffer cb) {
0, 1, &mb, 0, nullptr, 0, nullptr);
}
// Device writes reach a mapped pointer only through a barrier naming the host
// stage. Desktop drivers tolerate its absence; GB10 unified memory hands back
// stale lines without it.
void host_read_barrier(VkCommandBuffer cb) {
VkMemoryBarrier mb{VK_STRUCTURE_TYPE_MEMORY_BARRIER};
mb.srcAccessMask = VK_ACCESS_TRANSFER_WRITE_BIT |
VK_ACCESS_SHADER_WRITE_BIT;
mb.dstAccessMask = VK_ACCESS_HOST_READ_BIT;
vkCmdPipelineBarrier(
cb,
VK_PIPELINE_STAGE_TRANSFER_BIT | VK_PIPELINE_STAGE_COMPUTE_SHADER_BIT,
VK_PIPELINE_STAGE_HOST_BIT, 0, 1, &mb, 0, nullptr, 0, nullptr);
}
uint64_t flush_all_streams() {
std::vector<StreamImpl*> streams;
{
@@ -542,6 +556,7 @@ void staged_download_sync(void* dst, const ResolvedPtr& src, size_t bytes) {
vk::record_and_wait([&](VkCommandBuffer cb) {
VkBufferCopy c{src.offset + done, reg.offset, n};
vkCmdCopyBuffer(cb, src.alloc.buffer, reg.buffer, 1, &c);
vk::host_read_barrier(cb);
});
std::memcpy((char*)dst + done, reg.mapped, n);
done += n;
@@ -617,6 +632,7 @@ void memcpy_sync(void* dst, const void* src, size_t bytes, MemcpyKind kind) {
vk::record_and_wait([&](VkCommandBuffer cb) {
VkBufferCopy c{s.offset, d.offset, bytes};
vkCmdCopyBuffer(cb, s.alloc.buffer, d.alloc.buffer, 1, &c);
vk::host_read_barrier(cb);
});
} else {
staged_download_sync(dst, s, bytes);
@@ -716,6 +732,7 @@ void memcpy_async(void* dst, const void* src, size_t bytes, MemcpyKind kind,
VkBufferCopy c{s.offset, d.offset, bytes};
vkCmdCopyBuffer(cb, s.alloc.buffer, d.alloc.buffer, 1, &c);
vk::stream_barrier(cb);
vk::host_read_barrier(cb);
return;
}
// Pageable destination degrades to sync (as CUDA does).