mirror of
https://github.com/harry7557558/spirula-studio.git
synced 2026-10-02 02:44:54 +08:00
fix vulkan synchronization bugs that affect unified memory platforms
This commit is contained in:
@@ -623,12 +623,9 @@ uint64_t Context::submit(VkCommandBuffer cb) {
|
||||
|
||||
bool Context::wait(uint64_t value) {
|
||||
if (value == 0) return true;
|
||||
// Poll the counter first (the timeline analog of vkGetFenceStatus,
|
||||
// measurably cheaper than the blocking vkWaitSemaphores path on desktop
|
||||
// drivers). The spin is bounded so a stuck device (device fault) ends up
|
||||
// parked in the blocking wait instead of burning a core; CPU devices
|
||||
// (llvmpipe) skip it entirely — the spinning host thread would compete
|
||||
// with the driver's own worker threads.
|
||||
// Spin on the counter before the blocking wait to save the park/wake
|
||||
// round trip. It cannot replace that wait: reading the counter is a query,
|
||||
// not the host domain operation that makes device writes visible.
|
||||
if (_poll_waits) {
|
||||
const auto deadline = std::chrono::steady_clock::now() +
|
||||
std::chrono::milliseconds(100);
|
||||
@@ -640,7 +637,7 @@ bool Context::wait(uint64_t value) {
|
||||
set_error("vkGetSemaphoreCounterValue failed", r);
|
||||
return false;
|
||||
}
|
||||
if (current >= value) return true;
|
||||
if (current >= value) break;
|
||||
std::this_thread::yield();
|
||||
} while (std::chrono::steady_clock::now() < deadline);
|
||||
}
|
||||
|
||||
@@ -61,6 +61,7 @@ uint64_t stream_flush(StreamImpl* s);
|
||||
// Called after every dispatch/copy recorded into a batch; reproduces CUDA
|
||||
// stream ordering (see README).
|
||||
void stream_barrier(VkCommandBuffer cb);
|
||||
void host_read_barrier(VkCommandBuffer cb);
|
||||
|
||||
// Flushes every live stream and returns the highest submitted value.
|
||||
uint64_t flush_all_streams();
|
||||
|
||||
@@ -312,6 +312,20 @@ void stream_barrier(VkCommandBuffer cb) {
|
||||
0, 1, &mb, 0, nullptr, 0, nullptr);
|
||||
}
|
||||
|
||||
// Device writes reach a mapped pointer only through a barrier naming the host
|
||||
// stage. Desktop drivers tolerate its absence; GB10 unified memory hands back
|
||||
// stale lines without it.
|
||||
void host_read_barrier(VkCommandBuffer cb) {
|
||||
VkMemoryBarrier mb{VK_STRUCTURE_TYPE_MEMORY_BARRIER};
|
||||
mb.srcAccessMask = VK_ACCESS_TRANSFER_WRITE_BIT |
|
||||
VK_ACCESS_SHADER_WRITE_BIT;
|
||||
mb.dstAccessMask = VK_ACCESS_HOST_READ_BIT;
|
||||
vkCmdPipelineBarrier(
|
||||
cb,
|
||||
VK_PIPELINE_STAGE_TRANSFER_BIT | VK_PIPELINE_STAGE_COMPUTE_SHADER_BIT,
|
||||
VK_PIPELINE_STAGE_HOST_BIT, 0, 1, &mb, 0, nullptr, 0, nullptr);
|
||||
}
|
||||
|
||||
uint64_t flush_all_streams() {
|
||||
std::vector<StreamImpl*> streams;
|
||||
{
|
||||
@@ -542,6 +556,7 @@ void staged_download_sync(void* dst, const ResolvedPtr& src, size_t bytes) {
|
||||
vk::record_and_wait([&](VkCommandBuffer cb) {
|
||||
VkBufferCopy c{src.offset + done, reg.offset, n};
|
||||
vkCmdCopyBuffer(cb, src.alloc.buffer, reg.buffer, 1, &c);
|
||||
vk::host_read_barrier(cb);
|
||||
});
|
||||
std::memcpy((char*)dst + done, reg.mapped, n);
|
||||
done += n;
|
||||
@@ -617,6 +632,7 @@ void memcpy_sync(void* dst, const void* src, size_t bytes, MemcpyKind kind) {
|
||||
vk::record_and_wait([&](VkCommandBuffer cb) {
|
||||
VkBufferCopy c{s.offset, d.offset, bytes};
|
||||
vkCmdCopyBuffer(cb, s.alloc.buffer, d.alloc.buffer, 1, &c);
|
||||
vk::host_read_barrier(cb);
|
||||
});
|
||||
} else {
|
||||
staged_download_sync(dst, s, bytes);
|
||||
@@ -716,6 +732,7 @@ void memcpy_async(void* dst, const void* src, size_t bytes, MemcpyKind kind,
|
||||
VkBufferCopy c{s.offset, d.offset, bytes};
|
||||
vkCmdCopyBuffer(cb, s.alloc.buffer, d.alloc.buffer, 1, &c);
|
||||
vk::stream_barrier(cb);
|
||||
vk::host_read_barrier(cb);
|
||||
return;
|
||||
}
|
||||
// Pageable destination degrades to sync (as CUDA does).
|
||||
|
||||
Reference in New Issue
Block a user