Commit: 9176fb5c8ae3d5f82e11fe73e60b3201e745bd20
Parent: 4433143460dc818d69e144d67d54e3d7a477064e
Author: Randy Palamar
Date: Mon, 14 Sep 2026 20:52:53 -0700
vulkan: cleanup staging buffer setup/usage
First, I actually forgot to hook up the single_transfer_size
parameter so the buffer was the full size anyways. The size being
smaller is actually not desired though, to allow overlapping
uploads the staging buffer needs to have identical layout to the
GPU buffer (consider transfer operation still in progress while
CPU is filling next ring buffer slot). So actually use the extra
space and drop the unnecessary upload stall.
Furthermore, when the Host RW flags were cleared during buffer
allocation an transfer source/destination flags were not added to
each of the two buffers.
Diffstat:
3 files changed, 23 insertions(+), 21 deletions(-)
diff --git a/beamformer_core.c b/beamformer_core.c
@@ -277,9 +277,14 @@ gpu_resource_build_end(GPUResourceBuilder *rb, GPUBuffer *buffer)
//////////////////////////////////////
// NOTE(rnp): upload data
+ u64 last_wait_value = 0;
for (GPUResource *r = rb->resource_list; r; r = r->next)
if (r->data)
- gpu_buffer_range_upload(buffer, r->data, r->offset, r->size, 0);
+ last_wait_value = gpu_buffer_range_upload(buffer, r->data, r->offset, r->size, 0);
+
+ // TODO(rnp): cleanup this pointless stall
+ if (vk_buffer_needs_sync(buffer))
+ gpu_host_wait_timeline(GPUTimeline_Transfer, last_wait_value, -1ULL);
}
function BeamformerComputePlan *
@@ -1741,7 +1746,6 @@ DEBUG_EXPORT BEAMFORMER_RF_UPLOAD_FN(beamformer_rf_upload)
.size = countof(rf->upload_complete_values) * rf->active_rf_size,
.flags = GPUUsageFlag_HostWrite,
.label = str8("RawRFBuffer"),
- .single_transfer_size = rf->active_rf_size,
};
gpu_buffer_allocate(&rf->buffer, allocate_info);
}
@@ -1754,14 +1758,14 @@ DEBUG_EXPORT BEAMFORMER_RF_UPLOAD_FN(beamformer_rf_upload)
assert((ctx->shared_memory_size % os_system_info()->page_size) == 0 &&
(os_system_info()->page_size % gpu_round_up_to_sync_size(1, 64)) == 0);
- gpu_buffer_range_upload(&rf->buffer, beamformer_shared_memory_data_pointer(sm, ctx->shared_memory_size),
- slot * rf->active_rf_size, rf->active_rf_size, 1);
+ u64 wait_value = gpu_buffer_range_upload(&rf->buffer, beamformer_shared_memory_data_pointer(sm, ctx->shared_memory_size),
+ slot * rf->active_rf_size, rf->active_rf_size, 1);
store_fence();
beamformer_shared_memory_release_lock(ctx->shared_memory, (i32)scratch_lock);
post_sync_barrier(ctx->shared_memory, upload_lock);
- atomic_store_u64(rf->upload_complete_values + slot, gpu_host_signal_timeline(GPUTimeline_Transfer));
+ atomic_store_u64(rf->upload_complete_values + slot, wait_value);
atomic_add_u64(&rf->insertion_index, 1);
os_wake_all_waiters(ctx->compute_worker_sync);
diff --git a/beamformer_internal.h b/beamformer_internal.h
@@ -118,12 +118,6 @@ typedef struct {
GPUUsageFlags flags;
i64 size;
- // NOTE(rnp): when the buffer is used as a destination for CPU->GPU transfers
- // and the GPU doesn't support full UMA/ReBAR access this indicates the maximum
- // size the CPU will try to transfer in one go. This can be used to reduce to
- // reduce CPU memory overhead for large GPU side buffers
- i64 single_transfer_size;
-
// NOTE(rnp): only required if buffer will be used on multiple timelines
u32 timeline_count;
GPUTimeline *timelines_used;
@@ -149,7 +143,7 @@ DEBUG_IMPORT GPUInfo *gpu_info(void);
DEBUG_IMPORT void gpu_buffer_allocate(GPUBuffer *, GPUBufferAllocateInfo info);
DEBUG_IMPORT void gpu_buffer_release(GPUBuffer *);
-DEBUG_IMPORT void gpu_buffer_range_upload(GPUBuffer *, void *data, u64 offset, u64 size, b32 non_temporal);
+DEBUG_IMPORT u64 gpu_buffer_range_upload(GPUBuffer *, void *data, u64 offset, u64 size, b32 non_temporal);
DEBUG_IMPORT void gpu_buffer_range_download(void *output, GPUBuffer *, u64 source_offset, u64 size, b32 non_temporal);
DEBUG_IMPORT u64 gpu_round_up_to_sync_size(u64, u64 min);
diff --git a/vulkan.c b/vulkan.c
@@ -1086,6 +1086,8 @@ vk_buffer_allocate_common(VulkanBuffer *vb, VulkanBufferAllocateInfo *ai)
u32 transfer_queue_family = vk->queues[vk->queue_indices[VulkanQueueKind_Transfer]]->queue_family;
ai->flags &= ~GPUUsageFlag_HostReadWrite;
+ ai->flags |= GPUUsageFlag_TransferDestination;
+
b32 found = 0;
for EachElement(ai->queue_family_indices, it) {
if (ai->queue_family_indices[it] == transfer_queue_family) {
@@ -1103,7 +1105,7 @@ vk_buffer_allocate_common(VulkanBuffer *vb, VulkanBufferAllocateInfo *ai)
VulkanBufferAllocateInfo asi = {
.size = AlignUpPowerOfTwo(host_size, vk->memory_info.non_coherent_atom_size),
.index_type = VK_INDEX_TYPE_NONE_KHR,
- .flags = host_rw_flags,
+ .flags = host_rw_flags|GPUUsageFlag_TransferSource,
.queue_family_count = 1,
.queue_family_indices[0] = vk->queues[vk->queue_indices[VulkanQueueKind_Transfer]]->queue_family,
};
@@ -2050,12 +2052,13 @@ vk_command_copy_buffer(VkCommandBuffer cb, VkBuffer db, u64 destination_offset,
vkCmdCopyBuffer2(cb, ©_buffer_info);
}
-function force_inline void
+function force_inline u64
vk_buffer_buffer_copy(VulkanBuffer *destination, VulkanBuffer *source, u64 destination_offset, u64 source_offset, u64 size, b32 non_temporal)
{
VulkanContext *vk = vulkan_context;
(void)vk;
+ u64 result = 0;
switch (source->memory_kind) {
#if !ForceStagingBuffers
case VulkanMemoryKind_BAR:
@@ -2130,7 +2133,7 @@ vk_buffer_buffer_copy(VulkanBuffer *destination, VulkanBuffer *source, u64 desti
case VulkanMemoryKind_Device:{
VulkanBuffer *db = vk_entity_data((u64)destination->next, VulkanEntityKind_Buffer);
assert(size <= db->memory_size);
- void *dest = (u8 *)db->host_pointer;
+ void *dest = (u8 *)db->host_pointer + destination_offset;
void *src = (u8 *)source->host_pointer + source_offset;
// NOTE(rnp): don't trash the CPU cache for large data stores
if (non_temporal) memory_copy_non_temporal(dest, src, size);
@@ -2138,10 +2141,8 @@ vk_buffer_buffer_copy(VulkanBuffer *destination, VulkanBuffer *source, u64 desti
store_fence();
GPUCommandList cb = gpu_command_list_begin(GPUTimeline_Transfer);
- vk_command_copy_buffer(vk_command_buffer(cb), destination->buffer, destination_offset, db->buffer, 0, size);
- u64 wait_value = gpu_command_list_end(cb, (VulkanHandle){0}, (VulkanHandle){0});
- // TODO(rnp): asynchronous transfers
- gpu_host_wait_timeline(GPUTimeline_Transfer, wait_value, -1ULL);
+ vk_command_copy_buffer(vk_command_buffer(cb), destination->buffer, destination_offset, db->buffer, destination_offset, size);
+ result = gpu_command_list_end(cb, (VulkanHandle){0}, (VulkanHandle){0});
}break;
InvalidDefaultCase;
@@ -2175,9 +2176,11 @@ vk_buffer_buffer_copy(VulkanBuffer *destination, VulkanBuffer *source, u64 desti
InvalidDefaultCase;
}
+
+ return result;
}
-DEBUG_IMPORT void
+DEBUG_IMPORT u64
gpu_buffer_range_upload(GPUBuffer *b, void *data, u64 offset, u64 size, b32 non_temporal)
{
VulkanBuffer *db = vk_entity_data(b->handle.value, VulkanEntityKind_Buffer);
@@ -2185,7 +2188,8 @@ gpu_buffer_range_upload(GPUBuffer *b, void *data, u64 offset, u64 size, b32 non_
.host_pointer = data,
.memory_kind = VulkanMemoryKind_Host,
};
- vk_buffer_buffer_copy(db, &sb, offset, 0, size, non_temporal);
+ u64 result = vk_buffer_buffer_copy(db, &sb, offset, 0, size, non_temporal);
+ return result;
}
DEBUG_IMPORT void