Commit: c0686bd9a265c2acb7e9f35ebaa7c81af64b6dd9
Parent: a86ce25c884bbdae720165fb550b97bc35946406
Author: Randy Palamar
Date: Sat, 3 Oct 2026 17:03:17 -0700
gpu/core: use more explicit upload api to reduce latency
This improves latency in two ways:
1. move compute stall waiting for upload until after command
buffer recording. The compute thread doesn't need to know the
upload thread's signal value until it is time to submit it's
command buffer.
2. makes upload thread provide signal value before doing any
memory copying. this requires a more explicit api but in turn
allows for the compute command buffer to be fully submitted to the
device queue before the upload has completed.
We do not currently have a way of measuring end to end latency
(from rf capture completed to present) but once the UI is drawn
with Vulkan I can figure out a way to measure beamformer gets RF
data to presentation latency.
As was fully expected this change makes no measurable difference
to program throughput.
Diffstat:
5 files changed, 167 insertions(+), 71 deletions(-)
diff --git a/beamformer.c b/beamformer.c
@@ -152,6 +152,8 @@ function OS_THREAD_ENTRY_POINT_FN(beamformer_upload_entry_point)
GLWorkerThreadContext *ctx = user_context;
BeamformerUploadThreadContext *up = (typeof(up))ctx->user_context;
+ up->rf_buffer->upload_semaphore = gpu_semaphore_create(0);
+
for (;;) {
worker_thread_sleep(ctx, up->shared_memory);
beamformer_rf_upload(up);
diff --git a/beamformer_core.c b/beamformer_core.c
@@ -1,8 +1,5 @@
/* See LICENSE for license details. */
/* TODO(rnp):
- * [ ]: refactor: make better use Transfer timeline semaphore to not stall compute thread
- * while upload is occuring. rf thread should do: get next timeline semaphore value, insert
- * into wait values array, atomic_inc insertion index, start upload, signal timeline semaphore
* [ ]: backtrace dumping on SIGSEGV
* [ ]: cooperative shared memory loading in decode shader
* [ ]: refactor: save filter parameters with rest of parameters, whole slot thing is dumb
@@ -288,7 +285,7 @@ gpu_resource_build_end(GPUResourceBuilder *rb, GPUBuffer *buffer)
last_wait_value = gpu_buffer_range_upload(buffer, r->data, r->offset, r->size, 0);
// TODO(rnp): cleanup this pointless stall
- if (vk_buffer_needs_sync(buffer))
+ if (gpu_buffer_needs_sync(buffer))
gpu_host_wait_timeline(GPUTimeline_Transfer, last_wait_value, -1ULL);
}
@@ -1596,19 +1593,8 @@ complete_queue(BeamformerCtx *ctx, BeamformWorkQueue *q, Arena *arena)
// NOTE(rnp): the library/ui will ensure that any time a new dataset is uploaded
// the next time beamforming is issued it will be tagged as WaitThenCompute.
- // In this case we need to stall until the rf thread has finished its upload.
- if (work->kind == BeamformerWorkKind_WaitThenCompute) {
- // TODO(rnp): this shouldn't be necessary, there should be a way of communicating
- // what the value will be so that the only the command wait is needed.
- spin_wait(atomic_load_u64(&rf->insertion_index) <= compute_index);
-
- /* NOTE(rnp): if the GPU supports BAR there may be no need to synchronize
- * other than the above spin */
- if (vk_buffer_needs_sync(&rf->buffer))
- gpu_command_wait_timeline(cmd, GPUTimeline_Transfer, rf->upload_complete_values[slot]);
- } else {
+ if (work->kind != BeamformerWorkKind_WaitThenCompute)
slot = (compute_index - 1) % countof(rf->upload_complete_values);
- }
// NOTE(rnp): nvidia needs a memory barrier between pipeline stages
// for correct output. It doesn't seem to effect performance on nvidia cards.
@@ -1665,7 +1651,16 @@ complete_queue(BeamformerCtx *ctx, BeamformWorkQueue *q, Arena *arena)
do_compute_shader(cmd, cp, frame, 0, 0, i, 0);
gpu_command_timestamp(cmd);
}
- u64 end_timeline_value = gpu_command_list_end(cmd, (VulkanHandle){0}, (VulkanHandle){0});
+
+ u32 wait_info_count = 0;
+ GPUSemaphoreSignalInfo wait_info = {.semaphore = rf->upload_semaphore};
+ if (work->kind == BeamformerWorkKind_WaitThenCompute) {
+ spin_wait(atomic_load_u64(&rf->insertion_index) <= compute_index);
+ wait_info.value = atomic_load_u64(rf->upload_complete_values + slot);
+ wait_info_count = 1;
+ }
+
+ u64 end_timeline_value = gpu_command_list_end(cmd, &wait_info, wait_info_count, 0, 0);
atomic_store_u64(rf->compute_complete_values + slot, end_timeline_value);
if (work->kind == BeamformerWorkKind_WaitThenCompute)
@@ -1829,21 +1824,27 @@ DEBUG_EXPORT BEAMFORMER_RF_UPLOAD_FN(beamformer_rf_upload)
u64 slot = rf->insertion_index % countof(rf->upload_complete_values);
/* NOTE(rnp): don't overwrite slot if the compute thread hasn't processed it */
- spin_wait(atomic_load_u64(&rf->compute_index) < rf->insertion_index);
+ spin_wait(rf->insertion_index - atomic_load_u64(&rf->compute_index) >= countof(rf->upload_complete_values));
gpu_host_wait_timeline(GPUTimeline_Compute, rf->compute_complete_values[slot], -1ULL);
assert((ctx->shared_memory_size % os_system_info()->page_size) == 0 &&
(os_system_info()->page_size % gpu_round_up_to_sync_size(1, 64)) == 0);
- u64 wait_value = gpu_buffer_range_upload(&rf->buffer, beamformer_shared_memory_data_pointer(sm, ctx->shared_memory_size),
- slot * rf->active_rf_size, rf->active_rf_size, 1);
- store_fence();
-
- beamformer_shared_memory_release_lock(ctx->shared_memory, (i32)scratch_lock);
- post_sync_barrier(ctx->shared_memory, upload_lock);
+ u64 wait_value = gpu_semaphore_value(rf->upload_semaphore) + 1;
atomic_store_u64(rf->upload_complete_values + slot, wait_value);
atomic_add_u64(&rf->insertion_index, 1);
+ void *host_pointer = gpu_buffer_host_pointer(&rf->buffer);
+ memory_copy_non_temporal((u8 *)host_pointer + slot * rf->active_rf_size,
+ beamformer_shared_memory_data_pointer(sm, ctx->shared_memory_size),
+ rf->active_rf_size);
+ store_fence();
+
+ GPUSemaphoreSignalInfo signal_info = {.semaphore = rf->upload_semaphore, .value = wait_value};
+ gpu_buffer_make_visible(&rf->buffer, slot * rf->active_rf_size, rf->active_rf_size, &signal_info, 1);
+
+ beamformer_shared_memory_release_lock(ctx->shared_memory, (i32)scratch_lock);
+ post_sync_barrier(ctx->shared_memory, upload_lock);
os_wake_all_waiters(ctx->compute_worker_sync);
u64 current_time = os_timer_count();
diff --git a/beamformer_internal.h b/beamformer_internal.h
@@ -20,6 +20,7 @@
typedef struct { u64 value; } GPUHandle;
typedef struct { u64 value; } GPUCommandList;
+typedef struct { u64 value; } GPUSemaphore;
typedef struct { u64 value; } GPUSplitBarrier;
typedef struct { u64 value[1]; } VulkanHandle;
@@ -128,6 +129,12 @@ typedef struct {
} GPUBufferAllocateInfo;
typedef struct {
+ GPUSemaphore semaphore;
+ // NOTE(rnp): ignored if semaphore was exported
+ u64 value;
+} GPUSemaphoreSignalInfo;
+
+typedef struct {
GPUBuffer model;
u32 vertex_count;
u32 normals_offset;
@@ -163,16 +170,19 @@ DEBUG_IMPORT VulkanHandle vk_pipeline(VulkanPipelineCreateInfo *infos, u32 count
DEBUG_IMPORT b32 vk_pipeline_valid(VulkanHandle);
DEBUG_IMPORT void vk_pipeline_release(VulkanHandle);
-DEBUG_IMPORT b32 vk_buffer_needs_sync(GPUBuffer *);
+DEBUG_IMPORT GPUSemaphore gpu_semaphore_create(OSHandle *export);
+DEBUG_IMPORT u64 gpu_semaphore_value(GPUSemaphore);
+DEBUG_IMPORT void gpu_host_signal_semaphore(GPUSemaphore semaphore, u64 value);
-DEBUG_IMPORT VulkanHandle vk_create_semaphore(OSHandle *export);
+DEBUG_IMPORT b32 gpu_buffer_needs_sync(GPUBuffer *);
DEBUG_IMPORT b32 gpu_host_wait_timeline(GPUTimeline timeline, u64 value, u64 timeout_ns);
DEBUG_IMPORT u64 gpu_host_signal_timeline(GPUTimeline timeline);
DEBUG_IMPORT GPUCommandList gpu_command_list_begin(GPUTimeline timeline);
-// NOTE: extra semaphores only exist for synchronization with OpenGL and will be removed in the future
-DEBUG_IMPORT u64 gpu_command_list_end(GPUCommandList command, VulkanHandle wait_semaphore, VulkanHandle finished_semaphore);
+DEBUG_IMPORT u64 gpu_command_list_end(GPUCommandList command,
+ GPUSemaphoreSignalInfo *wait_infos, u64 wait_info_count,
+ GPUSemaphoreSignalInfo *signal_infos, u64 signal_info_count);
DEBUG_IMPORT void gpu_command_bind_pipeline(GPUCommandList command, VulkanHandle pipeline);
DEBUG_IMPORT void gpu_command_pipeline_barrier(GPUCommandList command, b32 memory);
@@ -196,6 +206,22 @@ DEBUG_IMPORT void gpu_command_copy_buffer(GPUCommandList command,
// NOTE: returns array of valid timestamps. Calling thread may stall until results available.
DEBUG_IMPORT u64 * gpu_read_timestamps(GPUTimeline timeline, u64 *count, Arena *arena);
+/////////////////////////
+// NOTE(rnp): advanced buffer manipulation
+
+// IMPORTANT: there is no guarantee that the GPU can see writes through these
+// these APIs unless you call gpu_buffer_make_visible with the memory range
+// that was written
+// NOTE(rnp): caller is responsible for passing a buffer that was created with
+// the required access (read, write, or both).
+DEBUG_IMPORT void * gpu_buffer_host_pointer(GPUBuffer *);
+// NOTE(rnp): returns a value to wait on from the Transfer timeline if it
+// was used (a wait value is never 0). Any passed in semaphores are always
+// signaled (possibly via a host signal operation).
+DEBUG_IMPORT u64 gpu_buffer_make_visible(GPUBuffer *, u64 offset, u64 size,
+ GPUSemaphoreSignalInfo *signal_infos,
+ u64 signal_info_count);
+
#if BEAMFORMER_RENDERDOC_HOOKS
DEBUG_IMPORT void * vk_renderdoc_instance_handle(void);
@@ -329,7 +355,8 @@ typedef struct {
u64 upload_complete_values[BeamformerMaxRawDataFramesInFlight];
u64 compute_complete_values[BeamformerMaxRawDataFramesInFlight];
- GPUBuffer buffer;
+ GPUBuffer buffer;
+ GPUSemaphore upload_semaphore;
u64 active_rf_size;
diff --git a/ui.c b/ui.c
@@ -365,7 +365,7 @@ typedef struct {
VulkanHandle pipelines[BeamformerShaderKind_RenderCount];
OSHandle render_semaphores_export[2];
- VulkanHandle render_semaphores[2];
+ GPUSemaphore render_semaphores[2];
u32 render_semaphores_gl[2];
GPUImage render_3d_image;
@@ -866,7 +866,7 @@ beamformer_ui_frame_view_copy_frame(BeamformerFrameView *new, BeamformerFrameVie
gpu_command_wait_timeline(cmd, GPUTimeline_Compute, old->frame.timeline_valid_value);
u64 offset = old->frame.gpu_pointer - buffer->gpu_pointer;
gpu_command_copy_buffer(cmd, &new->copy_buffer, 0, buffer, offset, frame_size);
- new->frame.timeline_valid_value = gpu_command_list_end(cmd, (VulkanHandle){0}, (VulkanHandle){0});
+ new->frame.timeline_valid_value = gpu_command_list_end(cmd, 0, 0, 0, 0);
}
function BeamformerFrameView *
@@ -1146,7 +1146,9 @@ update_frame_views(BeamformerUI *ui, Rect window)
render_2d_plane(view, cmd, &pc);
}
gpu_command_end_rendering(cmd);
- gpu_command_list_end(cmd, ui->render_semaphores[0], ui->render_semaphores[1]);
+ GPUSemaphoreSignalInfo wait_info = {.semaphore = ui->render_semaphores[0]};
+ GPUSemaphoreSignalInfo signal_info = {.semaphore = ui->render_semaphores[1]};
+ gpu_command_list_end(cmd, &wait_info, 1, &signal_info, 1);
glWaitSemaphoreEXT(ui->render_semaphores_gl[1], 0, 0, 1, &view->texture, (GLenum[]){GL_LAYOUT_COLOR_ATTACHMENT_EXT});
@@ -5119,7 +5121,7 @@ ui_init(BeamformerCtx *ctx, Arena *store)
glGenSemaphoresEXT(countof(ui->render_semaphores_gl), ui->render_semaphores_gl);
for EachElement(ui->render_semaphores, it)
- ui->render_semaphores[it] = vk_create_semaphore(ui->render_semaphores_export + it);
+ ui->render_semaphores[it] = gpu_semaphore_create(ui->render_semaphores_export + it);
if (OS_WINDOWS) {
glImportSemaphoreWin32HandleEXT(ui->render_semaphores_gl[0], GL_HANDLE_TYPE_OPAQUE_WIN32_EXT, (void *)ui->render_semaphores_export[0].value[0]);
diff --git a/vulkan.c b/vulkan.c
@@ -121,6 +121,9 @@ typedef alignas(64) struct {
i32 lock;
u32 next_command_buffer_index;
+ // NOTE(rnp): small arena for storing temporary data during submission
+ Arena *arena;
+
VulkanPipeline *bound_pipeline;
u64 last_submission_values[MaxCommandBuffersInFlight];
@@ -1828,6 +1831,8 @@ vk_load_queues(Arena *arena, Stream *err)
for EachElement(vk->command_pools, it) {
VulkanCommandPool *vcp = vk->command_pools[it];
+ vcp->arena = arena_create(.reserve_size = KB(64));
+
VkCommandPoolCreateInfo command_pool_create_info = {
.sType = VK_STRUCTURE_TYPE_COMMAND_POOL_CREATE_INFO,
.flags = VK_COMMAND_POOL_CREATE_RESET_COMMAND_BUFFER_BIT,
@@ -2025,25 +2030,6 @@ gpu_buffer_allocate(GPUBuffer *b, GPUBufferAllocateInfo info)
}
}
-DEBUG_IMPORT b32
-vk_buffer_needs_sync(GPUBuffer *b)
-{
- b32 result = 0;
- if (b->handle.value) {
- VulkanBuffer *vb = vk_entity_data(b->handle.value, VulkanEntityKind_Buffer);
- result = vb->next != 0;
- }
- return result;
-}
-
-DEBUG_IMPORT u64
-gpu_round_up_to_sync_size(u64 size, u64 min)
-{
- i64 round = (i64)Max(min, vulkan_context->memory_info.non_coherent_atom_size);
- u64 result = (u64)round_up_to((i64)size, round);
- return result;
-}
-
function void
vk_command_copy_buffer(VkCommandBuffer cb, VkBuffer db, u64 destination_offset, VkBuffer sb, u64 source_offset, u64 size)
{
@@ -2065,6 +2051,77 @@ vk_command_copy_buffer(VkCommandBuffer cb, VkBuffer db, u64 destination_offset,
vkCmdCopyBuffer2(cb, ©_buffer_info);
}
+DEBUG_IMPORT u64
+gpu_semaphore_value(GPUSemaphore semaphore)
+{
+ VulkanSemaphore *fs = vk_entity_data(semaphore.value, VulkanEntityKind_Semaphore);
+ u64 result = fs->value;
+ return result;
+}
+
+DEBUG_IMPORT void
+gpu_host_signal_semaphore(GPUSemaphore semaphore, u64 value)
+{
+ VulkanSemaphore *vs = vk_entity_data(semaphore.value, VulkanEntityKind_Semaphore);
+ assert(vs->value < value);
+ VkSemaphoreSignalInfo ssi = {
+ .sType = VK_STRUCTURE_TYPE_SEMAPHORE_SIGNAL_INFO,
+ .semaphore = vs->semaphore,
+ .value = value,
+ };
+ vs->value = value;
+ vkSignalSemaphore(vulkan_context->device, &ssi);
+}
+
+DEBUG_IMPORT void *
+gpu_buffer_host_pointer(GPUBuffer *b)
+{
+ void *result = 0;
+ if (b->handle.value) {
+ VulkanBuffer *vb = vk_entity_data(b->handle.value, VulkanEntityKind_Buffer);
+ result = vb->next ? vb->next->as.buffer.host_pointer : vb->host_pointer;
+ }
+ return result;
+}
+
+DEBUG_IMPORT u64
+gpu_buffer_make_visible(GPUBuffer *b, u64 offset, u64 size,
+ GPUSemaphoreSignalInfo *signal_infos, u64 signal_info_count)
+{
+ u64 result = 0;
+ if (b->handle.value) {
+ VulkanBuffer *vb = vk_entity_data(b->handle.value, VulkanEntityKind_Buffer);
+ if (vb->memory_kind == VulkanMemoryKind_Device && vb->next) {
+ GPUCommandList cb = gpu_command_list_begin(GPUTimeline_Transfer);
+ vk_command_copy_buffer(vk_command_buffer(cb), vb->buffer, offset, vb->next->as.buffer.buffer, offset, size);
+ result = gpu_command_list_end(cb, 0, 0, signal_infos, signal_info_count);
+ } else {
+ for EachIndex(signal_info_count, index)
+ gpu_host_signal_semaphore(signal_infos[index].semaphore, signal_infos[index].value);
+ }
+ }
+ return result;
+}
+
+DEBUG_IMPORT b32
+gpu_buffer_needs_sync(GPUBuffer *b)
+{
+ b32 result = 0;
+ if (b->handle.value) {
+ VulkanBuffer *vb = vk_entity_data(b->handle.value, VulkanEntityKind_Buffer);
+ result = vb->next != 0;
+ }
+ return result;
+}
+
+DEBUG_IMPORT u64
+gpu_round_up_to_sync_size(u64 size, u64 min)
+{
+ i64 round = (i64)Max(min, vulkan_context->memory_info.non_coherent_atom_size);
+ u64 result = (u64)round_up_to((i64)size, round);
+ return result;
+}
+
function force_inline u64
vk_buffer_buffer_copy(VulkanBuffer *destination, VulkanBuffer *source, u64 destination_offset, u64 source_offset, u64 size, b32 non_temporal)
{
@@ -2155,7 +2212,7 @@ vk_buffer_buffer_copy(VulkanBuffer *destination, VulkanBuffer *source, u64 desti
GPUCommandList cb = gpu_command_list_begin(GPUTimeline_Transfer);
vk_command_copy_buffer(vk_command_buffer(cb), destination->buffer, destination_offset, db->buffer, destination_offset, size);
- result = gpu_command_list_end(cb, (VulkanHandle){0}, (VulkanHandle){0});
+ result = gpu_command_list_end(cb, 0, 0, 0, 0);
}break;
InvalidDefaultCase;
@@ -2174,7 +2231,7 @@ vk_buffer_buffer_copy(VulkanBuffer *destination, VulkanBuffer *source, u64 desti
GPUCommandList cb = gpu_command_list_begin(GPUTimeline_Transfer);
vk_command_copy_buffer(vk_command_buffer(cb), sb->buffer, 0, source->buffer, source_offset, size);
- u64 wait_value = gpu_command_list_end(cb, (VulkanHandle){0}, (VulkanHandle){0});
+ u64 wait_value = gpu_command_list_end(cb, 0, 0, 0, 0);
// TODO(rnp): asynchronous transfers
gpu_host_wait_timeline(GPUTimeline_Transfer, wait_value, -1ULL);
@@ -2409,12 +2466,12 @@ vk_image_allocate(GPUImage *image, u32 width, u32 height, u32 mips, u32 samples,
}
}
-DEBUG_IMPORT VulkanHandle
-vk_create_semaphore(OSHandle *export)
+DEBUG_IMPORT GPUSemaphore
+gpu_semaphore_create(OSHandle *export)
{
VulkanEntity *e = vk_entity_allocate(VulkanEntityKind_Semaphore);
e->as.semaphore = vk_make_semaphore(export);
- VulkanHandle result = {(u64)e};
+ GPUSemaphore result = {(u64)e};
return result;
}
@@ -2692,7 +2749,8 @@ gpu_command_wait_timeline(GPUCommandList command, GPUTimeline timeline, u64 valu
}
DEBUG_IMPORT u64
-gpu_command_list_end(GPUCommandList command, VulkanHandle wait_semaphore, VulkanHandle finished_semaphore)
+gpu_command_list_end(GPUCommandList command, GPUSemaphoreSignalInfo *wait_infos, u64 wait_info_count,
+ GPUSemaphoreSignalInfo *signal_infos, u64 signal_info_count)
{
u64 result = -1;
if (command.value) {
@@ -2704,6 +2762,8 @@ gpu_command_list_end(GPUCommandList command, VulkanHandle wait_semaphore, Vulkan
vkEndCommandBuffer(vcp->buffers[vcb->buffer_index]);
+ arena_clear(vcp->arena);
+
DeferLoop(take_lock(&vq->lock, -1), release_lock(&vq->lock)) {
VkCommandBufferSubmitInfo command_buffer_submit_info = {
.sType = VK_STRUCTURE_TYPE_COMMAND_BUFFER_SUBMIT_INFO,
@@ -2712,27 +2772,30 @@ gpu_command_list_end(GPUCommandList command, VulkanHandle wait_semaphore, Vulkan
result = ++vs->value;
- u32 signal_submit_info_count = 1;
- VkSemaphoreSubmitInfo signal_submit_infos[2] = {{
+ VkSemaphoreSubmitInfo *signal_submit_infos = push_array(vcp->arena, VkSemaphoreSubmitInfo, signal_info_count + 1);
+ signal_submit_infos[0] = (VkSemaphoreSubmitInfo){
.sType = VK_STRUCTURE_TYPE_SEMAPHORE_SUBMIT_INFO,
.semaphore = vs->semaphore,
.value = result,
.stageMask = vq->pipeline_stage_flags,
- }};
+ };
- if ValidVulkanHandle(finished_semaphore) {
- VulkanSemaphore *fs = vk_entity_data(finished_semaphore.value[0], VulkanEntityKind_Semaphore);
- signal_submit_infos[signal_submit_info_count++] = (VkSemaphoreSubmitInfo){
+ for EachIndex(signal_info_count, index) {
+ VulkanSemaphore *fs = vk_entity_data(signal_infos[index].semaphore.value, VulkanEntityKind_Semaphore);
+ signal_submit_infos[index + 1] = (VkSemaphoreSubmitInfo){
.sType = VK_STRUCTURE_TYPE_SEMAPHORE_SUBMIT_INFO,
.semaphore = fs->semaphore,
+ .value = signal_infos[index].value,
.stageMask = vq->pipeline_stage_flags,
};
+ fs->value = signal_infos[index].value;
}
u32 wait_submit_info_count = 0;
- VkSemaphoreSubmitInfo wait_submit_infos[VulkanQueueKind_Count + 1];
- for (u32 i = 0; i < vk->unique_queues; i++) {
- u32 queue_index = vk->queue_indices[i];
+ VkSemaphoreSubmitInfo *wait_submit_infos = push_array(vcp->arena, VkSemaphoreSubmitInfo,
+ wait_info_count + VulkanQueueKind_Count + 1);
+ for EachIndex(vk->unique_queues, index) {
+ u32 queue_index = vk->queue_indices[index];
if (vcb->in_flight_wait_values[queue_index] > 0) {
VulkanQueue *q = vk->queues[queue_index];
VkSemaphoreSubmitInfo wait_ssi = {
@@ -2745,11 +2808,12 @@ gpu_command_list_end(GPUCommandList command, VulkanHandle wait_semaphore, Vulkan
}
}
- if ValidVulkanHandle(wait_semaphore) {
- VulkanSemaphore *ws = vk_entity_data(wait_semaphore.value[0], VulkanEntityKind_Semaphore);
+ for EachIndex(wait_info_count, index) {
+ VulkanSemaphore *fs = vk_entity_data(wait_infos[index].semaphore.value, VulkanEntityKind_Semaphore);
wait_submit_infos[wait_submit_info_count++] = (VkSemaphoreSubmitInfo){
.sType = VK_STRUCTURE_TYPE_SEMAPHORE_SUBMIT_INFO,
- .semaphore = ws->semaphore,
+ .semaphore = fs->semaphore,
+ .value = wait_infos[index].value,
.stageMask = vq->pipeline_stage_flags,
};
}
@@ -2760,7 +2824,7 @@ gpu_command_list_end(GPUCommandList command, VulkanHandle wait_semaphore, Vulkan
.pCommandBufferInfos = &command_buffer_submit_info,
.waitSemaphoreInfoCount = wait_submit_info_count,
.pWaitSemaphoreInfos = wait_submit_infos,
- .signalSemaphoreInfoCount = signal_submit_info_count,
+ .signalSemaphoreInfoCount = signal_info_count + 1,
.pSignalSemaphoreInfos = signal_submit_infos,
};