ogl_beamforming

Ultrasound Beamforming Implemented with OpenGL
git clone anongit@rnpnr.xyz:ogl_beamforming.git
Log | Files | Refs | Feed | Submodules | README | LICENSE

Commit: f17ace4486eb20f3002d5a61d7a0acb3331b03ae
Parent: 2865c89febccaa0e39ea5be76644bad86a291c95
Author: Randy Palamar
Date:   Wed, 16 Sep 2026 19:43:13 -0700

core: re-enable memory barrier between compute stages

this is required for correct output on my nVidia test system. It
does cause a very minor performance regression for AMD but since
its technically against the vulkan spec to skip we will just
enable them everywhere for now.

Diffstat:
Mbeamformer_core.c | 12++++++++++--
1 file changed, 10 insertions(+), 2 deletions(-)

diff --git a/beamformer_core.c b/beamformer_core.c @@ -1545,6 +1545,12 @@ complete_queue(BeamformerCtx *ctx, BeamformWorkQueue *q, Arena *arena) slot = (rf->compute_index - 1) % countof(rf->upload_complete_values); } + // NOTE(rnp): nvidia needs a memory barrier between pipeline stages + // for correct output. It doesn't seem to effect performance on nvidia cards. + // It does have a very minor performance decrease on AMD, but technically its + // against vulkan spec to skip them so for simplicity we will default to for now + b32 pipeline_memory_barrier = 1; //gpu_info()->vendor == GPUVendor_NVIDIA; + for (u32 channel_offset = 0; channel_offset < cp->channel_count; channel_offset += BeamformerChunkChannelCount) @@ -1567,6 +1573,8 @@ complete_queue(BeamformerCtx *ctx, BeamformWorkQueue *q, Arena *arena) gpu_command_copy_buffer(cmd, &cs->ping_pong_buffer, input_index * pp_size, &rf->buffer, copy_offset, copy_size); gpu_command_clear_buffer(cmd, &cs->ping_pong_buffer, input_index * pp_size + copy_size, clear_size, 0); + // TODO(rnp): technically this should be a transfer to compute barrier + // but a full memory barrier will also do the trick gpu_command_pipeline_barrier(cmd, 1); } @@ -1581,14 +1589,14 @@ complete_queue(BeamformerCtx *ctx, BeamformWorkQueue *q, Arena *arena) u64 output_pointer = pp_output_pointer; u64 input_pointer = (i == 0 && !special_handling) ? rf_pointer : pp_input_pointer; - if (i != 0) gpu_command_pipeline_barrier(cmd, 0); + if (i != 0) gpu_command_pipeline_barrier(cmd, pipeline_memory_barrier); do_compute_shader(cmd, cp, frame, input_pointer, output_pointer, i, channel_offset); gpu_command_timestamp(cmd); } } for (u32 i = cp->first_image_shader_index; i < cp->pipeline.shader_count; i++) { - gpu_command_pipeline_barrier(cmd, 0); + gpu_command_pipeline_barrier(cmd, pipeline_memory_barrier); do_compute_shader(cmd, cp, frame, 0, 0, i, 0); gpu_command_timestamp(cmd); }