ogl_beamforming

Ultrasound Beamforming Implemented with OpenGL
git clone anongit@rnpnr.xyz:ogl_beamforming.git
Log | Files | Refs | Feed | Submodules | README | LICENSE

Commit: 679ea5e28947e920d8a5c7c34c30e624f7fe73c2
Parent: 13828706d3d607c7144156c091f42395299bcd9d
Author: Randy Palamar
Date:   Sat, 12 Sep 2026 07:11:44 -0700

core: reduce ping pong buffer slots back to 2

The third slot is only needed if it is possible to allow overlap
of all pre DAS-shaders with the previous batch of DAS. As I have
yet to be successful in achieving this the 3rd slot is just wasted
memory. Reduce this back to 2 and re-enable filter/das overlap for
TPW/VLS.

Diffstat:
Mbeamformer.meta | 2+-
Mbeamformer_core.c | 16+++++-----------
Mbeamformer_internal.h | 12+++---------
Mgenerated/beamformer.c | 35+++++++++++++++++------------------
Mshaders/das.glsl | 8++++----
5 files changed, 30 insertions(+), 43 deletions(-)

diff --git a/beamformer.meta b/beamformer.meta @@ -394,7 +394,6 @@ @Bake { - [RFData U64] [FocalVectors U32] [Hadamard U32] [IncoherentFrame U32] @@ -430,6 +429,7 @@ [xdc_transform M4] [voxel_transform M4] [xdc_element_pitch V2] + [rf_data U64] [output_frame U64] [channel_offset S32] [readi_group U32] diff --git a/beamformer_core.c b/beamformer_core.c @@ -2,8 +2,6 @@ /* TODO(rnp): * [ ]: backtrace dumping on SIGSEGV * [ ]: cooperative shared memory loading in decode shader - * [ ]: refactor: when there are only two beamforming shaders switch back to ping pong input - * for DAS instead of fixed region to allow overlap with first stage * [ ]: refactor: save filter parameters with rest of parameters, whole slot thing is dumb * [ ]: upload previously exported data for display. maybe this is a UI thing but doing it * programatically would be nice. @@ -905,9 +903,6 @@ plan_compute_pipeline(BeamformerComputePlan *cp, BeamformerParameterBlock *pb, A db->OutputSizeZ = cp->output_points.z; db->TransmitReceiveOrientation = pb->parameters.transmit_receive_orientation; - u64 pp_size = beamformer_context->compute_context.ping_pong_buffer.size / PING_PONG_BUFFER_SLOTS; - db->RFData = beamformer_context->compute_context.ping_pong_buffer.gpu_pointer + (PING_PONG_BUFFER_SLOTS - 1) * pp_size; - // NOTE(rnp): old gcc will miscompile an assignment memory_copy(cp->xdc_transform.E, pb->parameters.xdc_transform.E, sizeof(cp->xdc_transform)); @@ -1293,6 +1288,7 @@ do_compute_shader(GPUCommandList cmd, BeamformerComputePlan *cp, BeamformerFrame case BeamformerShaderKind_DAS:{ BeamformerDASPushConstants pc = { .xdc_element_pitch = cp->xdc_element_pitch, + .rf_data = input_pointer, .output_frame = frame->gpu_pointer, .channel_offset = channel_offset, .readi_group = cp->readi_group, @@ -1571,17 +1567,15 @@ complete_queue(BeamformerCtx *ctx, BeamformWorkQueue *q, Arena *arena) for (u32 i = 0; i < cp->first_image_shader_index; i++) { u32 output_index = !cs->ping_pong_input_index; u32 input_index = cs->ping_pong_input_index; - u32 das_output_index = PING_PONG_BUFFER_SLOTS - 1; u64 pp_size = cs->ping_pong_buffer.size / PING_PONG_BUFFER_SLOTS; - u64 pp_input_pointer = cs->ping_pong_buffer.gpu_pointer + input_index * pp_size; - u64 pp_output_pointer = cs->ping_pong_buffer.gpu_pointer + output_index * pp_size; - u64 pp_das_pointer = cs->ping_pong_buffer.gpu_pointer + das_output_index * pp_size; + u64 pp_input_pointer = cs->ping_pong_buffer.gpu_pointer + input_index * pp_size; + u64 pp_output_pointer = cs->ping_pong_buffer.gpu_pointer + output_index * pp_size; - u64 output_pointer = ((i + 1) == (u32)das_index) ? pp_das_pointer : pp_output_pointer; + u64 output_pointer = pp_output_pointer; u64 input_pointer = (i == 0 && !special_handling) ? rf_pointer : pp_input_pointer; - if (i != 0 || i == ((u32)das_index - 1)) gpu_command_pipeline_barrier(cmd, 0); + if (i != 0) gpu_command_pipeline_barrier(cmd, 0); do_compute_shader(cmd, cp, frame, input_pointer, output_pointer, i, channel_offset); gpu_command_timestamp(cmd); } diff --git a/beamformer_internal.h b/beamformer_internal.h @@ -406,7 +406,7 @@ typedef struct { } BeamformerFrame; /* NOTE(rnp): backing storage for beamformed frames. The amount of backlog frames -* is dependant on the currently requested output size. */ + * is dependant on the currently requested output size. */ typedef struct { GPUBuffer buffer[1]; @@ -422,14 +422,8 @@ typedef struct { BeamformerComputePlan *compute_plans[BeamformerMaxParameterBlocks]; BeamformerComputePlan *compute_plan_freelist; - /* NOTE(rnp): used to ping pong data between compute stages. - * - * Allocate one extra slot for DAS output to allow overlap with the next - * channel chunk batch. To obtain optimal overlap we need 2 extra slots - * and we need to ping pong submissions between queues. This is not - * implemented so we only do 1 extra slot for now. - */ - #define PING_PONG_BUFFER_SLOTS (2 + 1) + // NOTE(rnp): used to ping pong data between compute stages. + #define PING_PONG_BUFFER_SLOTS 2 GPUBuffer ping_pong_buffer; OSHandle ping_pong_export_handle; u32 ping_pong_input_index; diff --git a/generated/beamformer.c b/generated/beamformer.c @@ -189,7 +189,6 @@ typedef struct { } BeamformerFilterBakeParameters; typedef struct { - u64 RFData; u32 FocalVectors; u32 Hadamard; u32 IncoherentFrame; @@ -247,6 +246,7 @@ typedef struct { m4 xdc_transform; m4 voxel_transform; v2 xdc_element_pitch; + u64 rf_data; u64 output_frame; i32 channel_offset; u32 readi_group; @@ -605,30 +605,29 @@ read_only global MetaStructMember *meta_struct_members_by_id[] = { {18, 48, 1, 0}, }, (MetaStructMember []){ - {17, 0, 1, 0}, + {18, 0, 1, 0}, + {18, 4, 1, 0}, {18, 8, 1, 0}, {18, 12, 1, 0}, {18, 16, 1, 0}, {18, 20, 1, 0}, - {18, 24, 1, 0}, - {18, 28, 1, 0}, + {10, 24, 1, 0}, + {10, 28, 1, 0}, {10, 32, 1, 0}, {10, 36, 1, 0}, - {10, 40, 1, 0}, - {10, 44, 1, 0}, + {8, 40, 1, 0}, + {8, 44, 1, 0}, {8, 48, 1, 0}, {8, 52, 1, 0}, - {8, 56, 1, 0}, + {18, 56, 1, 0}, {8, 60, 1, 0}, {18, 64, 1, 0}, {8, 68, 1, 0}, - {18, 72, 1, 0}, - {8, 76, 1, 0}, - {8, 80, 1, 0}, + {8, 72, 1, 0}, + {18, 76, 1, 0}, + {18, 80, 1, 0}, {18, 84, 1, 0}, {18, 88, 1, 0}, - {18, 92, 1, 0}, - {18, 96, 1, 0}, }, (MetaStructMember []){ {18, 0, 1, 0}, @@ -678,7 +677,6 @@ read_only global str8 *meta_struct_member_names_by_id[] = { str8_comp("OutputTransmitStride"), }, (str8 []){ - str8_comp("RFData"), str8_comp("FocalVectors"), str8_comp("Hadamard"), str8_comp("IncoherentFrame"), @@ -722,11 +720,11 @@ read_only global str8 *meta_struct_member_names_by_id[] = { }; read_only global MetaStructInfo meta_struct_info_by_id[] = { - {str8_comp("DecodeBakeParameters"), 11, 44, 0}, - {str8_comp("FilterBakeParameters"), 13, 52, 0}, - {str8_comp("DASBakeParameters"), 24, 100, 0}, - {str8_comp("CoherencyWeightingBakeParameters"), 3, 12, 0}, - {str8_comp("ReshapeBakeParameters"), 9, 36, 0}, + {str8_comp("DecodeBakeParameters"), 11, 44, 0}, + {str8_comp("FilterBakeParameters"), 13, 52, 0}, + {str8_comp("DASBakeParameters"), 23, 92, 0}, + {str8_comp("CoherencyWeightingBakeParameters"), 3, 12, 0}, + {str8_comp("ReshapeBakeParameters"), 9, 36, 0}, }; read_only global str8 beamformer_shader_names[] = { @@ -855,6 +853,7 @@ read_only global str8 beamformer_shader_global_header_strings[] = { " f32mat4 xdc_transform;\n" " f32mat4 voxel_transform;\n" " f32vec2 xdc_element_pitch;\n" + " uint64_t rf_data;\n" " uint64_t output_frame;\n" " int32_t channel_offset;\n" " uint32_t readi_group;\n" diff --git a/shaders/das.glsl b/shaders/das.glsl @@ -247,7 +247,7 @@ RESULT_TYPE RCA(const vec3 world_point) vec2 xdc_world_point = rca_plane_projection((xdc_transform * vec4(world_point, 1)).xyz, rx_rows); f32 transmit_index = sample_index(rca_transmit_distance(world_point, focal_vector, tx_rx_orientation)); - u64 rf_pointer = RFData + InputDataKindByteSize * acquisition * SampleCount; + u64 rf_pointer = rf_data + InputDataKindByteSize * acquisition * SampleCount; rf_pointer -= InputDataKindByteSize * u32(InterpolationMode == InterpolationMode_Cubic); for (f32 chunk_channel = 0.f; chunk_channel < f32(batch_channel_count()); chunk_channel += 1.f) { @@ -291,7 +291,7 @@ RESULT_TYPE HERCULES(const vec3 world_point) f32 element_receive_delta_squared = rx_world_point - rx_channel * rx_pitch; element_receive_delta_squared *= element_receive_delta_squared; - u64 rf_pointer = RFData + InputDataKindByteSize * (u32(chunk_channel) * SampleCount * AcquisitionCount + u32(Sparse) * SampleCount); + u64 rf_pointer = rf_data + InputDataKindByteSize * (u32(chunk_channel) * SampleCount * AcquisitionCount + u32(Sparse) * SampleCount); rf_pointer -= InputDataKindByteSize * u32(InterpolationMode == InterpolationMode_Cubic); for (f32 transmit = f32(Sparse); transmit < f32(AcquisitionCount); transmit += 1.f) { @@ -335,7 +335,7 @@ RESULT_TYPE FORCES(const vec3 world_point) f32 a_arg = abs(FNumber * receive_x_delta / xdc_world_point.z); if (a_arg < 0.5f) { - u64 rf_pointer = RFData + InputDataKindByteSize * (u32(chunk_channel) * SampleCount * AcquisitionCount + u32(Sparse) * SampleCount); + u64 rf_pointer = rf_data + InputDataKindByteSize * (u32(chunk_channel) * SampleCount * AcquisitionCount + u32(Sparse) * SampleCount); rf_pointer -= InputDataKindByteSize * u32(InterpolationMode == InterpolationMode_Cubic); f32 receive_index = sample_index(sqrt(receive_x_delta * receive_x_delta + z_delta_squared)); @@ -375,7 +375,7 @@ RESULT_TYPE READI_FORCES(const vec3 world_point) f32 a_arg = abs(FNumber * receive_x_delta / xdc_world_point.z); if (a_arg < 0.5f) { - u64 channel_rf_pointer = RFData + InputDataKindByteSize * u32(chunk_channel) * SampleCount * AcquisitionCount; + u64 channel_rf_pointer = rf_data + InputDataKindByteSize * u32(chunk_channel) * SampleCount * AcquisitionCount; channel_rf_pointer -= InputDataKindByteSize * u32(InterpolationMode == InterpolationMode_Cubic); f32 receive_index = sample_index(sqrt(receive_x_delta * receive_x_delta + z_delta_squared));