Commit: 679ea5e28947e920d8a5c7c34c30e624f7fe73c2
Parent: 13828706d3d607c7144156c091f42395299bcd9d
Author: Randy Palamar
Date: Sat, 12 Sep 2026 07:11:44 -0700
core: reduce ping pong buffer slots back to 2
The third slot is only needed if it is possible to allow overlap
of all pre DAS-shaders with the previous batch of DAS. As I have
yet to be successful in achieving this the 3rd slot is just wasted
memory. Reduce this back to 2 and re-enable filter/das overlap for
TPW/VLS.
Diffstat:
5 files changed, 30 insertions(+), 43 deletions(-)
diff --git a/beamformer.meta b/beamformer.meta
@@ -394,7 +394,6 @@
@Bake
{
- [RFData U64]
[FocalVectors U32]
[Hadamard U32]
[IncoherentFrame U32]
@@ -430,6 +429,7 @@
[xdc_transform M4]
[voxel_transform M4]
[xdc_element_pitch V2]
+ [rf_data U64]
[output_frame U64]
[channel_offset S32]
[readi_group U32]
diff --git a/beamformer_core.c b/beamformer_core.c
@@ -2,8 +2,6 @@
/* TODO(rnp):
* [ ]: backtrace dumping on SIGSEGV
* [ ]: cooperative shared memory loading in decode shader
- * [ ]: refactor: when there are only two beamforming shaders switch back to ping pong input
- * for DAS instead of fixed region to allow overlap with first stage
* [ ]: refactor: save filter parameters with rest of parameters, whole slot thing is dumb
* [ ]: upload previously exported data for display. maybe this is a UI thing but doing it
* programatically would be nice.
@@ -905,9 +903,6 @@ plan_compute_pipeline(BeamformerComputePlan *cp, BeamformerParameterBlock *pb, A
db->OutputSizeZ = cp->output_points.z;
db->TransmitReceiveOrientation = pb->parameters.transmit_receive_orientation;
- u64 pp_size = beamformer_context->compute_context.ping_pong_buffer.size / PING_PONG_BUFFER_SLOTS;
- db->RFData = beamformer_context->compute_context.ping_pong_buffer.gpu_pointer + (PING_PONG_BUFFER_SLOTS - 1) * pp_size;
-
// NOTE(rnp): old gcc will miscompile an assignment
memory_copy(cp->xdc_transform.E, pb->parameters.xdc_transform.E, sizeof(cp->xdc_transform));
@@ -1293,6 +1288,7 @@ do_compute_shader(GPUCommandList cmd, BeamformerComputePlan *cp, BeamformerFrame
case BeamformerShaderKind_DAS:{
BeamformerDASPushConstants pc = {
.xdc_element_pitch = cp->xdc_element_pitch,
+ .rf_data = input_pointer,
.output_frame = frame->gpu_pointer,
.channel_offset = channel_offset,
.readi_group = cp->readi_group,
@@ -1571,17 +1567,15 @@ complete_queue(BeamformerCtx *ctx, BeamformWorkQueue *q, Arena *arena)
for (u32 i = 0; i < cp->first_image_shader_index; i++) {
u32 output_index = !cs->ping_pong_input_index;
u32 input_index = cs->ping_pong_input_index;
- u32 das_output_index = PING_PONG_BUFFER_SLOTS - 1;
u64 pp_size = cs->ping_pong_buffer.size / PING_PONG_BUFFER_SLOTS;
- u64 pp_input_pointer = cs->ping_pong_buffer.gpu_pointer + input_index * pp_size;
- u64 pp_output_pointer = cs->ping_pong_buffer.gpu_pointer + output_index * pp_size;
- u64 pp_das_pointer = cs->ping_pong_buffer.gpu_pointer + das_output_index * pp_size;
+ u64 pp_input_pointer = cs->ping_pong_buffer.gpu_pointer + input_index * pp_size;
+ u64 pp_output_pointer = cs->ping_pong_buffer.gpu_pointer + output_index * pp_size;
- u64 output_pointer = ((i + 1) == (u32)das_index) ? pp_das_pointer : pp_output_pointer;
+ u64 output_pointer = pp_output_pointer;
u64 input_pointer = (i == 0 && !special_handling) ? rf_pointer : pp_input_pointer;
- if (i != 0 || i == ((u32)das_index - 1)) gpu_command_pipeline_barrier(cmd, 0);
+ if (i != 0) gpu_command_pipeline_barrier(cmd, 0);
do_compute_shader(cmd, cp, frame, input_pointer, output_pointer, i, channel_offset);
gpu_command_timestamp(cmd);
}
diff --git a/beamformer_internal.h b/beamformer_internal.h
@@ -406,7 +406,7 @@ typedef struct {
} BeamformerFrame;
/* NOTE(rnp): backing storage for beamformed frames. The amount of backlog frames
-* is dependant on the currently requested output size. */
+ * is dependant on the currently requested output size. */
typedef struct {
GPUBuffer buffer[1];
@@ -422,14 +422,8 @@ typedef struct {
BeamformerComputePlan *compute_plans[BeamformerMaxParameterBlocks];
BeamformerComputePlan *compute_plan_freelist;
- /* NOTE(rnp): used to ping pong data between compute stages.
- *
- * Allocate one extra slot for DAS output to allow overlap with the next
- * channel chunk batch. To obtain optimal overlap we need 2 extra slots
- * and we need to ping pong submissions between queues. This is not
- * implemented so we only do 1 extra slot for now.
- */
- #define PING_PONG_BUFFER_SLOTS (2 + 1)
+ // NOTE(rnp): used to ping pong data between compute stages.
+ #define PING_PONG_BUFFER_SLOTS 2
GPUBuffer ping_pong_buffer;
OSHandle ping_pong_export_handle;
u32 ping_pong_input_index;
diff --git a/generated/beamformer.c b/generated/beamformer.c
@@ -189,7 +189,6 @@ typedef struct {
} BeamformerFilterBakeParameters;
typedef struct {
- u64 RFData;
u32 FocalVectors;
u32 Hadamard;
u32 IncoherentFrame;
@@ -247,6 +246,7 @@ typedef struct {
m4 xdc_transform;
m4 voxel_transform;
v2 xdc_element_pitch;
+ u64 rf_data;
u64 output_frame;
i32 channel_offset;
u32 readi_group;
@@ -605,30 +605,29 @@ read_only global MetaStructMember *meta_struct_members_by_id[] = {
{18, 48, 1, 0},
},
(MetaStructMember []){
- {17, 0, 1, 0},
+ {18, 0, 1, 0},
+ {18, 4, 1, 0},
{18, 8, 1, 0},
{18, 12, 1, 0},
{18, 16, 1, 0},
{18, 20, 1, 0},
- {18, 24, 1, 0},
- {18, 28, 1, 0},
+ {10, 24, 1, 0},
+ {10, 28, 1, 0},
{10, 32, 1, 0},
{10, 36, 1, 0},
- {10, 40, 1, 0},
- {10, 44, 1, 0},
+ {8, 40, 1, 0},
+ {8, 44, 1, 0},
{8, 48, 1, 0},
{8, 52, 1, 0},
- {8, 56, 1, 0},
+ {18, 56, 1, 0},
{8, 60, 1, 0},
{18, 64, 1, 0},
{8, 68, 1, 0},
- {18, 72, 1, 0},
- {8, 76, 1, 0},
- {8, 80, 1, 0},
+ {8, 72, 1, 0},
+ {18, 76, 1, 0},
+ {18, 80, 1, 0},
{18, 84, 1, 0},
{18, 88, 1, 0},
- {18, 92, 1, 0},
- {18, 96, 1, 0},
},
(MetaStructMember []){
{18, 0, 1, 0},
@@ -678,7 +677,6 @@ read_only global str8 *meta_struct_member_names_by_id[] = {
str8_comp("OutputTransmitStride"),
},
(str8 []){
- str8_comp("RFData"),
str8_comp("FocalVectors"),
str8_comp("Hadamard"),
str8_comp("IncoherentFrame"),
@@ -722,11 +720,11 @@ read_only global str8 *meta_struct_member_names_by_id[] = {
};
read_only global MetaStructInfo meta_struct_info_by_id[] = {
- {str8_comp("DecodeBakeParameters"), 11, 44, 0},
- {str8_comp("FilterBakeParameters"), 13, 52, 0},
- {str8_comp("DASBakeParameters"), 24, 100, 0},
- {str8_comp("CoherencyWeightingBakeParameters"), 3, 12, 0},
- {str8_comp("ReshapeBakeParameters"), 9, 36, 0},
+ {str8_comp("DecodeBakeParameters"), 11, 44, 0},
+ {str8_comp("FilterBakeParameters"), 13, 52, 0},
+ {str8_comp("DASBakeParameters"), 23, 92, 0},
+ {str8_comp("CoherencyWeightingBakeParameters"), 3, 12, 0},
+ {str8_comp("ReshapeBakeParameters"), 9, 36, 0},
};
read_only global str8 beamformer_shader_names[] = {
@@ -855,6 +853,7 @@ read_only global str8 beamformer_shader_global_header_strings[] = {
" f32mat4 xdc_transform;\n"
" f32mat4 voxel_transform;\n"
" f32vec2 xdc_element_pitch;\n"
+ " uint64_t rf_data;\n"
" uint64_t output_frame;\n"
" int32_t channel_offset;\n"
" uint32_t readi_group;\n"
diff --git a/shaders/das.glsl b/shaders/das.glsl
@@ -247,7 +247,7 @@ RESULT_TYPE RCA(const vec3 world_point)
vec2 xdc_world_point = rca_plane_projection((xdc_transform * vec4(world_point, 1)).xyz, rx_rows);
f32 transmit_index = sample_index(rca_transmit_distance(world_point, focal_vector, tx_rx_orientation));
- u64 rf_pointer = RFData + InputDataKindByteSize * acquisition * SampleCount;
+ u64 rf_pointer = rf_data + InputDataKindByteSize * acquisition * SampleCount;
rf_pointer -= InputDataKindByteSize * u32(InterpolationMode == InterpolationMode_Cubic);
for (f32 chunk_channel = 0.f; chunk_channel < f32(batch_channel_count()); chunk_channel += 1.f) {
@@ -291,7 +291,7 @@ RESULT_TYPE HERCULES(const vec3 world_point)
f32 element_receive_delta_squared = rx_world_point - rx_channel * rx_pitch;
element_receive_delta_squared *= element_receive_delta_squared;
- u64 rf_pointer = RFData + InputDataKindByteSize * (u32(chunk_channel) * SampleCount * AcquisitionCount + u32(Sparse) * SampleCount);
+ u64 rf_pointer = rf_data + InputDataKindByteSize * (u32(chunk_channel) * SampleCount * AcquisitionCount + u32(Sparse) * SampleCount);
rf_pointer -= InputDataKindByteSize * u32(InterpolationMode == InterpolationMode_Cubic);
for (f32 transmit = f32(Sparse); transmit < f32(AcquisitionCount); transmit += 1.f) {
@@ -335,7 +335,7 @@ RESULT_TYPE FORCES(const vec3 world_point)
f32 a_arg = abs(FNumber * receive_x_delta / xdc_world_point.z);
if (a_arg < 0.5f) {
- u64 rf_pointer = RFData + InputDataKindByteSize * (u32(chunk_channel) * SampleCount * AcquisitionCount + u32(Sparse) * SampleCount);
+ u64 rf_pointer = rf_data + InputDataKindByteSize * (u32(chunk_channel) * SampleCount * AcquisitionCount + u32(Sparse) * SampleCount);
rf_pointer -= InputDataKindByteSize * u32(InterpolationMode == InterpolationMode_Cubic);
f32 receive_index = sample_index(sqrt(receive_x_delta * receive_x_delta + z_delta_squared));
@@ -375,7 +375,7 @@ RESULT_TYPE READI_FORCES(const vec3 world_point)
f32 a_arg = abs(FNumber * receive_x_delta / xdc_world_point.z);
if (a_arg < 0.5f) {
- u64 channel_rf_pointer = RFData + InputDataKindByteSize * u32(chunk_channel) * SampleCount * AcquisitionCount;
+ u64 channel_rf_pointer = rf_data + InputDataKindByteSize * u32(chunk_channel) * SampleCount * AcquisitionCount;
channel_rf_pointer -= InputDataKindByteSize * u32(InterpolationMode == InterpolationMode_Cubic);
f32 receive_index = sample_index(sqrt(receive_x_delta * receive_x_delta + z_delta_squared));