ogl_beamforming

Ultrasound Beamforming Implemented with OpenGL
git clone anongit@rnpnr.xyz:ogl_beamforming.git
Log | Files | Refs | Feed | Submodules | README | LICENSE

Commit: 91078ec7d254d9b5d0d21c673a8205cf4eb0e6cb
Parent: f17ace4486eb20f3002d5a61d7a0acb3331b03ae
Author: Randy Palamar
Date:   Sat, 26 Sep 2026 08:32:00 -0700

core: make rf data shuffling more flexible & robust

while the existing shuffling handling was working fine it was
difficult to keep track of what was being combined to form each
stride. This makes it difficult to explore other layouts as well
which may be needed to improve performance in DAS.

DAS is now passed explicit Channel and Acquisition byte strides
which are used instead of the fixed/assumed layout. For now the
sample stride is still assumed to be 1 as I don't see any scenario
where issuing more loads will help performance. Swapping the
Channels and Acquisitions gives minor improvement in the TPW/VLS
path due to its loop structure so that is now done.

Diffstat:
Mbeamformer.meta | 2++
Mbeamformer_core.c | 213++++++++++++++++++++++++++++++++++++++++++++++++++-----------------------------
Mbeamformer_core.meta | 16++++++++++++++++
Mgenerated/beamformer.c | 28+++++++++++++++++-----------
Mgenerated/beamformer_core.c | 16++++++++++++++++
Mshaders/das.glsl | 27+++++++++++++++------------
6 files changed, 202 insertions(+), 100 deletions(-)

diff --git a/beamformer.meta b/beamformer.meta @@ -405,6 +405,8 @@ [ReceiveChannelCount S32] [ChunkChannelCount S32] [SampleCount S32] + [AcquisitionByteStride U32] + [ChannelByteStride U32] [SamplingFrequency F32] [DemodulationFrequency F32] diff --git a/beamformer_core.c b/beamformer_core.c @@ -467,15 +467,17 @@ struct BeamformerComputeGraphNode { // the shader requires a fixed layout for input, output, or both. When two adjacent // nodes require incompatible layouts the second pass over the graph will insert // Reshape shaders in between. - BeamformerDataKind input_data_kind; - iv3 input_stride; + BeamformerDataKind input_data_kind; + BeamformerDataLayout input_data_layout; - BeamformerDataKind output_data_kind; - iv3 output_stride; + BeamformerDataKind output_data_kind; + BeamformerDataLayout output_data_layout; - b32 requires_fp_input; + uv3 rf_dimensions; - i32 user_pipeline_index; + b32 requires_fp_input; + + i32 user_pipeline_index; BeamformerComputeGraphNode *prev; BeamformerComputeGraphNode *next; @@ -510,14 +512,59 @@ push_compute_graph_node(BeamformerComputeGraph *graph, BeamformerShaderKind kind { BeamformerComputeGraphNode *result = push_struct(arena, BeamformerComputeGraphNode); if (graph) { + if (graph->last) result->rf_dimensions = graph->last->rf_dimensions; DLLInsertLast(0, graph->first, graph->last, result, next, prev); graph->count++; } result->kind = kind; result->user_pipeline_index = -1; - // NOTE(rnp): initially don't care data kind - result->input_data_kind = BeamformerDataKind_Count; - result->output_data_kind = BeamformerDataKind_Count; + // NOTE(rnp): initially don't care data parameters + result->input_data_kind = BeamformerDataKind_Count; + result->output_data_kind = BeamformerDataKind_Count; + result->input_data_layout = BeamformerDataLayout_Count; + result->output_data_layout = BeamformerDataLayout_Count; + return result; +} + +function uv3 +beamformer_data_strides(BeamformerDataLayout layout, uv3 rf_dimensions) +{ + u32 channel_count = rf_dimensions.E[BeamformerRFDimension_Channels]; + u32 event_count = rf_dimensions.E[BeamformerRFDimension_Events]; + u32 sample_count = rf_dimensions.E[BeamformerRFDimension_Samples]; + + uv3 result = {0}; + switch (layout) { + InvalidDefaultCase; + + case BeamformerDataLayout_Image:{}break; + + case BeamformerDataLayout_ChannelEventSample:{ + result.E[BeamformerRFDimension_Channels] = event_count * sample_count; + result.E[BeamformerRFDimension_Events] = sample_count; + result.E[BeamformerRFDimension_Samples] = 1; + }break; + + case BeamformerDataLayout_ChannelSampleEvent:{ + result.E[BeamformerRFDimension_Channels] = event_count * sample_count; + result.E[BeamformerRFDimension_Events] = 1; + result.E[BeamformerRFDimension_Samples] = event_count; + }break; + + case BeamformerDataLayout_EventChannelSample:{ + result.E[BeamformerRFDimension_Events] = channel_count * sample_count; + result.E[BeamformerRFDimension_Channels] = sample_count; + result.E[BeamformerRFDimension_Samples] = 1; + }break; + + case BeamformerDataLayout_SampleChannelEvent:{ + result.E[BeamformerRFDimension_Samples] = channel_count * event_count; + result.E[BeamformerRFDimension_Channels] = event_count; + result.E[BeamformerRFDimension_Events] = 1; + }break; + + } + return result; } @@ -607,14 +654,14 @@ plan_compute_pipeline(BeamformerComputePlan *cp, BeamformerParameterBlock *pb, A // NOTE(rnp): First Pass: build initial graph and insert hard layout constraints BeamformerComputeGraph graph = {0}; BeamformerComputeGraphNode *root_node = push_compute_graph_node(&graph, BeamformerShaderKind_Count, scratch); - root_node->input_data_kind = input_data_kind; - root_node->input_stride.x = 1; // Sample Stride - root_node->input_stride.y = pb->parameters.sample_count * acquisition_count; // Channel Stride - root_node->input_stride.z = pb->parameters.sample_count; // Receive Event Stride - root_node->output_data_kind = input_data_kind; - root_node->output_stride.x = 1; // Sample Stride - root_node->output_stride.y = pb->parameters.sample_count * acquisition_count; // Channel Stride - root_node->output_stride.z = pb->parameters.sample_count; // Receive Event Stride + root_node->input_data_kind = input_data_kind; + root_node->input_data_layout = BeamformerDataLayout_ChannelEventSample; + root_node->output_data_kind = input_data_kind; + root_node->output_data_layout = BeamformerDataLayout_ChannelEventSample; + + root_node->rf_dimensions.E[BeamformerRFDimension_Channels] = chunk_channel_count; + root_node->rf_dimensions.E[BeamformerRFDimension_Events] = pb->parameters.acquisition_count; + root_node->rf_dimensions.E[BeamformerRFDimension_Samples] = pb->parameters.sample_count; for EachIndex(pb->pipeline.shader_count, it) { // NOTE(rnp): skip unnecessary shaders @@ -639,6 +686,10 @@ plan_compute_pipeline(BeamformerComputePlan *cp, BeamformerParameterBlock *pb, A BeamformerComputeGraphNode *node = push_compute_graph_node(&graph, pb->pipeline.shaders[it], scratch); node->user_pipeline_index = (i32)it; switch (pb->pipeline.shaders[it]) { + case BeamformerShaderKind_Demodulate:{ + node->rf_dimensions.E[BeamformerRFDimension_Samples] /= (2 * decimation_rate); + }break; + case BeamformerShaderKind_Decode:{ b32 low_precision = beamformer_data_kind_element_size[input_data_kind] < 4; b32 use_coop_matrix = gpu_info()->cooperative_matrix && @@ -649,14 +700,12 @@ plan_compute_pipeline(BeamformerComputePlan *cp, BeamformerParameterBlock *pb, A node->requires_fp_input = 1; // NOTE(rnp): fixed input layout required for reasonable performance - node->input_stride.x = chunk_channel_count * acquisition_count; - node->input_stride.y = acquisition_count; - node->input_stride.z = 1; + node->input_data_layout = BeamformerDataLayout_SampleChannelEvent; if (use_coop_matrix) { - node->input_data_kind = BeamformerDataKind_Float16; - node->output_data_kind = data_kind_to_element_kind[das_data_kind]; - node->output_stride = node->input_stride; + node->input_data_kind = BeamformerDataKind_Float16; + node->output_data_kind = data_kind_to_element_kind[das_data_kind]; + node->output_data_layout = node->input_data_layout; } }break; @@ -667,13 +716,15 @@ plan_compute_pipeline(BeamformerComputePlan *cp, BeamformerParameterBlock *pb, A if (pb->parameters.decode_mode != BeamformerDecodeMode_None) node->input_data_kind = das_data_kind; - node->input_stride.x = 1; // Sample Stride - node->input_stride.y = input_sample_count * acquisition_count; // Channel Stride - node->input_stride.z = input_sample_count; // Receive Event Stride - node->output_stride.x = 1; - node->output_stride.y = cp->output_points.x; - node->output_stride.z = cp->output_points.x * cp->output_points.y; - node->output_data_kind = das_data_kind; + node->input_data_layout = BeamformerDataLayout_ChannelEventSample; + if (pb->parameters.acquisition_kind == BeamformerAcquisitionKind_RCA_TPW || + pb->parameters.acquisition_kind == BeamformerAcquisitionKind_RCA_VLS) + { + node->input_data_layout = BeamformerDataLayout_EventChannelSample; + } + + node->output_data_layout = BeamformerDataLayout_Image; + node->output_data_kind = das_data_kind; // NOTE(rnp): insert implicit CoherencyWeighting node if (pb->parameters.coherency_weighting) @@ -691,19 +742,19 @@ plan_compute_pipeline(BeamformerComputePlan *cp, BeamformerParameterBlock *pb, A // NOTE(rnp): data strides { - b32 input_dont_care = bv3_any(iv3_equal(node->input_stride, (iv3){0})); - b32 prev_output_dont_care = bv3_any(iv3_equal(node->prev->output_stride, (iv3){0})); + b32 input_dont_care = node->input_data_layout == BeamformerDataLayout_Count; + b32 prev_output_dont_care = node->prev->output_data_layout == BeamformerDataLayout_Count; if (prev_output_dont_care && !input_dont_care) - node->prev->output_stride = node->input_stride; + node->prev->output_data_layout = node->input_data_layout; if (!prev_output_dont_care && input_dont_care) - node->input_stride = node->prev->output_stride; + node->input_data_layout = node->prev->output_data_layout; if (prev_output_dont_care && input_dont_care) - node->input_stride = node->prev->output_stride = node->prev->input_stride; + node->input_data_layout = node->prev->output_data_layout = node->prev->input_data_layout; - needs_reshape |= !bv3_all(iv3_equal(node->input_stride, node->prev->output_stride)); + needs_reshape |= node->input_data_layout != node->prev->output_data_layout; } // NOTE(rnp): data kinds @@ -735,10 +786,11 @@ plan_compute_pipeline(BeamformerComputePlan *cp, BeamformerParameterBlock *pb, A BeamformerComputeGraphNode *last = node->prev; DLLInsertLast(0, node, last, new, next, prev); graph.count++; - new->input_data_kind = new->prev->output_data_kind; - new->input_stride = new->prev->output_stride; - new->output_data_kind = new->next->input_data_kind; - new->output_stride = new->next->input_stride; + new->rf_dimensions = new->prev->rf_dimensions; + new->input_data_kind = new->prev->output_data_kind; + new->input_data_layout = new->prev->output_data_layout; + new->output_data_kind = new->next->input_data_kind; + new->output_data_layout = new->next->input_data_layout; } } @@ -755,7 +807,7 @@ plan_compute_pipeline(BeamformerComputePlan *cp, BeamformerParameterBlock *pb, A GPUResourceBuilder *resource_builder = gpu_resource_build_begin(scratch); for (BeamformerComputeGraphNode *node = root_node->next; node; node = node->next) { assert(node->prev->output_data_kind == node->input_data_kind); - assert(bv3_all(iv3_equal(node->prev->output_stride, node->input_stride))); + assert(node->prev->output_data_layout == node->input_data_layout); BeamformerShaderParameters *sp = 0; if (node->user_pipeline_index >= 0) @@ -764,19 +816,22 @@ plan_compute_pipeline(BeamformerComputePlan *cp, BeamformerParameterBlock *pb, A if (compute_plan_push_shader(cp, node, sp)) { BeamformerShaderDescriptor *sd = cp->shader_descriptors + cp->pipeline.shader_count - 1; + uv3 output_stride = beamformer_data_strides(node->output_data_layout, node->rf_dimensions); + uv3 input_stride = beamformer_data_strides(node->input_data_layout, node->prev->rf_dimensions); + switch (node->kind) { case BeamformerShaderKind_Decode:{ BeamformerDecodeBakeParameters *db = &sd->bake.Decode; - u32 decode_sample_count = input_sample_count; - db->DecodeMode = pb->parameters.decode_mode; - db->TransmitCount = pb->parameters.acquisition_count; - db->ChunkChannelCount = chunk_channel_count; + u32 decode_sample_count = node->rf_dimensions.E[BeamformerRFDimension_Samples]; + db->DecodeMode = pb->parameters.decode_mode; + db->TransmitCount = node->rf_dimensions.E[BeamformerRFDimension_Events]; + db->ChunkChannelCount = node->rf_dimensions.E[BeamformerRFDimension_Channels]; // NOTE(rnp): ignored when using coop matrices - db->OutputSampleStride = node->output_stride.x; - db->OutputChannelStride = node->output_stride.y; - db->OutputTransmitStride = node->output_stride.z; + db->OutputSampleStride = output_stride.E[BeamformerRFDimension_Samples]; + db->OutputChannelStride = output_stride.E[BeamformerRFDimension_Channels]; + db->OutputTransmitStride = output_stride.E[BeamformerRFDimension_Events]; db->ToProcess = 1; @@ -811,21 +866,21 @@ plan_compute_pipeline(BeamformerComputePlan *cp, BeamformerParameterBlock *pb, A sd->layout.y = 4; sd->layout.z = 1; - sd->dispatch.x = (u32)ceil_f32((f32)pb->parameters.acquisition_count / (f32)sd->layout.x / (f32)db->ToProcess); - sd->dispatch.y = (u32)ceil_f32((f32)chunk_channel_count / (f32)sd->layout.y); - sd->dispatch.z = (u32)ceil_f32((f32)decode_sample_count / (f32)sd->layout.z); + sd->dispatch.x = (u32)ceil_f32((f32)db->TransmitCount / (f32)sd->layout.x / (f32)db->ToProcess); + sd->dispatch.y = (u32)ceil_f32((f32)db->ChunkChannelCount / (f32)sd->layout.y); + sd->dispatch.z = (u32)ceil_f32((f32)decode_sample_count / (f32)sd->layout.z); } else { /* NOTE(rnp): register caching. using more threads will cause the compiler to do * contortions to avoid spilling registers. using less gives higher performance */ sd->layout = (uv3){{subgroup_size / 2, 1, 1}}; - sd->dispatch.x = (u32)ceil_f32((f32)decode_sample_count / (f32)sd->layout.x); - sd->dispatch.y = (u32)ceil_f32((f32)chunk_channel_count / (f32)sd->layout.y); + sd->dispatch.x = (u32)ceil_f32((f32)decode_sample_count / (f32)sd->layout.x); + sd->dispatch.y = (u32)ceil_f32((f32)db->ChunkChannelCount / (f32)sd->layout.y); sd->dispatch.z = 1; } sd->uses_heap = 1; - u32 order = pb->parameters.acquisition_count; + u32 order = db->TransmitCount; db->Hadamard = gpu_resource_push(resource_builder, f16, order * order, .data = make_hadamard_transpose(scratch, order, use_coop_matrix), .name = str8("hadamard")); @@ -850,21 +905,21 @@ plan_compute_pipeline(BeamformerComputePlan *cp, BeamformerParameterBlock *pb, A .data = f->data, .name = push_str8_f(scratch, "filter_%u", sp->filter_slot)); - fb->SampleCount = input_sample_count; + fb->SampleCount = node->rf_dimensions.E[BeamformerRFDimension_Samples]; fb->DecimationRate = demod ? decimation_rate : 1; b32 deinterleave = beamformer_data_kind_complex[node->input_data_kind] && !beamformer_data_kind_complex[node->output_data_kind]; if (deinterleave) - fb->BatchSampleCount = chunk_channel_count * input_sample_count * pb->parameters.acquisition_count; + fb->BatchSampleCount = node->rf_dimensions.x * node->rf_dimensions.y * node->rf_dimensions.z; - fb->OutputSampleStride = node->output_stride.x; - fb->OutputChannelStride = node->output_stride.y; - fb->OutputTransmitStride = node->output_stride.z; + fb->OutputSampleStride = output_stride.E[BeamformerRFDimension_Samples]; + fb->OutputChannelStride = output_stride.E[BeamformerRFDimension_Channels]; + fb->OutputTransmitStride = output_stride.E[BeamformerRFDimension_Events]; - fb->InputSampleStride = node->input_stride.x; - fb->InputChannelStride = node->input_stride.y; - fb->InputTransmitStride = node->input_stride.z; + fb->InputSampleStride = input_stride.E[BeamformerRFDimension_Samples]; + fb->InputChannelStride = input_stride.E[BeamformerRFDimension_Channels]; + fb->InputTransmitStride = input_stride.E[BeamformerRFDimension_Events]; /* NOTE(rnp): when we are demodulating we pretend that the sampler was alternating * between sampling the I portion and the Q portion of an IQ signal. Therefore there @@ -881,9 +936,9 @@ plan_compute_pipeline(BeamformerComputePlan *cp, BeamformerParameterBlock *pb, A } sd->layout = (uv3){{subgroup_size, 1, 1}}; - sd->dispatch.x = (u32)ceil_f32((f32)input_sample_count / (f32)sd->layout.x); - sd->dispatch.y = (u32)ceil_f32((f32)chunk_channel_count / (f32)sd->layout.y); - sd->dispatch.z = (u32)ceil_f32((f32)pb->parameters.acquisition_count / (f32)sd->layout.z); + sd->dispatch.x = (u32)ceil_f32((f32)node->rf_dimensions.E[BeamformerRFDimension_Samples] / (f32)sd->layout.x); + sd->dispatch.y = (u32)ceil_f32((f32)node->rf_dimensions.E[BeamformerRFDimension_Channels] / (f32)sd->layout.y); + sd->dispatch.z = (u32)ceil_f32((f32)node->rf_dimensions.E[BeamformerRFDimension_Events] / (f32)sd->layout.z); }break; case BeamformerShaderKind_DAS:{ @@ -896,10 +951,14 @@ plan_compute_pipeline(BeamformerComputePlan *cp, BeamformerParameterBlock *pb, A db->TimeOffset = time_offset; db->FNumber = pb->parameters.f_number; db->AcquisitionKind = pb->parameters.acquisition_kind; - db->SampleCount = input_sample_count; + db->SampleCount = node->rf_dimensions.E[BeamformerRFDimension_Samples]; db->ReceiveChannelCount = pb->parameters.channel_count; - db->AcquisitionCount = pb->parameters.acquisition_count; - db->ChunkChannelCount = chunk_channel_count; + db->AcquisitionCount = node->rf_dimensions.E[BeamformerRFDimension_Events]; + db->ChunkChannelCount = node->rf_dimensions.E[BeamformerRFDimension_Channels]; + db->ChannelByteStride = input_stride.E[BeamformerRFDimension_Channels] + * beamformer_data_kind_byte_size[node->input_data_kind]; + db->AcquisitionByteStride = input_stride.E[BeamformerRFDimension_Events] + * beamformer_data_kind_byte_size[node->input_data_kind]; db->InterpolationMode = pb->parameters.interpolation_mode; db->TransmitAngle = pb->parameters.focal_vector.E[0]; db->FocusDepth = pb->parameters.focal_vector.E[1]; @@ -993,17 +1052,17 @@ plan_compute_pipeline(BeamformerComputePlan *cp, BeamformerParameterBlock *pb, A sd->compile_flags |= BeamformerReshapeCompileFlags_Deinterleave * deinterleave; sd->compile_flags |= BeamformerReshapeCompileFlags_Interleave * interleave; - rb->InputStrideX = node->input_stride.x; - rb->InputStrideY = node->input_stride.y; - rb->InputStrideZ = node->input_stride.z; - rb->OutputStrideX = node->output_stride.x; - rb->OutputStrideY = node->output_stride.y; - rb->OutputStrideZ = node->output_stride.z; + rb->InputStrideX = input_stride.x; + rb->InputStrideY = input_stride.y; + rb->InputStrideZ = input_stride.z; + rb->OutputStrideX = output_stride.x; + rb->OutputStrideY = output_stride.y; + rb->OutputStrideZ = output_stride.z; // NOTE(rnp): order doesn't really matter here but it must match the dispatch layout - rb->SizeX = input_sample_count; - rb->SizeY = chunk_channel_count; - rb->SizeZ = acquisition_count; + rb->SizeX = node->rf_dimensions.x; + rb->SizeY = node->rf_dimensions.y; + rb->SizeZ = node->rf_dimensions.z; sd->layout.x = 1; sd->layout.z = Min(subgroup_size, rb->SizeZ); diff --git a/beamformer_core.meta b/beamformer_core.meta @@ -1,4 +1,20 @@ // See LICENSE for license details. +@Enumeration DataLayout +{ + ChannelEventSample + ChannelSampleEvent + EventChannelSample + SampleChannelEvent + Image +} + +@Enumeration RFDimension +{ + Channels + Events + Samples +} + @Flags PanelFlags { List diff --git a/generated/beamformer.c b/generated/beamformer.c @@ -199,6 +199,8 @@ typedef struct { i32 ReceiveChannelCount; i32 ChunkChannelCount; i32 SampleCount; + u32 AcquisitionByteStride; + u32 ChannelByteStride; f32 SamplingFrequency; f32 DemodulationFrequency; f32 SpeedOfSound; @@ -615,19 +617,21 @@ read_only global MetaStructMember *meta_struct_members_by_id[] = { {10, 28, 1, 0}, {10, 32, 1, 0}, {10, 36, 1, 0}, - {8, 40, 1, 0}, - {8, 44, 1, 0}, + {18, 40, 1, 0}, + {18, 44, 1, 0}, {8, 48, 1, 0}, {8, 52, 1, 0}, - {18, 56, 1, 0}, + {8, 56, 1, 0}, {8, 60, 1, 0}, {18, 64, 1, 0}, {8, 68, 1, 0}, - {8, 72, 1, 0}, - {18, 76, 1, 0}, - {18, 80, 1, 0}, + {18, 72, 1, 0}, + {8, 76, 1, 0}, + {8, 80, 1, 0}, {18, 84, 1, 0}, {18, 88, 1, 0}, + {18, 92, 1, 0}, + {18, 96, 1, 0}, }, (MetaStructMember []){ {18, 0, 1, 0}, @@ -687,6 +691,8 @@ read_only global str8 *meta_struct_member_names_by_id[] = { str8_comp("ReceiveChannelCount"), str8_comp("ChunkChannelCount"), str8_comp("SampleCount"), + str8_comp("AcquisitionByteStride"), + str8_comp("ChannelByteStride"), str8_comp("SamplingFrequency"), str8_comp("DemodulationFrequency"), str8_comp("SpeedOfSound"), @@ -720,11 +726,11 @@ read_only global str8 *meta_struct_member_names_by_id[] = { }; read_only global MetaStructInfo meta_struct_info_by_id[] = { - {str8_comp("DecodeBakeParameters"), 11, 44, 0}, - {str8_comp("FilterBakeParameters"), 13, 52, 0}, - {str8_comp("DASBakeParameters"), 23, 92, 0}, - {str8_comp("CoherencyWeightingBakeParameters"), 3, 12, 0}, - {str8_comp("ReshapeBakeParameters"), 9, 36, 0}, + {str8_comp("DecodeBakeParameters"), 11, 44, 0}, + {str8_comp("FilterBakeParameters"), 13, 52, 0}, + {str8_comp("DASBakeParameters"), 25, 100, 0}, + {str8_comp("CoherencyWeightingBakeParameters"), 3, 12, 0}, + {str8_comp("ReshapeBakeParameters"), 9, 36, 0}, }; read_only global str8 beamformer_shader_names[] = { diff --git a/generated/beamformer_core.c b/generated/beamformer_core.c @@ -3,6 +3,22 @@ // GENERATED CODE typedef enum { + BeamformerDataLayout_ChannelEventSample = 0, + BeamformerDataLayout_ChannelSampleEvent = 1, + BeamformerDataLayout_EventChannelSample = 2, + BeamformerDataLayout_SampleChannelEvent = 3, + BeamformerDataLayout_Image = 4, + BeamformerDataLayout_Count, +} BeamformerDataLayout; + +typedef enum { + BeamformerRFDimension_Channels = 0, + BeamformerRFDimension_Events = 1, + BeamformerRFDimension_Samples = 2, + BeamformerRFDimension_Count, +} BeamformerRFDimension; + +typedef enum { BeamformerPanelKind_Nil = 0, BeamformerPanelKind_Split = 1, BeamformerPanelKind_TabGroup = 2, diff --git a/shaders/das.glsl b/shaders/das.glsl @@ -79,6 +79,13 @@ u32 batch_channel_count() return result; } +u64 rf_data_pointer(const u32 channel, const u32 acquisition) +{ + u64 result = rf_data + ChannelByteStride * channel + AcquisitionByteStride * acquisition; + result -= InputDataKindByteSize * u32(InterpolationMode == InterpolationMode_Cubic); + return result; +} + /* NOTE: See: https://cubic.org/docs/hermite.htm */ SAMPLE_TYPE cubic(const u64 rf_pointer, const f32 t) { @@ -247,8 +254,7 @@ RESULT_TYPE RCA(const vec3 world_point) vec2 xdc_world_point = rca_plane_projection((xdc_transform * vec4(world_point, 1)).xyz, rx_rows); f32 transmit_index = sample_index(rca_transmit_distance(world_point, focal_vector, tx_rx_orientation)); - u64 rf_pointer = rf_data + InputDataKindByteSize * acquisition * SampleCount; - rf_pointer -= InputDataKindByteSize * u32(InterpolationMode == InterpolationMode_Cubic); + u64 rf_pointer = rf_data_pointer(0u, u32(acquisition)); for (f32 chunk_channel = 0.f; chunk_channel < f32(batch_channel_count()); chunk_channel += 1.f) { f32 rx_channel = f32(channel_offset) + chunk_channel; @@ -261,7 +267,7 @@ RESULT_TYPE RCA(const vec3 world_point) SAMPLE_TYPE value = apodize(a_arg) * sample_rf(rf_pointer, index); result += RESULT_STORE(value); } - rf_pointer += InputDataKindByteSize * SampleCount * AcquisitionCount; + rf_pointer += ChannelByteStride; } } return result; @@ -291,8 +297,7 @@ RESULT_TYPE HERCULES(const vec3 world_point) f32 element_receive_delta_squared = rx_world_point - rx_channel * rx_pitch; element_receive_delta_squared *= element_receive_delta_squared; - u64 rf_pointer = rf_data + InputDataKindByteSize * (u32(chunk_channel) * SampleCount * AcquisitionCount + u32(Sparse) * SampleCount); - rf_pointer -= InputDataKindByteSize * u32(InterpolationMode == InterpolationMode_Cubic); + u64 rf_pointer = rf_data_pointer(u32(chunk_channel), u32(Sparse)); for (f32 transmit = f32(Sparse); transmit < f32(AcquisitionCount); transmit += 1.f) { f32 tx_channel = Sparse ? f32(S16(HeapBase + SparseElements - 2 * u32(Sparse)).x[s32(transmit)]) : transmit; @@ -311,7 +316,7 @@ RESULT_TYPE HERCULES(const vec3 world_point) result += RESULT_STORE(value); } - rf_pointer += InputDataKindByteSize * SampleCount; + rf_pointer += AcquisitionByteStride; } } return result; @@ -335,8 +340,7 @@ RESULT_TYPE FORCES(const vec3 world_point) f32 a_arg = abs(FNumber * receive_x_delta / xdc_world_point.z); if (a_arg < 0.5f) { - u64 rf_pointer = rf_data + InputDataKindByteSize * (u32(chunk_channel) * SampleCount * AcquisitionCount + u32(Sparse) * SampleCount); - rf_pointer -= InputDataKindByteSize * u32(InterpolationMode == InterpolationMode_Cubic); + u64 rf_pointer = rf_data_pointer(u32(chunk_channel), u32(Sparse)); f32 receive_index = sample_index(sqrt(receive_x_delta * receive_x_delta + z_delta_squared)); f32 apodization = apodize(a_arg); @@ -347,7 +351,7 @@ RESULT_TYPE FORCES(const vec3 world_point) SAMPLE_TYPE value = apodization * sample_rf(rf_pointer, receive_index + transmit_index); result += RESULT_STORE(value); - rf_pointer += InputDataKindByteSize * SampleCount; + rf_pointer += AcquisitionByteStride; } } } @@ -375,8 +379,7 @@ RESULT_TYPE READI_FORCES(const vec3 world_point) f32 a_arg = abs(FNumber * receive_x_delta / xdc_world_point.z); if (a_arg < 0.5f) { - u64 channel_rf_pointer = rf_data + InputDataKindByteSize * u32(chunk_channel) * SampleCount * AcquisitionCount; - channel_rf_pointer -= InputDataKindByteSize * u32(InterpolationMode == InterpolationMode_Cubic); + u64 channel_rf_pointer = rf_data_pointer(u32(chunk_channel), 0); f32 receive_index = sample_index(sqrt(receive_x_delta * receive_x_delta + z_delta_squared)); f32 apodization = apodize(a_arg); @@ -395,7 +398,7 @@ RESULT_TYPE READI_FORCES(const vec3 world_point) SAMPLE_TYPE value = group_apodization * sample_rf(rf_pointer, receive_index + transmit_index); result += RESULT_STORE(value); - rf_pointer += InputDataKindByteSize * SampleCount; + rf_pointer += AcquisitionByteStride; } } }