ogl_beamforming

Ultrasound Beamforming Implemented with OpenGL
git clone anongit@rnpnr.xyz:ogl_beamforming.git
Log | Files | Refs | Feed | Submodules | README | LICENSE

Commit: a86ce25c884bbdae720165fb550b97bc35946406
Parent: 7478e30e20675a15ab4cfb48772ad2e1ace81177
Author: Randy Palamar
Date:   Sat,  3 Oct 2026 05:23:14 -0700

lib: minor refactoring towards rebeamforming already upload data

parameter handling still needs some work for this to be useful.

Diffstat:
Mbase_types.h | 1+
Mbeamformer_core.c | 26++++++++++++++++----------
Mbeamformer_internal.h | 2+-
Mbeamformer_shared_memory.c | 7+++++--
Mlib/ogl_beamformer_lib.c | 40+++++++++++++++++++++++++++++++---------
5 files changed, 54 insertions(+), 22 deletions(-)

diff --git a/base_types.h b/base_types.h @@ -38,6 +38,7 @@ typedef char c8; typedef u8 b8; typedef u16 b16; typedef u32 b32; +typedef u64 b64; typedef _Float16 f16; typedef float f32; typedef double f64; diff --git a/beamformer_core.c b/beamformer_core.c @@ -1,5 +1,8 @@ /* See LICENSE for license details. */ /* TODO(rnp): + * [ ]: refactor: make better use Transfer timeline semaphore to not stall compute thread + * while upload is occuring. rf thread should do: get next timeline semaphore value, insert + * into wait values array, atomic_inc insertion index, start upload, signal timeline semaphore * [ ]: backtrace dumping on SIGSEGV * [ ]: cooperative shared memory loading in decode shader * [ ]: refactor: save filter parameters with rest of parameters, whole slot thing is dumb @@ -1508,7 +1511,7 @@ complete_queue(BeamformerCtx *ctx, BeamformWorkQueue *q, Arena *arena) cp->filter_parameters[slot] = fctx->parameters; }break; - case BeamformerWorkKind_ComputeIndirect: + case BeamformerWorkKind_WaitThenCompute: case BeamformerWorkKind_Compute: { push_compute_timing_info(ctx->compute_timing_table, @@ -1588,10 +1591,13 @@ complete_queue(BeamformerCtx *ctx, BeamformWorkQueue *q, Arena *arena) } BeamformerRFBuffer *rf = &cs->rf_buffer; - u32 compute_index = rf->compute_index; - u32 slot = compute_index % countof(rf->upload_complete_values); + u64 compute_index = rf->compute_index; + u64 slot = compute_index % countof(rf->upload_complete_values); - if (work->kind == BeamformerWorkKind_ComputeIndirect) { + // NOTE(rnp): the library/ui will ensure that any time a new dataset is uploaded + // the next time beamforming is issued it will be tagged as WaitThenCompute. + // In this case we need to stall until the rf thread has finished its upload. + if (work->kind == BeamformerWorkKind_WaitThenCompute) { // TODO(rnp): this shouldn't be necessary, there should be a way of communicating // what the value will be so that the only the command wait is needed. spin_wait(atomic_load_u64(&rf->insertion_index) <= compute_index); @@ -1601,7 +1607,7 @@ complete_queue(BeamformerCtx *ctx, BeamformWorkQueue *q, Arena *arena) if (vk_buffer_needs_sync(&rf->buffer)) gpu_command_wait_timeline(cmd, GPUTimeline_Transfer, rf->upload_complete_values[slot]); } else { - slot = (rf->compute_index - 1) % countof(rf->upload_complete_values); + slot = (compute_index - 1) % countof(rf->upload_complete_values); } // NOTE(rnp): nvidia needs a memory barrier between pipeline stages @@ -1660,10 +1666,10 @@ complete_queue(BeamformerCtx *ctx, BeamformWorkQueue *q, Arena *arena) gpu_command_timestamp(cmd); } u64 end_timeline_value = gpu_command_list_end(cmd, (VulkanHandle){0}, (VulkanHandle){0}); - if (work->kind == BeamformerWorkKind_ComputeIndirect) { - atomic_store_u64(rf->compute_complete_values + slot, end_timeline_value); + atomic_store_u64(rf->compute_complete_values + slot, end_timeline_value); + + if (work->kind == BeamformerWorkKind_WaitThenCompute) atomic_add_u64(&rf->compute_index, 1); - } atomic_store_u64(&frame->timeline_valid_value, end_timeline_value); @@ -1809,8 +1815,8 @@ DEBUG_EXPORT BEAMFORMER_RF_UPLOAD_FN(beamformer_rf_upload) BeamformerRFBuffer *rf = ctx->rf_buffer; - rf->active_rf_size = gpu_round_up_to_sync_size(rf_block_rf_size & 0xFFFFFFFFULL, 64); - if unlikely(rf->buffer.size < countof(rf->upload_complete_values) * rf->active_rf_size) { + rf->active_rf_size = gpu_round_up_to_sync_size(rf_block_rf_size & 0x00FFFFFFFFFFFFFFull, 64); + if unlikely((u64)rf->buffer.size < countof(rf->upload_complete_values) * rf->active_rf_size) { gpu_buffer_allocate(&rf->buffer, (GPUBufferAllocateInfo){ .size = countof(rf->upload_complete_values) * rf->active_rf_size, .flags = GPUUsageFlag_HostWrite, diff --git a/beamformer_internal.h b/beamformer_internal.h @@ -331,7 +331,7 @@ typedef struct { GPUBuffer buffer; - u32 active_rf_size; + u64 active_rf_size; u64 timestamp; diff --git a/beamformer_shared_memory.c b/beamformer_shared_memory.c @@ -1,9 +1,9 @@ /* See LICENSE for license details. */ -#define BEAMFORMER_SHARED_MEMORY_VERSION (35UL) +#define BEAMFORMER_SHARED_MEMORY_VERSION (36UL) typedef enum { BeamformerWorkKind_Compute, - BeamformerWorkKind_ComputeIndirect, + BeamformerWorkKind_WaitThenCompute, BeamformerWorkKind_CreateFilter, BeamformerWorkKind_ExportBuffer, } BeamformerWorkKind; @@ -40,6 +40,7 @@ typedef enum {BEAMFORMER_SHARED_MEMORY_LOCKS BeamformerSharedMemoryLockKind_Coun typedef struct { BeamformerViewPlaneTag view_plane; u32 parameter_block; + u64 rf_offset; } BeamformerComputeWorkContext; /* NOTE: discriminated union based on type */ @@ -144,6 +145,8 @@ typedef struct { /* TODO(rnp): this is really sucky. we need a better way to communicate this */ u64 rf_block_rf_size; + b64 new_rf_upload; + // NOTE(rnp): currently this cannot be directly user readable. its interpretation // requires beamformer implementation details u64 beamformed_frame_buffer_size; diff --git a/lib/ogl_beamformer_lib.c b/lib/ogl_beamformer_lib.c @@ -484,7 +484,7 @@ BEAMFORMER_REDUCE_A1S2_CONTRAST_LIST #undef X function b32 -beamformer_push_data_base(void *data, u32 data_size, i32 timeout_ms, u32 block) +beamformer_push_data_base(void *data, u64 data_size, i32 timeout_ms, u32 block) { b32 result = 0; Arena *scratch = beamformer_shared_memory_scratch_arena(g_beamformer_library_context.bp, @@ -553,8 +553,9 @@ beamformer_push_data_base(void *data, u32 data_size, i32 timeout_ms, u32 block) lib_release_lock(BeamformerSharedMemoryLockKind_ScratchSpace); /* TODO(rnp): need a better way to communicate this */ - u64 rf_block_rf_size = (u64)block << 32ULL | (u64)rf_size; + u64 rf_block_rf_size = ((u64)block << 56ull) | (rf_size & 0x00FFFFFFFFFFFFFFull); atomic_store_u64(&g_beamformer_library_context.bp->rf_block_rf_size, rf_block_rf_size); + atomic_store_u64(&g_beamformer_library_context.bp->new_rf_upload, 1ull); result = 1; } } @@ -562,21 +563,22 @@ beamformer_push_data_base(void *data, u32 data_size, i32 timeout_ms, u32 block) return result; } -b32 -beamformer_push_data_with_compute(void *data, u32 data_size, u32 image_plane_tag, u32 parameter_slot) +function b32 +beamformer_start_compute(u64 offset, u32 image_plane_tag, u32 parameter_slot) { b32 result = 0; if (check_shared_memory()) { u32 reserved_blocks = g_beamformer_library_context.bp->reserved_parameter_blocks; - if (lib_error_check(image_plane_tag < BeamformerViewPlaneTag_Count, InvalidImagePlane) && - lib_error_check(parameter_slot < reserved_blocks, ParameterBlockUnallocated) && - beamformer_push_data_base(data, data_size, g_beamformer_library_context.timeout_ms, parameter_slot)) + if (lib_error_check(parameter_slot < reserved_blocks, ParameterBlockUnallocated) && + lib_error_check(image_plane_tag < BeamformerViewPlaneTag_Count, InvalidImagePlane)) { BeamformWork *work = try_push_work_queue(); if (work) { - work->kind = BeamformerWorkKind_ComputeIndirect; + b64 new_rf = atomic_swap_u64(&g_beamformer_library_context.bp->new_rf_upload, 0ull); + work->kind = new_rf ? BeamformerWorkKind_WaitThenCompute : BeamformerWorkKind_Compute; work->compute_context.view_plane = image_plane_tag; work->compute_context.parameter_block = parameter_slot; + work->compute_context.rf_offset = offset; beamform_work_queue_push_commit(&g_beamformer_library_context.bp->external_work_queue); beamformer_flush_commands(); result = 1; @@ -586,6 +588,26 @@ beamformer_push_data_with_compute(void *data, u32 data_size, u32 image_plane_tag return result; } +function b32 +beamformer_push_data(void *data, u64 data_size, u32 parameter_slot) +{ + b32 result = 0; + if (check_shared_memory()) { + u32 reserved_blocks = g_beamformer_library_context.bp->reserved_parameter_blocks; + result = lib_error_check(parameter_slot < reserved_blocks, ParameterBlockUnallocated) && + beamformer_push_data_base(data, data_size, g_beamformer_library_context.timeout_ms, parameter_slot); + } + return result; +} + +b32 +beamformer_push_data_with_compute(void *data, u32 data_size, u32 image_plane_tag, u32 parameter_slot) +{ + b32 result = beamformer_push_data(data, data_size, parameter_slot) && + beamformer_start_compute(0, image_plane_tag, parameter_slot); + return result; +} + b32 beamformer_push_parameters_at(BeamformerParameters *bp, u32 block) { @@ -729,7 +751,7 @@ beamformer_beamform_data(BeamformerSimpleParameters *bp, void *data, uint32_t da return result; } -BEAMFORMER_LIB_EXPORT b32 +function b32 beamformer_compute_timings(BeamformerComputeStatsTable *output, i32 timeout_ms) { b32 result = 0;