Commit: a86ce25c884bbdae720165fb550b97bc35946406
Parent: 7478e30e20675a15ab4cfb48772ad2e1ace81177
Author: Randy Palamar
Date: Sat, 3 Oct 2026 05:23:14 -0700
lib: minor refactoring towards rebeamforming already upload data
parameter handling still needs some work for this to be useful.
Diffstat:
5 files changed, 54 insertions(+), 22 deletions(-)
diff --git a/base_types.h b/base_types.h
@@ -38,6 +38,7 @@ typedef char c8;
typedef u8 b8;
typedef u16 b16;
typedef u32 b32;
+typedef u64 b64;
typedef _Float16 f16;
typedef float f32;
typedef double f64;
diff --git a/beamformer_core.c b/beamformer_core.c
@@ -1,5 +1,8 @@
/* See LICENSE for license details. */
/* TODO(rnp):
+ * [ ]: refactor: make better use Transfer timeline semaphore to not stall compute thread
+ * while upload is occuring. rf thread should do: get next timeline semaphore value, insert
+ * into wait values array, atomic_inc insertion index, start upload, signal timeline semaphore
* [ ]: backtrace dumping on SIGSEGV
* [ ]: cooperative shared memory loading in decode shader
* [ ]: refactor: save filter parameters with rest of parameters, whole slot thing is dumb
@@ -1508,7 +1511,7 @@ complete_queue(BeamformerCtx *ctx, BeamformWorkQueue *q, Arena *arena)
cp->filter_parameters[slot] = fctx->parameters;
}break;
- case BeamformerWorkKind_ComputeIndirect:
+ case BeamformerWorkKind_WaitThenCompute:
case BeamformerWorkKind_Compute:
{
push_compute_timing_info(ctx->compute_timing_table,
@@ -1588,10 +1591,13 @@ complete_queue(BeamformerCtx *ctx, BeamformWorkQueue *q, Arena *arena)
}
BeamformerRFBuffer *rf = &cs->rf_buffer;
- u32 compute_index = rf->compute_index;
- u32 slot = compute_index % countof(rf->upload_complete_values);
+ u64 compute_index = rf->compute_index;
+ u64 slot = compute_index % countof(rf->upload_complete_values);
- if (work->kind == BeamformerWorkKind_ComputeIndirect) {
+ // NOTE(rnp): the library/ui will ensure that any time a new dataset is uploaded
+ // the next time beamforming is issued it will be tagged as WaitThenCompute.
+ // In this case we need to stall until the rf thread has finished its upload.
+ if (work->kind == BeamformerWorkKind_WaitThenCompute) {
// TODO(rnp): this shouldn't be necessary, there should be a way of communicating
// what the value will be so that the only the command wait is needed.
spin_wait(atomic_load_u64(&rf->insertion_index) <= compute_index);
@@ -1601,7 +1607,7 @@ complete_queue(BeamformerCtx *ctx, BeamformWorkQueue *q, Arena *arena)
if (vk_buffer_needs_sync(&rf->buffer))
gpu_command_wait_timeline(cmd, GPUTimeline_Transfer, rf->upload_complete_values[slot]);
} else {
- slot = (rf->compute_index - 1) % countof(rf->upload_complete_values);
+ slot = (compute_index - 1) % countof(rf->upload_complete_values);
}
// NOTE(rnp): nvidia needs a memory barrier between pipeline stages
@@ -1660,10 +1666,10 @@ complete_queue(BeamformerCtx *ctx, BeamformWorkQueue *q, Arena *arena)
gpu_command_timestamp(cmd);
}
u64 end_timeline_value = gpu_command_list_end(cmd, (VulkanHandle){0}, (VulkanHandle){0});
- if (work->kind == BeamformerWorkKind_ComputeIndirect) {
- atomic_store_u64(rf->compute_complete_values + slot, end_timeline_value);
+ atomic_store_u64(rf->compute_complete_values + slot, end_timeline_value);
+
+ if (work->kind == BeamformerWorkKind_WaitThenCompute)
atomic_add_u64(&rf->compute_index, 1);
- }
atomic_store_u64(&frame->timeline_valid_value, end_timeline_value);
@@ -1809,8 +1815,8 @@ DEBUG_EXPORT BEAMFORMER_RF_UPLOAD_FN(beamformer_rf_upload)
BeamformerRFBuffer *rf = ctx->rf_buffer;
- rf->active_rf_size = gpu_round_up_to_sync_size(rf_block_rf_size & 0xFFFFFFFFULL, 64);
- if unlikely(rf->buffer.size < countof(rf->upload_complete_values) * rf->active_rf_size) {
+ rf->active_rf_size = gpu_round_up_to_sync_size(rf_block_rf_size & 0x00FFFFFFFFFFFFFFull, 64);
+ if unlikely((u64)rf->buffer.size < countof(rf->upload_complete_values) * rf->active_rf_size) {
gpu_buffer_allocate(&rf->buffer, (GPUBufferAllocateInfo){
.size = countof(rf->upload_complete_values) * rf->active_rf_size,
.flags = GPUUsageFlag_HostWrite,
diff --git a/beamformer_internal.h b/beamformer_internal.h
@@ -331,7 +331,7 @@ typedef struct {
GPUBuffer buffer;
- u32 active_rf_size;
+ u64 active_rf_size;
u64 timestamp;
diff --git a/beamformer_shared_memory.c b/beamformer_shared_memory.c
@@ -1,9 +1,9 @@
/* See LICENSE for license details. */
-#define BEAMFORMER_SHARED_MEMORY_VERSION (35UL)
+#define BEAMFORMER_SHARED_MEMORY_VERSION (36UL)
typedef enum {
BeamformerWorkKind_Compute,
- BeamformerWorkKind_ComputeIndirect,
+ BeamformerWorkKind_WaitThenCompute,
BeamformerWorkKind_CreateFilter,
BeamformerWorkKind_ExportBuffer,
} BeamformerWorkKind;
@@ -40,6 +40,7 @@ typedef enum {BEAMFORMER_SHARED_MEMORY_LOCKS BeamformerSharedMemoryLockKind_Coun
typedef struct {
BeamformerViewPlaneTag view_plane;
u32 parameter_block;
+ u64 rf_offset;
} BeamformerComputeWorkContext;
/* NOTE: discriminated union based on type */
@@ -144,6 +145,8 @@ typedef struct {
/* TODO(rnp): this is really sucky. we need a better way to communicate this */
u64 rf_block_rf_size;
+ b64 new_rf_upload;
+
// NOTE(rnp): currently this cannot be directly user readable. its interpretation
// requires beamformer implementation details
u64 beamformed_frame_buffer_size;
diff --git a/lib/ogl_beamformer_lib.c b/lib/ogl_beamformer_lib.c
@@ -484,7 +484,7 @@ BEAMFORMER_REDUCE_A1S2_CONTRAST_LIST
#undef X
function b32
-beamformer_push_data_base(void *data, u32 data_size, i32 timeout_ms, u32 block)
+beamformer_push_data_base(void *data, u64 data_size, i32 timeout_ms, u32 block)
{
b32 result = 0;
Arena *scratch = beamformer_shared_memory_scratch_arena(g_beamformer_library_context.bp,
@@ -553,8 +553,9 @@ beamformer_push_data_base(void *data, u32 data_size, i32 timeout_ms, u32 block)
lib_release_lock(BeamformerSharedMemoryLockKind_ScratchSpace);
/* TODO(rnp): need a better way to communicate this */
- u64 rf_block_rf_size = (u64)block << 32ULL | (u64)rf_size;
+ u64 rf_block_rf_size = ((u64)block << 56ull) | (rf_size & 0x00FFFFFFFFFFFFFFull);
atomic_store_u64(&g_beamformer_library_context.bp->rf_block_rf_size, rf_block_rf_size);
+ atomic_store_u64(&g_beamformer_library_context.bp->new_rf_upload, 1ull);
result = 1;
}
}
@@ -562,21 +563,22 @@ beamformer_push_data_base(void *data, u32 data_size, i32 timeout_ms, u32 block)
return result;
}
-b32
-beamformer_push_data_with_compute(void *data, u32 data_size, u32 image_plane_tag, u32 parameter_slot)
+function b32
+beamformer_start_compute(u64 offset, u32 image_plane_tag, u32 parameter_slot)
{
b32 result = 0;
if (check_shared_memory()) {
u32 reserved_blocks = g_beamformer_library_context.bp->reserved_parameter_blocks;
- if (lib_error_check(image_plane_tag < BeamformerViewPlaneTag_Count, InvalidImagePlane) &&
- lib_error_check(parameter_slot < reserved_blocks, ParameterBlockUnallocated) &&
- beamformer_push_data_base(data, data_size, g_beamformer_library_context.timeout_ms, parameter_slot))
+ if (lib_error_check(parameter_slot < reserved_blocks, ParameterBlockUnallocated) &&
+ lib_error_check(image_plane_tag < BeamformerViewPlaneTag_Count, InvalidImagePlane))
{
BeamformWork *work = try_push_work_queue();
if (work) {
- work->kind = BeamformerWorkKind_ComputeIndirect;
+ b64 new_rf = atomic_swap_u64(&g_beamformer_library_context.bp->new_rf_upload, 0ull);
+ work->kind = new_rf ? BeamformerWorkKind_WaitThenCompute : BeamformerWorkKind_Compute;
work->compute_context.view_plane = image_plane_tag;
work->compute_context.parameter_block = parameter_slot;
+ work->compute_context.rf_offset = offset;
beamform_work_queue_push_commit(&g_beamformer_library_context.bp->external_work_queue);
beamformer_flush_commands();
result = 1;
@@ -586,6 +588,26 @@ beamformer_push_data_with_compute(void *data, u32 data_size, u32 image_plane_tag
return result;
}
+function b32
+beamformer_push_data(void *data, u64 data_size, u32 parameter_slot)
+{
+ b32 result = 0;
+ if (check_shared_memory()) {
+ u32 reserved_blocks = g_beamformer_library_context.bp->reserved_parameter_blocks;
+ result = lib_error_check(parameter_slot < reserved_blocks, ParameterBlockUnallocated) &&
+ beamformer_push_data_base(data, data_size, g_beamformer_library_context.timeout_ms, parameter_slot);
+ }
+ return result;
+}
+
+b32
+beamformer_push_data_with_compute(void *data, u32 data_size, u32 image_plane_tag, u32 parameter_slot)
+{
+ b32 result = beamformer_push_data(data, data_size, parameter_slot) &&
+ beamformer_start_compute(0, image_plane_tag, parameter_slot);
+ return result;
+}
+
b32
beamformer_push_parameters_at(BeamformerParameters *bp, u32 block)
{
@@ -729,7 +751,7 @@ beamformer_beamform_data(BeamformerSimpleParameters *bp, void *data, uint32_t da
return result;
}
-BEAMFORMER_LIB_EXPORT b32
+function b32
beamformer_compute_timings(BeamformerComputeStatsTable *output, i32 timeout_ms)
{
b32 result = 0;