ogl_beamforming

Ultrasound Beamforming Implemented with OpenGL
git clone anongit@rnpnr.xyz:ogl_beamforming.git
Log | Files | Refs | Feed | Submodules | README | LICENSE

beamformer_core.c (84914B)


      1 /* See LICENSE for license details. */
      2 /* TODO(rnp):
      3  * [ ]: backtrace dumping on SIGSEGV
      4  * [ ]: cooperative shared memory loading in decode shader
      5  * [ ]: refactor: save filter parameters with rest of parameters, whole slot thing is dumb
      6  * [ ]: upload previously exported data for display. maybe this is a UI thing but doing it
      7  *      programatically would be nice.
      8  * [ ]: Add interface for multi frame upload. RF upload already uses an offset into SM so
      9  *      that part works fine. We just need a way of specify a multi frame upload. (Data must
     10  *      be organized for simple offset access per frame).
     11  * [ ]: refactor: do_compute should build its own "command graph" which tracks
     12  *      dependencies better. It is very important that unnecessary barriers are
     13  *      not placed between compute stages which requires knowledge of the entire
     14  *      graph.
     15  * [ ]: refactor: replace UploadRF with just the scratch_rf_size variable,
     16  *      use below to spin wait in library
     17  * [ ]: utilize umonitor/umwait (intel), monitorx/mwaitx (amd), and wfe/sev (aarch64)
     18  *      for power efficient low latency waiting
     19  * [ ]: BeamformWorkQueue -> BeamformerWorkQueue
     20  * [ ]: refactor: work queue needs a cleanup, we should only have a single one
     21  *      - that queue isn't really considered hot so a lock is probably fine
     22  * [ ]: bug: reinit cuda on hot-reload
     23  *
     24  * [ ]: export for special frames/data
     25  *    - Color Map Data
     26  *    - Recursive Imaging Result
     27  *    - Incoherent sum
     28  *
     29  * [ ]: Tiled Array Handling
     30  *    [ ]: add tile count uv2
     31  *    [ ]: make xdc_transform an array
     32  *    [ ]: modify CPU DAS dispatch code to lookup xdc_transform based on tile index
     33  *    [ ]: modify CPU DAS dispatch code to set rf_element_offset based on tile index
     34  *       - need to check if this works for HERCULES or if the shader just needs to
     35  *         know about the extra tiles
     36  *    [ ]: write simple backpropagation code to optimize 2D tile position and 2D tilt
     37  *         (4 variables per tile).
     38  *       - two tests:
     39  *         1. Maximize value of wire target
     40  *         2. Maximize value of single cyst contrast
     41  *
     42  * [ ]: Recursive Imaging Handling
     43  *    [ ]: add array of image offsets to Compute Array Parameters
     44  *    [ ]: add array of weighting coeffiecents to Compute Array Parameters
     45  *    [ ]: Update 2D sum shader to use these
     46  *
     47  * [ ]: Power Doppler
     48  *    [ ]: also needs array of image offsets
     49  *    [ ]: make modified filter that grabs each sample from a different image offset
     50  *       - NOTE: this can't use existing striding mechanism because one image may
     51  *               be at the start of the ring buffer and another at the end of the
     52  *               ring buffer and the image size may not cleanly divide the ring buffer size.
     53  *       - Probably want this shader to just output the estimated power map directly.
     54  *    [ ]: make a modified render shader that takes two images: one structural and one that
     55  *         is used to index into a color map. (or second render pass that applies color overlay)
     56  */
     57 
     58 #include "base_platform.h"
     59 
     60 #if defined(BEAMFORMER_DEBUG) && !defined(BEAMFORMER_EXPORT) && OS_WINDOWS
     61   #define BEAMFORMER_EXPORT __declspec(dllexport)
     62 #endif
     63 
     64 #include "beamformer_internal.h"
     65 
     66 #define GPU_RESOURCE_HASH_TABLE_COUNT 256
     67 
     68 typedef struct GPUResource GPUResource;
     69 struct GPUResource {
     70 	str8 name;
     71 	u64  size;
     72 	u64  offset;
     73 	u64  alignment;
     74 
     75 	void *data;
     76 
     77 	u64  hash;
     78 
     79 	GPUResource *next;
     80 	GPUResource *hash_next, *hash_prev;
     81 };
     82 typedef struct {GPUResource *first, *last;} GPUResourceHashBucket;
     83 
     84 typedef struct {
     85 	Arena *arena;
     86 	u64    position;
     87 
     88 	GPUResource *resource_list;
     89 	GPUResourceHashBucket hash_table[GPU_RESOURCE_HASH_TABLE_COUNT];
     90 } GPUResourceBuilder;
     91 
     92 read_only BeamformerFrame       beamformer_nil_frame;
     93 read_only BeamformerComputePlan beamformer_nil_compute_plan;
     94 
     95 global BeamformerCtx   *beamformer_context;
     96 global BeamformerInput *beamformer_input;
     97 global f32 dt_for_frame;
     98 
     99 #define beamformer_frame_arena() (beamformer_context->frame_arenas[beamformer_context->frame_index % countof(beamformer_context->frame_arenas)])
    100 #define beamformer_registers() (&beamformer_context->registers->v)
    101 #define beamformer_push_registers(...) beamformer_push_registers_(&(BeamformerRegisters){beamformer_registers_init_literal __VA_ARGS__})
    102 #define BeamformerRegistersScope(...) DeferLoop(beamformer_push_registers(__VA_ARGS__), beamformer_pop_registers())
    103 #define beamformer_command(name, ...) beamformer_push_command(name, &(BeamformerRegisters){beamformer_registers_init_literal __VA_ARGS__})
    104 
    105 function BeamformerRegisters *
    106 beamformer_pop_registers(void)
    107 {
    108 	BeamformerRegisters *result = &beamformer_context->registers->v;
    109 	SLLStackPop(beamformer_context->registers, next);
    110 	if (beamformer_context->registers == 0)
    111 		beamformer_context->registers = &beamformer_context->base_registers;
    112 	return result;
    113 }
    114 
    115 function BeamformerRegisters *
    116 beamformer_push_registers_(BeamformerRegisters *registers)
    117 {
    118 	BeamformerRegistersNode *node   = push_struct(beamformer_frame_arena(), BeamformerRegistersNode);
    119 	BeamformerRegisters     *result = &node->v;
    120 	memory_copy(result, registers, sizeof(node->v));
    121 	SLLStackPush(beamformer_context->registers, node, next);
    122 	return result;
    123 }
    124 
    125 function void
    126 beamformer_command_list_push_new(Arena *arena, BeamformerCommandList *commands, str8 name, BeamformerRegisters *registers)
    127 {
    128 	BeamformerCommandNode *node = push_struct(arena, BeamformerCommandNode);
    129 	node->command.registers = push_struct_no_zero(arena, BeamformerRegisters);
    130 	node->command.name      = push_str8(arena, name);
    131 	memory_copy(node->command.registers, registers, sizeof(*registers));
    132 	DLLInsertLast(0, commands->first, commands->last, node, next, prev);
    133 	commands->count += 1;
    134 }
    135 
    136 function void
    137 beamformer_push_command(str8 name, BeamformerRegisters *registers)
    138 {
    139 	beamformer_command_list_push_new(beamformer_frame_arena(), beamformer_context->command_queues + 0,
    140 	                                 name, registers);
    141 }
    142 
    143 function BeamformerCommandKind
    144 beamformer_command_kind_from_string(str8 s)
    145 {
    146 	BeamformerCommandKind result = BeamformerCommandKind_Nil;
    147 	for EachElement(beamformer_command_infos, it) {
    148 		if (str8_equal(beamformer_command_infos[it].string, s)) {
    149 			result = (BeamformerCommandKind)it;
    150 			break;
    151 		}
    152 	}
    153 	return result;
    154 }
    155 
    156 function BeamformerPanelKind
    157 beamformer_panel_kind_from_string(str8 s)
    158 {
    159 	BeamformerPanelKind result = BeamformerPanelKind_Nil;
    160 	for EachElement(beamformer_panel_infos, it) {
    161 		if (str8_equal(beamformer_panel_infos[it].string, s)) {
    162 			result = (BeamformerPanelKind)it;
    163 			break;
    164 		}
    165 	}
    166 	return result;
    167 }
    168 
    169 function BeamformerFrame *
    170 beamformer_frame_from_index(u64 index)
    171 {
    172 	BeamformerFrame *result = (BeamformerFrame *)&beamformer_nil_frame;
    173 	if (index < countof(beamformer_context->compute_context.backlog.frames)) {
    174 		BeamformerFrame *frame = beamformer_context->compute_context.backlog.frames + index;
    175 		if (frame->timeline_valid_value != 0)
    176 			result = frame;
    177 	}
    178 	return result;
    179 }
    180 
    181 function b32
    182 beamformer_frame_valid(u64 index)
    183 {
    184 	b32 result = beamformer_frame_from_index(index) != &beamformer_nil_frame;
    185 	return result;
    186 }
    187 
    188 function void
    189 beamformer_compute_plan_release(BeamformerComputeContext *cc, u32 block)
    190 {
    191 	assert(block < countof(cc->compute_plans));
    192 	BeamformerComputePlan *cp = cc->compute_plans[block];
    193 	if (cp) {
    194 		gpu_buffer_release(&cp->gpu_temp_arena);
    195 		cc->compute_plans[block] = 0;
    196 		SLLPushFreelist(cp, cc->compute_plan_freelist);
    197 	}
    198 }
    199 
    200 function GPUResource *
    201 gpu_resource_from_hash(GPUResourceBuilder *rb, u64 hash)
    202 {
    203 	GPUResource *result = 0;
    204 
    205 	GPUResourceHashBucket *hb = rb->hash_table + (hash % GPU_RESOURCE_HASH_TABLE_COUNT);
    206 	for (GPUResource *r = hb->first; r; r = r->hash_next) {
    207 		if (hash == r->hash) {
    208 			result = r;
    209 			break;
    210 		}
    211 	}
    212 
    213 	return result;
    214 }
    215 
    216 typedef struct {
    217 	str8  name;
    218 	u64   align;
    219 	u64   size;
    220 	void *data;
    221 } GPUResourcePushInfo;
    222 #define gpu_resource_push(rb, t, count, ...) gpu_resource_push_(rb, (GPUResourcePushInfo){\
    223 	.align = Max(alignof(t), 16), \
    224 	.size  = sizeof(t) * count, \
    225 	__VA_ARGS__})
    226 
    227 function u32
    228 gpu_resource_push_(GPUResourceBuilder *rb, GPUResourcePushInfo info)
    229 {
    230 	assert(info.size > 0 && info.name.length > 0 && IsPowerOfTwo(info.align));
    231 
    232 	u64 hash = u64_hash_from_str8(info.name);
    233 	GPUResource *r = gpu_resource_from_hash(rb, hash);
    234 	if (!r) {
    235 		r = push_struct(rb->arena, GPUResource);
    236 		GPUResourceHashBucket *hb = rb->hash_table + (hash % GPU_RESOURCE_HASH_TABLE_COUNT);
    237 		DLLInsert(0, hb->first, hb->last, r, hash_next, hash_prev);
    238 		SLLStackPush(rb->resource_list, r, next);
    239 
    240 		r->hash      = hash;
    241 		r->name      = info.name;
    242 		r->alignment = Max(16, info.align);
    243 		r->offset    = AlignUpPowerOfTwo(rb->position, r->alignment);
    244 		r->size      = info.size;
    245 		rb->position = r->offset + r->size;
    246 	}
    247 
    248 	u32 result = r->offset;
    249 
    250 	// NOTE(rnp): if this is a new resource and no data is provided it is likely
    251 	// a temporary GPU side buffer. if this is not a new resource and no data
    252 	// is provided then maybe it is shared and someone else will provide it
    253 	if (info.data) r->data = info.data;
    254 
    255 	return result;
    256 }
    257 
    258 function GPUResourceBuilder *
    259 gpu_resource_build_begin(Arena *arena)
    260 {
    261 	GPUResourceBuilder *result = push_struct(arena, GPUResourceBuilder);
    262 	result->arena = arena;
    263 	return result;
    264 }
    265 
    266 function void
    267 gpu_resource_build_end(GPUResourceBuilder *rb, GPUBuffer *buffer)
    268 {
    269 	u64 size = gpu_round_up_to_sync_size(rb->position, 64);
    270 	if (size != (u64)buffer->size) {
    271 		gpu_buffer_allocate(buffer, (GPUBufferAllocateInfo){
    272 			.size  = size,
    273 			.flags = GPUUsageFlag_HostWrite,
    274 			.label = push_str8_f(rb->arena, "GPU Temp Arena [%p]", buffer),
    275 			.timelines_used = (GPUTimeline []){GPUTimeline_Compute},
    276 			.timeline_count = 1,
    277 		});
    278 	}
    279 
    280 	//////////////////////////////////////
    281 	// NOTE(rnp): upload data
    282 	u64 last_wait_value = 0;
    283 	for (GPUResource *r = rb->resource_list; r; r = r->next)
    284 		if (r->data)
    285 			last_wait_value = gpu_buffer_range_upload(buffer, r->data, r->offset, r->size, 0);
    286 
    287 	// TODO(rnp): cleanup this pointless stall
    288 	if (gpu_buffer_needs_sync(buffer))
    289 		gpu_host_wait_timeline(GPUTimeline_Transfer, last_wait_value, -1ULL);
    290 }
    291 
    292 function BeamformerComputePlan *
    293 beamformer_compute_plan_for_block(BeamformerComputeContext *cc, u32 block, Arena *arena)
    294 {
    295 	assert(block < countof(cc->compute_plans));
    296 	BeamformerComputePlan *result = cc->compute_plans[block];
    297 	if (!result) {
    298 		result = SLLPopFreelist(cc->compute_plan_freelist);
    299 		if (!result) result = push_struct_no_zero(arena, BeamformerComputePlan);
    300 		zero_struct(result);
    301 		cc->compute_plans[block] = result;
    302 
    303 		result->ui_voxel_transform = m4_identity();
    304 	}
    305 	return result;
    306 }
    307 
    308 function BeamformerFilter *
    309 beamformer_filter_create(Arena *arena, BeamformerFilterParameters fp)
    310 {
    311 	BeamformerFilter *result = push_struct(arena, BeamformerFilter);
    312 	switch (fp.kind) {
    313 	case BeamformerFilterKind_Kaiser:{
    314 		/* TODO(rnp): this should also support complex */
    315 		/* TODO(rnp): implement this as an IFIR filter instead to reduce computation */
    316 		result->data = kaiser_low_pass_filter(arena, fp.kaiser.cutoff_frequency, fp.sampling_frequency,
    317 		                                      fp.kaiser.beta, (i32)fp.kaiser.length);
    318 		result->length     = (i32)fp.kaiser.length;
    319 		result->time_delay = (f32)result->length / 2.0f / fp.sampling_frequency;
    320 	}break;
    321 
    322 	case BeamformerFilterKind_MatchedChirp:{
    323 		typeof(fp.matched_chirp) *mc = &fp.matched_chirp;
    324 		f32 fs = fp.sampling_frequency;
    325 		result->length = (i32)(mc->duration * fs);
    326 		if (fp.complex) {
    327 			result->data = baseband_chirp(arena, mc->min_frequency, mc->max_frequency, fs, result->length, 1, 0.5f);
    328 			result->time_delay = complex_filter_first_moment(result->data, result->length, fs);
    329 		} else {
    330 			result->data = rf_chirp(arena, mc->min_frequency, mc->max_frequency, fs, result->length, 1);
    331 			result->time_delay = real_filter_first_moment(result->data, result->length, fs);
    332 		}
    333 	}break;
    334 
    335 	InvalidDefaultCase;
    336 	}
    337 
    338 	result->parameters = fp;
    339 	return result;
    340 }
    341 
    342 function iv3
    343 das_valid_points(iv3 points)
    344 {
    345 	iv3 result;
    346 	result.x = Max(points.x, 1);
    347 	result.y = Max(points.y, 1);
    348 	result.z = Max(points.z, 1);
    349 	return result;
    350 }
    351 
    352 function GPUBuffer *
    353 beamformer_gpu_buffer_from_frame(BeamformerFrame *frame)
    354 {
    355 	GPUBuffer *result = beamformer_context->compute_context.backlog.buffer;
    356 	if (!Between(frame->gpu_pointer, result->gpu_pointer, result->gpu_pointer + result->size)) {
    357 		result = 0;
    358 		BeamformerComputePlan *cp = beamformer_context->compute_context.compute_plans[frame->parameter_block];
    359 		if (cp) {
    360 			result = &cp->gpu_temp_arena;
    361 			assert(Between(frame->gpu_pointer, result->gpu_pointer, result->gpu_pointer + result->size));
    362 		}
    363 	}
    364 	return result;
    365 }
    366 
    367 function u64
    368 beamformer_frame_byte_size(iv3 points, BeamformerDataKind kind)
    369 {
    370 	u64 result = points.x * points.y * points.z * beamformer_data_kind_byte_size[kind];
    371 	result = round_up_to(result, 64);
    372 	return result;
    373 }
    374 
    375 function u64
    376 beamformer_incoherent_frame_byte_size(iv3 points, BeamformerDataKind kind)
    377 {
    378 	u64 result = beamformer_frame_byte_size(points, kind) / beamformer_data_kind_element_count[kind];
    379 	return result;
    380 }
    381 
    382 function BeamformerFrame *
    383 beamformer_frame_next(BeamformerComputeContext *cc, iv3 output_points, b32 complex)
    384 {
    385 	BeamformerFrameBacklog *bl = &cc->backlog;
    386 
    387 	BeamformerDataKind kind = complex ? BeamformerDataKind_Float32Complex : BeamformerDataKind_Float32;
    388 	u64 frame_size = beamformer_frame_byte_size(output_points, kind);
    389 
    390 	// TODO(rnp): handle this somewhat gracefully (even it produces garbled output)
    391 	assert(frame_size <= (u64)bl->buffer->size);
    392 
    393 	if (bl->next_offset > (u64)bl->buffer->size - frame_size)
    394 		bl->next_offset = 0;
    395 
    396 	u64 id = bl->counter++;
    397 
    398 	BeamformerFrame *result = bl->frames + (id % countof(bl->frames));
    399 	atomic_store_u64(&result->timeline_valid_value, -1ULL);
    400 	result->id            = id & U32_MAX;
    401 	result->gpu_pointer   = bl->buffer->gpu_pointer + bl->next_offset;
    402 	result->points        = output_points;
    403 	result->data_kind     = kind;
    404 
    405 	bl->next_offset += frame_size;
    406 
    407 	return result;
    408 }
    409 
    410 function void
    411 push_compute_timing_info(ComputeTimingTable *t, ComputeTimingInfo info)
    412 {
    413 	u32 index = atomic_add_u32(&t->write_index, 1) % countof(t->buffer);
    414 	t->buffer[index] = info;
    415 }
    416 
    417 function uv3
    418 layout_for_output(iv3 points)
    419 {
    420 	uv3 result = {{1, 1, 1}};
    421 
    422 	b32 has_x = points.x > 1;
    423 	b32 has_y = points.y > 1;
    424 	b32 has_z = points.z > 1;
    425 
    426 	u32 subgroup_size  = gpu_info()->subgroup_size;
    427 	u32 grid_3d_z_size = Max(1, subgroup_size / (4 * 4));
    428 	u32 grid_2d_y_size = Max(1, subgroup_size / 8);
    429 
    430 	switch (iv3_dimension(points)) {
    431 	case 1:{
    432 		if (has_x) result.x = subgroup_size;
    433 		if (has_y) result.y = subgroup_size;
    434 		if (has_z) result.z = subgroup_size;
    435 	}break;
    436 
    437 	case 2:{
    438 		if (has_x && has_y) {result.x = 8; result.y = grid_2d_y_size;}
    439 		if (has_x && has_z) {result.x = 8; result.z = grid_2d_y_size;}
    440 		if (has_y && has_z) {result.y = 8; result.z = grid_2d_y_size;}
    441 	}break;
    442 
    443 	case 3:{result = (uv3){{4, 4, grid_3d_z_size}};}break;
    444 
    445 	InvalidDefaultCase;
    446 	}
    447 
    448 	return result;
    449 }
    450 
    451 function uv3
    452 dispatch_for_output(uv3 layout, iv3 points)
    453 {
    454 	uv3 result;
    455 	result.x = (u32)ceil_f32((f32)points.x / layout.x);
    456 	result.y = (u32)ceil_f32((f32)points.y / layout.y);
    457 	result.z = (u32)ceil_f32((f32)points.z / layout.z);
    458 	return result;
    459 }
    460 
    461 typedef struct BeamformerComputeGraphNode BeamformerComputeGraphNode;
    462 struct BeamformerComputeGraphNode {
    463 	// NOTE(rnp): will be BeamformerShaderKind_Count for root node
    464 	BeamformerShaderKind kind;
    465 
    466 	// NOTE(rnp): when any of input or output stride is assigned it is assumed that
    467 	// the shader requires a fixed layout for input, output, or both. When two adjacent
    468 	// nodes require incompatible layouts the second pass over the graph will insert
    469 	// Reshape shaders in between.
    470 	BeamformerDataKind   input_data_kind;
    471 	BeamformerDataLayout input_data_layout;
    472 
    473 	BeamformerDataKind   output_data_kind;
    474 	BeamformerDataLayout output_data_layout;
    475 
    476 	uv3 rf_dimensions;
    477 
    478 	b32 requires_fp_input;
    479 
    480 	i32 user_pipeline_index;
    481 
    482 	BeamformerComputeGraphNode *prev;
    483 	BeamformerComputeGraphNode *next;
    484 };
    485 
    486 typedef struct {
    487 	BeamformerComputeGraphNode *first;
    488 	BeamformerComputeGraphNode *last;
    489 	u64                         count;
    490 } BeamformerComputeGraph;
    491 
    492 function b32
    493 compute_plan_push_shader(BeamformerComputePlan *p, BeamformerComputeGraphNode *node, BeamformerShaderParameters *sp)
    494 {
    495 	b32 result = 0;
    496 	if (p->pipeline.shader_count < countof(p->pipeline.shaders)) {
    497 		u32 index = p->pipeline.shader_count++;
    498 		p->pipeline.shaders[index]    = node->kind;
    499 		zero_struct(p->shader_descriptors + index);
    500 		p->pipeline.parameters[index] = sp ? *sp : (BeamformerShaderParameters){0};
    501 
    502 		p->shader_descriptors[index].input_data_kind  = node->input_data_kind;
    503 		p->shader_descriptors[index].output_data_kind = node->output_data_kind;
    504 
    505 		result = 1;
    506 	}
    507 	return result;
    508 }
    509 
    510 function BeamformerComputeGraphNode *
    511 push_compute_graph_node(BeamformerComputeGraph *graph, BeamformerShaderKind kind, Arena *arena)
    512 {
    513 	BeamformerComputeGraphNode *result = push_struct(arena, BeamformerComputeGraphNode);
    514 	if (graph) {
    515 		if (graph->last) result->rf_dimensions = graph->last->rf_dimensions;
    516 		DLLInsertLast(0, graph->first, graph->last, result, next, prev);
    517 		graph->count++;
    518 	}
    519 	result->kind = kind;
    520 	result->user_pipeline_index = -1;
    521 	// NOTE(rnp): initially don't care data parameters
    522 	result->input_data_kind    = BeamformerDataKind_Count;
    523 	result->output_data_kind   = BeamformerDataKind_Count;
    524 	result->input_data_layout  = BeamformerDataLayout_Count;
    525 	result->output_data_layout = BeamformerDataLayout_Count;
    526 	return result;
    527 }
    528 
    529 function uv3
    530 beamformer_data_strides(BeamformerDataLayout layout, uv3 rf_dimensions)
    531 {
    532 	u32 channel_count = rf_dimensions.E[BeamformerRFDimension_Channels];
    533 	u32 event_count   = rf_dimensions.E[BeamformerRFDimension_Events];
    534 	u32 sample_count  = rf_dimensions.E[BeamformerRFDimension_Samples];
    535 
    536 	uv3 result = {0};
    537 	switch (layout) {
    538 	InvalidDefaultCase;
    539 
    540 	case BeamformerDataLayout_Image:{}break;
    541 
    542 	case BeamformerDataLayout_ChannelEventSample:{
    543 		result.E[BeamformerRFDimension_Channels] = event_count * sample_count;
    544 		result.E[BeamformerRFDimension_Events]   = sample_count;
    545 		result.E[BeamformerRFDimension_Samples]  = 1;
    546 	}break;
    547 
    548 	case BeamformerDataLayout_ChannelSampleEvent:{
    549 		result.E[BeamformerRFDimension_Channels] = event_count * sample_count;
    550 		result.E[BeamformerRFDimension_Events]   = 1;
    551 		result.E[BeamformerRFDimension_Samples]  = event_count;
    552 	}break;
    553 
    554 	case BeamformerDataLayout_EventChannelSample:{
    555 		result.E[BeamformerRFDimension_Events]   = channel_count * sample_count;
    556 		result.E[BeamformerRFDimension_Channels] = sample_count;
    557 		result.E[BeamformerRFDimension_Samples]  = 1;
    558 	}break;
    559 
    560 	case BeamformerDataLayout_SampleChannelEvent:{
    561 		result.E[BeamformerRFDimension_Samples]  = channel_count * event_count;
    562 		result.E[BeamformerRFDimension_Channels] = event_count;
    563 		result.E[BeamformerRFDimension_Events]   = 1;
    564 	}break;
    565 
    566 	}
    567 
    568 	return result;
    569 }
    570 
    571 function void
    572 plan_compute_pipeline(BeamformerComputePlan *cp, BeamformerParameterBlock *pb, Arena *scratch)
    573 {
    574 	b32 run_hilbert = 0;
    575 	b32 demodulate  = 0;
    576 
    577 	for (u32 i = 0; i < pb->pipeline.shader_count; i++) {
    578 		switch (pb->pipeline.shaders[i]) {
    579 		case BeamformerShaderKind_Hilbert:{run_hilbert = 1;}break;
    580 		case BeamformerShaderKind_Demodulate:{demodulate = 1;}break;
    581 		default:{}break;
    582 		}
    583 	}
    584 
    585 	if (demodulate) run_hilbert = 0;
    586 
    587 	f32 sampling_frequency = pb->parameters.sampling_frequency;
    588 	u32 input_sample_count = pb->parameters.sample_count;
    589 	u32 acquisition_count  = pb->parameters.acquisition_count;
    590 	u32 decimation_rate    = Max(pb->parameters.decimation_rate, 1);
    591 
    592 	cp->raw_channel_byte_stride = pb->parameters.sample_count * pb->parameters.acquisition_count
    593 	                              * beamformer_data_kind_byte_size[pb->pipeline.data_kind];
    594 
    595 	BeamformerDataKind input_data_kind = pb->pipeline.data_kind;
    596 	if (demodulate) {
    597 		switch (input_data_kind) {
    598 		case BeamformerDataKind_Int16:{  input_data_kind = BeamformerDataKind_Int16Complex;  }break;
    599 		case BeamformerDataKind_Float16:{input_data_kind = BeamformerDataKind_Float16Complex;}break;
    600 		case BeamformerDataKind_Float32:{input_data_kind = BeamformerDataKind_Float32Complex;}break;
    601 		default:{}break;
    602 		}
    603 		input_sample_count /= (2 * decimation_rate);
    604 		sampling_frequency /= (2 * decimation_rate);
    605 	}
    606 
    607 	cp->iq_pipeline = beamformer_data_kind_complex[input_data_kind] || run_hilbert;
    608 
    609 	BeamformerDataKind das_data_kind = cp->iq_pipeline ? BeamformerDataKind_Float32Complex
    610 	                                                   : BeamformerDataKind_Float32;
    611 
    612 	cp->channel_count = pb->parameters.channel_count;
    613 	u32 chunk_channel_count = Min(cp->channel_count, BeamformerChunkChannelCount);
    614 
    615 	cp->rf_size = input_sample_count * pb->parameters.acquisition_count * chunk_channel_count
    616 	              * beamformer_data_kind_byte_size[das_data_kind];
    617 
    618 	i64 buffer_size = PING_PONG_BUFFER_SLOTS * round_up_to(cp->rf_size, 64);
    619 	if (beamformer_context->compute_context.ping_pong_buffer.size < buffer_size) {
    620 		b32 cuda = cuda_supported();
    621 		GPUBufferAllocateInfo allocate_info = {
    622 			.size   = buffer_size,
    623 			.export = cuda ? &beamformer_context->compute_context.ping_pong_export_handle : 0,
    624 			.label  = str8("PingPongBuffer"),
    625 		};
    626 		gpu_buffer_allocate(&beamformer_context->compute_context.ping_pong_buffer, allocate_info);
    627 
    628 		// TODO(rnp): figure out how to share with CUDA
    629 		// IMPORTANT: on linux the handle is returned to os and should be cleared after import
    630 		// see usage of glImportMemoryFdEXT and surrounding code in ui.c for examples
    631 		if (cuda) {
    632 		}
    633 	}
    634 
    635 	read_only BeamformerDataKind data_kind_to_element_kind[] = {
    636 		[BeamformerDataKind_Int16]          = BeamformerDataKind_Float16,
    637 		[BeamformerDataKind_Float16]        = BeamformerDataKind_Float16,
    638 		[BeamformerDataKind_Float32]        = BeamformerDataKind_Float32,
    639 		[BeamformerDataKind_Int16Complex]   = BeamformerDataKind_Float16,
    640 		[BeamformerDataKind_Float16Complex] = BeamformerDataKind_Float16,
    641 		[BeamformerDataKind_Float32Complex] = BeamformerDataKind_Float32,
    642 	};
    643 
    644 	read_only BeamformerDataKind data_kind_to_fp_kind[] = {
    645 		[BeamformerDataKind_Int16]          = BeamformerDataKind_Float16,
    646 		[BeamformerDataKind_Float16]        = BeamformerDataKind_Float16,
    647 		[BeamformerDataKind_Float32]        = BeamformerDataKind_Float32,
    648 		[BeamformerDataKind_Int16Complex]   = BeamformerDataKind_Float16Complex,
    649 		[BeamformerDataKind_Float16Complex] = BeamformerDataKind_Float16Complex,
    650 		[BeamformerDataKind_Float32Complex] = BeamformerDataKind_Float32Complex,
    651 	};
    652 
    653 	//////////////////////////////////////
    654 	// NOTE(rnp): First Pass: build initial graph and insert hard layout constraints
    655 	BeamformerComputeGraph graph = {0};
    656 	BeamformerComputeGraphNode *root_node = push_compute_graph_node(&graph, BeamformerShaderKind_Count, scratch);
    657 	root_node->input_data_kind    = input_data_kind;
    658 	root_node->input_data_layout  = BeamformerDataLayout_ChannelEventSample;
    659 	root_node->output_data_kind   = input_data_kind;
    660 	root_node->output_data_layout = BeamformerDataLayout_ChannelEventSample;
    661 
    662 	root_node->rf_dimensions.E[BeamformerRFDimension_Channels] = chunk_channel_count;
    663 	root_node->rf_dimensions.E[BeamformerRFDimension_Events]   = pb->parameters.acquisition_count;
    664 	root_node->rf_dimensions.E[BeamformerRFDimension_Samples]  = pb->parameters.sample_count;
    665 
    666 	for EachIndex(pb->pipeline.shader_count, it) {
    667 		// NOTE(rnp): skip unnecessary shaders
    668 		switch (pb->pipeline.shaders[it]) {
    669 		case BeamformerShaderKind_Hilbert:{if (!run_hilbert) continue;}break;
    670 
    671 		case BeamformerShaderKind_Decode:{
    672 			if (pb->parameters.decode_mode == BeamformerDecodeMode_None)
    673 				continue;
    674 		}break;
    675 
    676 		case BeamformerShaderKind_Sum:
    677 		case BeamformerShaderKind_MinMax:
    678 		{
    679 			// NOTE(rnp): currently unsupported
    680 			continue;
    681 		}break;
    682 
    683 		default:{}break;
    684 		}
    685 
    686 		BeamformerComputeGraphNode *node = push_compute_graph_node(&graph, pb->pipeline.shaders[it], scratch);
    687 		node->user_pipeline_index = (i32)it;
    688 		switch (pb->pipeline.shaders[it]) {
    689 		case BeamformerShaderKind_Demodulate:{
    690 			node->rf_dimensions.E[BeamformerRFDimension_Samples] /= (2 * decimation_rate);
    691 		}break;
    692 
    693 		case BeamformerShaderKind_Decode:{
    694 			b32 low_precision   = beamformer_data_kind_element_size[input_data_kind] < 4;
    695 			b32 use_coop_matrix = gpu_info()->cooperative_matrix &&
    696 			                      low_precision &&
    697 			                      (acquisition_count   % 16 == 0) &&
    698 			                      (chunk_channel_count % 16 == 0);
    699 
    700 			node->requires_fp_input = 1;
    701 
    702 			// NOTE(rnp): fixed input layout required for reasonable performance
    703 			node->input_data_layout = BeamformerDataLayout_SampleChannelEvent;
    704 
    705 			if (use_coop_matrix) {
    706 				node->input_data_kind    = BeamformerDataKind_Float16;
    707 				node->output_data_kind   = data_kind_to_element_kind[das_data_kind];
    708 				node->output_data_layout = node->input_data_layout;
    709 			}
    710 		}break;
    711 
    712 		case BeamformerShaderKind_DAS:{
    713 			// NOTE(rnp): if we aren't decoding we can let the previous stage's
    714 			// data kind pass through as long as it is floating point
    715 			node->requires_fp_input = 1;
    716 			if (pb->parameters.decode_mode != BeamformerDecodeMode_None)
    717 				node->input_data_kind = das_data_kind;
    718 
    719 			node->input_data_layout = BeamformerDataLayout_ChannelEventSample;
    720 			if (pb->parameters.acquisition_kind == BeamformerAcquisitionKind_RCA_TPW ||
    721 			    pb->parameters.acquisition_kind == BeamformerAcquisitionKind_RCA_VLS)
    722 			{
    723 				node->input_data_layout = BeamformerDataLayout_EventChannelSample;
    724 			}
    725 
    726 			node->output_data_layout = BeamformerDataLayout_Image;
    727 			node->output_data_kind   = das_data_kind;
    728 
    729 			// NOTE(rnp): insert implicit CoherencyWeighting node
    730 			if (pb->parameters.coherency_weighting)
    731 				node = push_compute_graph_node(&graph, BeamformerShaderKind_CoherencyWeighting, scratch);
    732 		}break;
    733 
    734 		default:{}break;
    735 		}
    736 	}
    737 
    738 	//////////////////////////////////////
    739 	// NOTE(rnp): Second Pass: resolve layout constraints
    740 	for (BeamformerComputeGraphNode *node = root_node->next; node; node = node->next) {
    741 		b32 needs_reshape = 0;
    742 
    743 		// NOTE(rnp): data strides
    744 		{
    745 			b32 input_dont_care       = node->input_data_layout        == BeamformerDataLayout_Count;
    746 			b32 prev_output_dont_care = node->prev->output_data_layout == BeamformerDataLayout_Count;
    747 
    748 			if (prev_output_dont_care && !input_dont_care)
    749 				node->prev->output_data_layout = node->input_data_layout;
    750 
    751 			if (!prev_output_dont_care && input_dont_care)
    752 				node->input_data_layout = node->prev->output_data_layout;
    753 
    754 			if (prev_output_dont_care && input_dont_care)
    755 				node->input_data_layout = node->prev->output_data_layout = node->prev->input_data_layout;
    756 
    757 			needs_reshape |= node->input_data_layout != node->prev->output_data_layout;
    758 		}
    759 
    760 		// NOTE(rnp): data kinds
    761 		{
    762 			b32 input_dont_care       = node->input_data_kind        == BeamformerDataKind_Count;
    763 			b32 prev_output_dont_care = node->prev->output_data_kind == BeamformerDataKind_Count;
    764 
    765 			if (prev_output_dont_care && !input_dont_care)
    766 				node->prev->output_data_kind = node->input_data_kind;
    767 
    768 			if (!prev_output_dont_care && input_dont_care)
    769 				node->input_data_kind = node->prev->output_data_kind;
    770 
    771 			if (prev_output_dont_care && input_dont_care)
    772 				node->input_data_kind = node->prev->output_data_kind = node->prev->input_data_kind;
    773 
    774 			if (node->requires_fp_input) {
    775 				node->input_data_kind = data_kind_to_fp_kind[node->input_data_kind];
    776 				if (prev_output_dont_care)
    777 					node->prev->output_data_kind = node->input_data_kind;
    778 			}
    779 
    780 			needs_reshape |= node->input_data_kind != node->prev->output_data_kind;
    781 		}
    782 
    783 		// NOTE(rnp): insert reshape if needed
    784 		if (needs_reshape) {
    785 			BeamformerComputeGraphNode *new = push_compute_graph_node(0, BeamformerShaderKind_Reshape, scratch);
    786 			BeamformerComputeGraphNode *last  = node->prev;
    787 			DLLInsertLast(0, node, last, new, next, prev);
    788 			graph.count++;
    789 			new->rf_dimensions      = new->prev->rf_dimensions;
    790 			new->input_data_kind    = new->prev->output_data_kind;
    791 			new->input_data_layout  = new->prev->output_data_layout;
    792 			new->output_data_kind   = new->next->input_data_kind;
    793 			new->output_data_layout = new->next->input_data_layout;
    794 		}
    795 	}
    796 
    797 	// NOTE(rnp): ensure last node descriptor gets proper values for output data kind
    798 	if (graph.last->output_data_kind == BeamformerDataKind_Count)
    799 		graph.last->output_data_kind = graph.last->input_data_kind;
    800 
    801 	f32 time_offset   = pb->parameters.time_offset;
    802 	u32 subgroup_size = gpu_info()->subgroup_size;
    803 
    804 	cp->first_image_shader_index = 0;
    805 	cp->pipeline.shader_count = 0;
    806 
    807 	GPUResourceBuilder *resource_builder = gpu_resource_build_begin(scratch);
    808 	for (BeamformerComputeGraphNode *node = root_node->next; node; node = node->next) {
    809 		assert(node->prev->output_data_kind == node->input_data_kind);
    810 		assert(node->prev->output_data_layout == node->input_data_layout);
    811 
    812 		BeamformerShaderParameters *sp = 0;
    813 		if (node->user_pipeline_index >= 0)
    814 			sp = pb->pipeline.parameters + node->user_pipeline_index;
    815 
    816 		if (compute_plan_push_shader(cp, node, sp)) {
    817 			BeamformerShaderDescriptor *sd = cp->shader_descriptors + cp->pipeline.shader_count - 1;
    818 
    819 			uv3 output_stride = beamformer_data_strides(node->output_data_layout, node->rf_dimensions);
    820 			uv3 input_stride  = beamformer_data_strides(node->input_data_layout,  node->prev->rf_dimensions);
    821 
    822 			switch (node->kind) {
    823 			case BeamformerShaderKind_Decode:{
    824 				BeamformerDecodeBakeParameters *db = &sd->bake.Decode;
    825 
    826 				u32 decode_sample_count = node->rf_dimensions.E[BeamformerRFDimension_Samples];
    827 				db->DecodeMode        = pb->parameters.decode_mode;
    828 				db->TransmitCount     = node->rf_dimensions.E[BeamformerRFDimension_Events];
    829 				db->ChunkChannelCount = node->rf_dimensions.E[BeamformerRFDimension_Channels];
    830 
    831 				// NOTE(rnp): ignored when using coop matrices
    832 				db->OutputSampleStride   = output_stride.E[BeamformerRFDimension_Samples];
    833 				db->OutputChannelStride  = output_stride.E[BeamformerRFDimension_Channels];
    834 				db->OutputTransmitStride = output_stride.E[BeamformerRFDimension_Events];
    835 
    836 				db->ToProcess = 1;
    837 
    838 				b32 use_coop_matrix = gpu_info()->cooperative_matrix &&
    839 				                      node->input_data_kind == BeamformerDataKind_Float16 &&
    840 				                      (db->TransmitCount % 16 == 0) &&
    841 				                      (chunk_channel_count % 16 == 0);
    842 				if (use_coop_matrix) {
    843 					// TODO(rnp): shared memory for larger sizes
    844 					sd->layout = (uv3){{subgroup_size, 1, 1}};
    845 
    846 					if (demodulate)
    847 						decode_sample_count *= 2;
    848 
    849 					sd->compile_flags |= BeamformerDecodeCompileFlags_CooperativeMatrix;
    850 					db->CooperativeMatrixM = 16;
    851 					db->CooperativeMatrixN = 16;
    852 					db->CooperativeMatrixK = 16;
    853 
    854 					sd->dispatch.x = db->TransmitCount   / db->CooperativeMatrixN;
    855 					sd->dispatch.y = chunk_channel_count / db->CooperativeMatrixM;
    856 					sd->dispatch.z = decode_sample_count;
    857 				} else if (db->TransmitCount > 40) {
    858 					sd->compile_flags |= BeamformerDecodeCompileFlags_UseSharedMemory;
    859 
    860 					if (db->TransmitCount == 48)
    861 						db->ToProcess = db->TransmitCount / 16;
    862 
    863 					b32 use_16x  = db->TransmitCount == 48 || db->TransmitCount == 80 ||
    864 					               db->TransmitCount == 96 || db->TransmitCount == 160;
    865 					sd->layout.x = use_16x ? 16 : 32;
    866 					sd->layout.y = 4;
    867 					sd->layout.z = 1;
    868 
    869 					sd->dispatch.x = (u32)ceil_f32((f32)db->TransmitCount     / (f32)sd->layout.x / (f32)db->ToProcess);
    870 					sd->dispatch.y = (u32)ceil_f32((f32)db->ChunkChannelCount / (f32)sd->layout.y);
    871 					sd->dispatch.z = (u32)ceil_f32((f32)decode_sample_count   / (f32)sd->layout.z);
    872 				} else {
    873 					/* NOTE(rnp): register caching. using more threads will cause the compiler to do
    874 					 * contortions to avoid spilling registers. using less gives higher performance */
    875 					sd->layout = (uv3){{subgroup_size / 2, 1, 1}};
    876 
    877 					sd->dispatch.x = (u32)ceil_f32((f32)decode_sample_count   / (f32)sd->layout.x);
    878 					sd->dispatch.y = (u32)ceil_f32((f32)db->ChunkChannelCount / (f32)sd->layout.y);
    879 					sd->dispatch.z = 1;
    880 				}
    881 
    882 				sd->uses_heap = 1;
    883 				u32 order = db->TransmitCount;
    884 				db->Hadamard = gpu_resource_push(resource_builder, f16, order * order,
    885 				                                 .data = make_hadamard_transpose(scratch, order, use_coop_matrix),
    886 				                                 .name = str8("hadamard"));
    887 			}break;
    888 
    889 			case BeamformerShaderKind_Demodulate:
    890 			case BeamformerShaderKind_Filter:
    891 			{
    892 				b32 demod = node->kind == BeamformerShaderKind_Demodulate;
    893 				BeamformerFilter *f = beamformer_filter_create(scratch, cp->filter_parameters[sp->filter_slot]);
    894 
    895 				sd->compile_flags |= BeamformerFilterCompileFlags_Demodulate * demod;
    896 				sd->compile_flags |= BeamformerFilterCompileFlags_ComplexFilter * f->parameters.complex;
    897 
    898 				time_offset += f->time_delay;
    899 
    900 				BeamformerFilterBakeParameters *fb = &sd->bake.Filter;
    901 
    902 				sd->uses_heap = 1;
    903 				fb->FilterLength = (u32)f->length;
    904 				fb->FilterCoefficients = gpu_resource_push(resource_builder, f32, f->length * (f->parameters.complex ? 2 : 1),
    905 				                                           .data = f->data,
    906 				                                           .name = push_str8_f(scratch, "filter_%u", sp->filter_slot));
    907 
    908 				fb->SampleCount    = node->rf_dimensions.E[BeamformerRFDimension_Samples];
    909 				fb->DecimationRate = demod ? decimation_rate : 1;
    910 
    911 				b32 deinterleave =  beamformer_data_kind_complex[node->input_data_kind] &&
    912 				                   !beamformer_data_kind_complex[node->output_data_kind];
    913 				if (deinterleave)
    914 					fb->BatchSampleCount = node->rf_dimensions.x * node->rf_dimensions.y * node->rf_dimensions.z;
    915 
    916 				fb->OutputSampleStride   = output_stride.E[BeamformerRFDimension_Samples];
    917 				fb->OutputChannelStride  = output_stride.E[BeamformerRFDimension_Channels];
    918 				fb->OutputTransmitStride = output_stride.E[BeamformerRFDimension_Events];
    919 
    920 				fb->InputSampleStride    = input_stride.E[BeamformerRFDimension_Samples];
    921 				fb->InputChannelStride   = input_stride.E[BeamformerRFDimension_Channels];
    922 				fb->InputTransmitStride  = input_stride.E[BeamformerRFDimension_Events];
    923 
    924 				/* NOTE(rnp): when we are demodulating we pretend that the sampler was alternating
    925 				 * between sampling the I portion and the Q portion of an IQ signal. Therefore there
    926 				 * is an implicit decimation factor of 2 which must always be included. All code here
    927 				 * assumes that the signal was sampled in such a way that supports this operation.
    928 				 * To recover IQ[n] from the sampled data (RF[n]) we do the following:
    929 				 *   I[n]  = RF[n]
    930 				 *   Q[n]  = RF[n + 1]
    931 				 *   IQ[n] = I[n] - j*Q[n]
    932 				 */
    933 				if (demod) {
    934 					fb->DemodulationFrequency = pb->parameters.demodulation_frequency;
    935 					fb->SamplingFrequency     = pb->parameters.sampling_frequency / 2;
    936 				}
    937 
    938 				sd->layout     = (uv3){{subgroup_size, 1, 1}};
    939 				sd->dispatch.x = (u32)ceil_f32((f32)node->rf_dimensions.E[BeamformerRFDimension_Samples]  / (f32)sd->layout.x);
    940 				sd->dispatch.y = (u32)ceil_f32((f32)node->rf_dimensions.E[BeamformerRFDimension_Channels] / (f32)sd->layout.y);
    941 				sd->dispatch.z = (u32)ceil_f32((f32)node->rf_dimensions.E[BeamformerRFDimension_Events]   / (f32)sd->layout.z);
    942 			}break;
    943 
    944 			case BeamformerShaderKind_DAS:{
    945 				cp->first_image_shader_index = cp->pipeline.shader_count;
    946 
    947 				BeamformerDASBakeParameters *db = &sd->bake.DAS;
    948 				db->SamplingFrequency     = sampling_frequency;
    949 				db->DemodulationFrequency = pb->parameters.demodulation_frequency;
    950 				db->SpeedOfSound          = pb->parameters.speed_of_sound;
    951 				db->TimeOffset            = time_offset;
    952 				db->FNumber               = pb->parameters.f_number;
    953 				db->AcquisitionKind       = pb->parameters.acquisition_kind;
    954 				db->SampleCount           = node->rf_dimensions.E[BeamformerRFDimension_Samples];
    955 				db->ReceiveChannelCount   = pb->parameters.channel_count;
    956 				db->AcquisitionCount      = node->rf_dimensions.E[BeamformerRFDimension_Events];
    957 				db->ChunkChannelCount     = node->rf_dimensions.E[BeamformerRFDimension_Channels];
    958 				db->ChannelByteStride     = input_stride.E[BeamformerRFDimension_Channels]
    959 				                            * beamformer_data_kind_byte_size[node->input_data_kind];
    960 				db->AcquisitionByteStride = input_stride.E[BeamformerRFDimension_Events]
    961 				                            * beamformer_data_kind_byte_size[node->input_data_kind];
    962 				db->InterpolationMode     = pb->parameters.interpolation_mode;
    963 				db->TransmitAngle         = pb->parameters.focal_vector.E[0];
    964 				db->FocusDepth            = pb->parameters.focal_vector.E[1];
    965 				db->ReadiGroupCount       = pb->parameters.readi_group_count;
    966 				db->OutputSizeX           = cp->output_points.x;
    967 				db->OutputSizeY           = cp->output_points.y;
    968 				db->OutputSizeZ           = cp->output_points.z;
    969 				db->TransmitReceiveOrientation = pb->parameters.transmit_receive_orientation;
    970 
    971 				// NOTE(rnp): old gcc will miscompile an assignment
    972 				memory_copy(cp->xdc_transform.E, pb->parameters.xdc_transform.E, sizeof(cp->xdc_transform));
    973 
    974 				cp->voxel_transform   = m4_mul(cp->ui_voxel_transform, pb->parameters.das_voxel_transform);
    975 				cp->xdc_element_pitch = pb->parameters.xdc_element_pitch;
    976 
    977 				memory_copy(cp->das_voxel_transform.E, cp->voxel_transform.E, sizeof(cp->voxel_transform));
    978 
    979 				u32 id = pb->parameters.acquisition_kind;
    980 				b32 sparse = id == BeamformerAcquisitionKind_UFORCES || id == BeamformerAcquisitionKind_UHERCULES;
    981 				b32 single_focus       = pb->parameters.single_focus != 0;
    982 				b32 single_orientation = pb->parameters.single_orientation != 0;
    983 
    984 				sd->compile_flags |= BeamformerDASCompileFlags_CoherencyWeighting * (pb->parameters.coherency_weighting != 0);
    985 				sd->compile_flags |= BeamformerDASCompileFlags_SingleFocus        * single_focus;
    986 				sd->compile_flags |= BeamformerDASCompileFlags_SingleOrientation  * single_orientation;
    987 				sd->compile_flags |= BeamformerDASCompileFlags_Sparse             * sparse;
    988 
    989 				sd->layout   = layout_for_output(cp->output_points);
    990 				sd->dispatch = dispatch_for_output(sd->layout, cp->output_points);
    991 
    992 				if (id != BeamformerAcquisitionKind_UFORCES && id != BeamformerAcquisitionKind_FORCES && !single_focus) {
    993 					db->FocalVectors = gpu_resource_push(resource_builder, v2, db->AcquisitionCount,
    994 					                                     .data = pb->focal_vectors,
    995 					                                     .name = str8("focal_vectors"));
    996 					sd->uses_heap = 1;
    997 				}
    998 
    999 				if (id != BeamformerAcquisitionKind_UFORCES && id != BeamformerAcquisitionKind_FORCES && !single_orientation) {
   1000 					db->TransmitReceiveOrientations = gpu_resource_push(resource_builder, u8, db->AcquisitionCount,
   1001 					                                                    .data = pb->transmit_receive_orientations,
   1002 					                                                    .name = str8("transmit_receive_orientations"));
   1003 					sd->uses_heap = 1;
   1004 				}
   1005 
   1006 				if (sparse) {
   1007 					db->SparseElements = gpu_resource_push(resource_builder, i16, db->AcquisitionCount,
   1008 					                                       .data = pb->sparse_elements,
   1009 					                                       .name = str8("sparse_elements"));
   1010 					sd->uses_heap = 1;
   1011 				}
   1012 
   1013 				if (pb->parameters.coherency_weighting) {
   1014 					db->IncoherentFrame = gpu_resource_push(resource_builder, u32, 0,
   1015 					                                        .size = beamformer_incoherent_frame_byte_size(cp->output_points, das_data_kind),
   1016 					                                        .name = str8("incoherent_buffer"));
   1017 					sd->uses_heap = 1;
   1018 				}
   1019 
   1020 				cp->readi_group = pb->parameters.readi_group;
   1021 				if (db->ReadiGroupCount > 1) {
   1022 					u32 order = db->ReadiGroupCount;
   1023 					db->Hadamard = gpu_resource_push(resource_builder, f16, order * order,
   1024 					                                 .data = make_hadamard_transpose(scratch, order, 0),
   1025 					                                 .name = str8("readi_hadamard"));
   1026 					sd->uses_heap = 1;
   1027 				}
   1028 			}break;
   1029 
   1030 			case BeamformerShaderKind_CoherencyWeighting:{
   1031 				// NOTE(rnp): beamformed data is stored in linear order; making the layout 2D or 3D
   1032 				// here is just slower
   1033 				sd->layout   = (uv3){{subgroup_size, 1, 1}};
   1034 				sd->dispatch = dispatch_for_output(sd->layout, cp->output_points);
   1035 
   1036 				BeamformerCoherencyWeightingBakeParameters *cw = &sd->bake.CoherencyWeighting;
   1037 				cw->Scale         = 1.f;
   1038 				cw->OutputVoxels  = cp->output_points.x * cp->output_points.y * cp->output_points.z;
   1039 				cw->IncoherentSum = gpu_resource_push(resource_builder, u32, 0,
   1040 				                                      .size = beamformer_incoherent_frame_byte_size(cp->output_points, das_data_kind),
   1041 				                                      .name = str8("incoherent_buffer"));
   1042 				sd->uses_heap = 1;
   1043 			}break;
   1044 
   1045 			case BeamformerShaderKind_Reshape:{
   1046 				BeamformerReshapeBakeParameters *rb = &sd->bake.Reshape;
   1047 				b32 deinterleave =  beamformer_data_kind_complex[node->input_data_kind] &&
   1048 				                   !beamformer_data_kind_complex[node->output_data_kind];
   1049 				b32 interleave   = !beamformer_data_kind_complex[node->input_data_kind] &&
   1050 				                    beamformer_data_kind_complex[node->output_data_kind];
   1051 				assert(interleave == 0 || (interleave != deinterleave));
   1052 				sd->compile_flags |= BeamformerReshapeCompileFlags_Deinterleave * deinterleave;
   1053 				sd->compile_flags |= BeamformerReshapeCompileFlags_Interleave   * interleave;
   1054 
   1055 				rb->InputStrideX   = input_stride.x;
   1056 				rb->InputStrideY   = input_stride.y;
   1057 				rb->InputStrideZ   = input_stride.z;
   1058 				rb->OutputStrideX  = output_stride.x;
   1059 				rb->OutputStrideY  = output_stride.y;
   1060 				rb->OutputStrideZ  = output_stride.z;
   1061 
   1062 				// NOTE(rnp): order doesn't really matter here but it must match the dispatch layout
   1063 				rb->SizeX          = node->rf_dimensions.x;
   1064 				rb->SizeY          = node->rf_dimensions.y;
   1065 				rb->SizeZ          = node->rf_dimensions.z;
   1066 
   1067 				sd->layout.x = 1;
   1068 				sd->layout.z = Min(subgroup_size, rb->SizeZ);
   1069 				sd->layout.y = subgroup_size / sd->layout.z;
   1070 
   1071 				sd->dispatch.x = (u32)(ceil_f32((f32)rb->SizeX / sd->layout.x));
   1072 				sd->dispatch.y = (u32)(ceil_f32((f32)rb->SizeY / sd->layout.y));
   1073 				sd->dispatch.z = (u32)(ceil_f32((f32)rb->SizeZ / sd->layout.z));
   1074 			}break;
   1075 
   1076 			default:{}break;
   1077 
   1078 			#if 0
   1079 			case BeamformerShaderKind_Sum:{
   1080 				sd->bake.data_kind = BeamformerDataKind_Float32;
   1081 				if (cp->iq_pipeline)
   1082 					sd->bake.data_kind = BeamformerDataKind_Float32Complex;
   1083 
   1084 				sd->layout   = layout_for_output(cp->output_points);
   1085 				sd->dispatch = dispatch_for_output(sd->layout, cp->output_points);
   1086 
   1087 				commit = 1;
   1088 			}break;
   1089 			#endif
   1090 
   1091 			}
   1092 		}
   1093 	}
   1094 
   1095 	cp->pipeline.data_kind = input_data_kind;
   1096 
   1097 	if (cp->first_image_shader_index == 0)
   1098 		cp->first_image_shader_index = cp->pipeline.shader_count;
   1099 
   1100 	gpu_resource_build_end(resource_builder, &cp->gpu_temp_arena);
   1101 }
   1102 
   1103 function void
   1104 stream_append_shader_header(Stream *s, i32 reloadable_index, u64 gpu_heap_pointer, BeamformerShaderDescriptor *sd, uv3 layout)
   1105 {
   1106 	stream_append_str8(s, str8("#version 460 core\n\n"
   1107 	"#extension GL_EXT_buffer_reference : require\n"
   1108 	"#extension GL_EXT_shader_16bit_storage : require\n"
   1109 	"#extension GL_EXT_shader_explicit_arithmetic_types : require\n\n"
   1110 	"#define f32     float32_t\n"
   1111 	"#define f16     float16_t\n"
   1112 	"#define s32     int32_t\n"
   1113 	"#define u64     uint64_t\n"
   1114 	"#define u32     uint32_t\n"
   1115 	"#define s16     int16_t\n"
   1116 	"#define u16     uint16_t\n"
   1117 	"#define u8      uint8_t\n"
   1118 	"#define s32vec2 i32vec2\n"
   1119 	"#define s16vec2 i16vec2\n"
   1120 	"\n"));
   1121 
   1122 	i32  header_vector_length = beamformer_shader_header_vector_lengths[reloadable_index];
   1123 	i32 *header_vector        = (i32 *)beamformer_shader_header_vectors[reloadable_index];
   1124 	for (i32 index = 0; index < header_vector_length; index++)
   1125 		stream_append_str8(s, beamformer_shader_global_header_strings[header_vector[index]]);
   1126 
   1127 	if (layout.x != 0) {
   1128 		stream_append_str8(s, str8("layout(local_size_x = "));
   1129 		stream_append_u64(s,  layout.x);
   1130 		stream_append_str8(s, str8(", local_size_y = "));
   1131 		stream_append_u64(s,  layout.y);
   1132 		stream_append_str8(s, str8(", local_size_z = "));
   1133 		stream_append_u64(s,  layout.z);
   1134 		stream_append_str8(s, str8(") in;\n\n"));
   1135 	}
   1136 
   1137 	{
   1138 		u32 max_length = 0;
   1139 		for EachElement(beamformer_data_kind_str8, it)
   1140 			max_length = Max(max_length, (u32)beamformer_data_kind_str8[it].length);
   1141 
   1142 		for EachElement(beamformer_data_kind_str8, it) {
   1143 			stream_append_str8s(s, str8("#define DataKind_"), beamformer_data_kind_str8[it]);
   1144 			stream_pad(s, ' ', max_length - beamformer_data_kind_str8[it].length + 1);
   1145 			stream_append_u64(s, it);
   1146 			stream_append_byte(s, '\n');
   1147 		}
   1148 		stream_append_byte(s, '\n');
   1149 	}
   1150 
   1151 	if (gpu_heap_pointer) {
   1152 		stream_append_str8(s, str8("#define HeapBase u64(0x"));
   1153 		stream_append_hex_u64(s, gpu_heap_pointer);
   1154 		stream_append_str8(s, str8("ul)\n"));
   1155 	}
   1156 
   1157 	if (sd) {
   1158 		BeamformerDataKind data_kinds[] = {sd->input_data_kind, sd->output_data_kind};
   1159 		str8 line_prefixes[] = {str8_comp("Input"), str8_comp("Output")};
   1160 		for EachElement(data_kinds, it) {
   1161 			if (data_kinds[it] != BeamformerDataKind_Count) {
   1162 				stream_append_str8s(s, str8("#define "), line_prefixes[it], str8("DataType "),
   1163 				                    beamformer_data_kind_glsl_type[data_kinds[it]],
   1164 				                    str8("\n#define "), line_prefixes[it], str8("DataKind DataKind_"),
   1165 				                    beamformer_data_kind_str8[data_kinds[it]],
   1166 				                    str8("\n#define "), line_prefixes[it], str8("DataKindByteSize "));
   1167 				stream_append_u64(s, beamformer_data_kind_byte_size[data_kinds[it]]);
   1168 				stream_append_byte(s, '\n');
   1169 			}
   1170 		}
   1171 		stream_append_byte(s, '\n');
   1172 
   1173 		stream_append_str8(s, str8("#define CompileFlags (0x"));
   1174 		stream_append_hex_u64_width(s, sd->compile_flags, 8);
   1175 		stream_append_str8(s, str8(")\n"));
   1176 
   1177 		i32 struct_id = beamformer_base_shader_to_bake_struct_id[reloadable_index];
   1178 		if (struct_id != -1) {
   1179 			const str8             *names = meta_struct_member_names_by_id[struct_id];
   1180 			const MetaStructInfo   *si    = meta_struct_info_by_id + struct_id;
   1181 			const MetaStructMember *sm    = meta_struct_members_by_id[struct_id];
   1182 			for (u32 index = 0; index < si->member_count; index++) {
   1183 				str8 type = meta_kind_glsl_types[sm[index].type_id];
   1184 				stream_append_str8(s, str8("layout(constant_id = "));
   1185 				stream_append_u64(s, index);
   1186 				stream_append_str8s(s, str8(") const "), type, str8(" "), names[index], str8(" = "), type, str8("(1);\n"));
   1187 			}
   1188 		}
   1189 	}
   1190 
   1191 	if (!renderdoc_attached())
   1192 		stream_append_str8(s, str8("\n\n#line 1\n"));
   1193 }
   1194 
   1195 function void
   1196 beamformer_reload_pipeline(VulkanHandle *pipeline, BeamformerShaderReloadInfo *sris, u32 count, Arena *scratch)
   1197 {
   1198 	assume(count <= 2);
   1199 	str8 paths[2];
   1200 	VulkanPipelineCreateInfo infos[2];
   1201 
   1202 	if (!BakeShaders) {
   1203 		for (u32 i = 0; i < count; i++)
   1204 			paths[i] = push_str8_from_parts(scratch, os_path_separator(), str8("shaders"), sris[i].filename_or_data);
   1205 	}
   1206 
   1207 	u32 push_constants_size = 0;
   1208 	for (u32 i = 0; i < count; i++) {
   1209 		Stream shader_stream = arena_stream(scratch);
   1210 		i32 reloadable_index = beamformer_shader_reloadable_index_by_shader[sris[i].shader];
   1211 		if (i == 0) push_constants_size = beamformer_shader_push_constant_sizes[reloadable_index];
   1212 		else        assert(push_constants_size == beamformer_shader_push_constant_sizes[reloadable_index]);
   1213 
   1214 		stream_append_shader_header(&shader_stream, reloadable_index, sris[i].gpu_heap_pointer,
   1215 		                            sris[i].shader_descriptor, sris[i].layout);
   1216 
   1217 		str8 shader_text;
   1218 		if (BakeShaders) {
   1219 			stream_append_str8(&shader_stream, sris[i].filename_or_data);
   1220 			shader_text = arena_stream_commit_zero(scratch, &shader_stream);
   1221 		} else {
   1222 			str8 stream_data = arena_stream_commit(scratch, &shader_stream);
   1223 			str8 shader_data = os_read_entire_file(scratch, (c8 *)paths[i].data);
   1224 			// NOTE(rnp): kinda sucky but need to make sure these are a contiguous string
   1225 			shader_text = push_str8_from_parts(scratch, str8(""), stream_data, shader_data);
   1226 		}
   1227 
   1228 		infos[i].kind = sris[i].shader_kind;
   1229 		infos[i].text = shader_text;
   1230 		infos[i].name = beamformer_shader_names[sris[i].shader];
   1231 		infos[i].specialization_data      = sris[i].shader_descriptor ? &sris[i].shader_descriptor->bake : 0;
   1232 		infos[i].specialization_struct_id = beamformer_base_shader_to_bake_struct_id[reloadable_index];
   1233 
   1234 		//str8 line = str8("---------------\n");
   1235 		//str8 nl   = str8("\n");
   1236 		//os_console_log(line.data, line.length);
   1237 		//os_console_log(infos[i].name.data, infos[i].name.length);
   1238 		//os_console_log(nl.data, nl.length);
   1239 		//os_console_log(line.data, line.length);
   1240 		//os_console_log(infos[i].text.data, infos[i].text.length);
   1241 		//os_console_log(line.data, line.length);
   1242 	}
   1243 
   1244 	vk_pipeline_release(*pipeline);
   1245 	*pipeline = vk_pipeline(infos, count, push_constants_size);
   1246 }
   1247 
   1248 function void
   1249 beamformer_reload_render_pipeline(VulkanHandle *pipeline, BeamformerShaderKind shader, Arena *scratch)
   1250 {
   1251 	i32 index = beamformer_shader_reloadable_index_by_shader[shader];
   1252 	BeamformerShaderReloadInfo infos[2] = {
   1253 		{
   1254 			.shader      = shader,
   1255 			.shader_kind = beamformer_shader_primitive_is_vertex[index] ? VulkanShaderKind_Vertex : VulkanShaderKind_Mesh,
   1256 			.filename_or_data = BakeShaders ? beamformer_shader_data[index][0]
   1257 			                                : beamformer_reloadable_shader_files[index][0],
   1258 		},
   1259 		{
   1260 			.shader           = shader,
   1261 			.shader_kind      = VulkanShaderKind_Fragment,
   1262 			.filename_or_data = BakeShaders ? beamformer_shader_data[index][1]
   1263 			                                : beamformer_reloadable_shader_files[index][1],
   1264 		},
   1265 	};
   1266 	beamformer_reload_pipeline(pipeline, infos, countof(infos), scratch);
   1267 }
   1268 
   1269 function void
   1270 beamformer_commit_parameter_block(BeamformerCtx *ctx, BeamformerComputePlan *cp, u32 block, Arena *scratch)
   1271 {
   1272 	BeamformerParameterBlock *pb;
   1273 	DeferLoop(pb = beamformer_parameter_block_lock(ctx->shared_memory, block, -1),
   1274 	          beamformer_parameter_block_unlock(ctx->shared_memory, block))
   1275 	for EachBit(pb->region_update_flags, region)
   1276 	{
   1277 		pb->region_update_flags &= ~(1ul << region);
   1278 		switch (region) {
   1279 		case BeamformerParameterDirtyFlag_NotifyUI:{
   1280 			atomic_store_u32(&ctx->ui_dirty_parameter_blocks, 1u << block);
   1281 		}break;
   1282 
   1283 		case BeamformerParameterDirtyFlag_Parameters:{
   1284 			cp->output_points  = das_valid_points(pb->parameters.output_points.xyz);
   1285 			cp->average_frames = pb->parameters.output_points.E[3];
   1286 
   1287 			plan_compute_pipeline(cp, pb, scratch);
   1288 
   1289 			b32 heap_pointer_updated = cp->last_gpu_heap_pointer != cp->gpu_temp_arena.gpu_pointer;
   1290 			cp->last_gpu_heap_pointer = cp->gpu_temp_arena.gpu_pointer;
   1291 
   1292 			for (u32 shader_slot = 0; shader_slot < cp->pipeline.shader_count; shader_slot++) {
   1293 				u128 hash = u128_hash_from_data(cp->shader_descriptors + shader_slot, sizeof(BeamformerShaderDescriptor));
   1294 				if (!u128_equal(hash, cp->shader_hashes[shader_slot]) ||
   1295 				    (cp->shader_descriptors[shader_slot].uses_heap && heap_pointer_updated))
   1296 				{
   1297 					cp->dirty_programs |= 1 << shader_slot;
   1298 				}
   1299 				cp->shader_hashes[shader_slot] = hash;
   1300 			}
   1301 
   1302 			cp->acquisition_count = pb->parameters.acquisition_count;
   1303 			cp->acquisition_kind  = pb->parameters.acquisition_kind;
   1304 			cp->contrast_mode     = pb->parameters.contrast_mode;
   1305 		}break;
   1306 		}
   1307 	}
   1308 }
   1309 
   1310 function void
   1311 do_compute_shader(GPUCommandList cmd, BeamformerComputePlan *cp, BeamformerFrame *frame,
   1312                   u64 input_pointer, u64 output_pointer, u32 shader_slot, u32 channel_offset)
   1313 {
   1314 	BeamformerComputeContext *cc = &beamformer_context->compute_context;
   1315 
   1316 	uv3 dispatch = cp->shader_descriptors[shader_slot].dispatch;
   1317 
   1318 	gpu_command_bind_pipeline(cmd, cp->vulkan_pipelines[shader_slot]);
   1319 
   1320 	switch (cp->pipeline.shaders[shader_slot]) {
   1321 
   1322 	case BeamformerShaderKind_Decode:{
   1323 		BeamformerDecodePushConstants pc = {
   1324 			.rf_buffer     = input_pointer,
   1325 			.output_buffer = output_pointer,
   1326 		};
   1327 
   1328 		gpu_command_push_constants(cmd, 0, sizeof(pc), &pc);
   1329 		gpu_command_dispatch_compute(cmd, dispatch);
   1330 
   1331 		cc->ping_pong_input_index = !cc->ping_pong_input_index;
   1332 	}break;
   1333 
   1334 	case BeamformerShaderKind_Hilbert:{
   1335 		//cuda_hilbert(input_index, output_index);
   1336 		//cc->ping_pong_input_index = !cc->ping_pong_input_index;
   1337 	}break;
   1338 
   1339 	case BeamformerShaderKind_Filter:
   1340 	case BeamformerShaderKind_Demodulate:
   1341 	{
   1342 		BeamformerFilterPushConstants pc = {
   1343 			.input_buffer  = input_pointer,
   1344 			.output_buffer = output_pointer,
   1345 		};
   1346 
   1347 		gpu_command_push_constants(cmd, 0, sizeof(pc), &pc);
   1348 		gpu_command_dispatch_compute(cmd, dispatch);
   1349 
   1350 		cc->ping_pong_input_index = !cc->ping_pong_input_index;
   1351 	}break;
   1352 
   1353 	case BeamformerShaderKind_DAS:{
   1354 		BeamformerDASPushConstants pc = {
   1355 			.xdc_element_pitch = cp->xdc_element_pitch,
   1356 			.rf_data           = input_pointer,
   1357 			.output_frame      = frame->gpu_pointer,
   1358 			.channel_offset    = channel_offset,
   1359 			.readi_group       = cp->readi_group,
   1360 		};
   1361 		memory_copy(pc.voxel_transform.E, cp->das_voxel_transform.E, sizeof(pc.voxel_transform));
   1362 		memory_copy(pc.xdc_transform.E,   cp->xdc_transform.E,       sizeof(pc.xdc_transform));
   1363 
   1364 		gpu_command_push_constants(cmd, 0, sizeof(pc), &pc);
   1365 		gpu_command_dispatch_compute(cmd, dispatch);
   1366 	}break;
   1367 
   1368 	case BeamformerShaderKind_CoherencyWeighting:{
   1369 		BeamformerCoherencyWeightingPushConstants pc = {.coherent_sum = frame->gpu_pointer};
   1370 		gpu_command_push_constants(cmd, 0, sizeof(pc), &pc);
   1371 		gpu_command_dispatch_compute(cmd, dispatch);
   1372 	}break;
   1373 
   1374 	case BeamformerShaderKind_Reshape:{
   1375 		BeamformerDataKind input_data_kind = cp->shader_descriptors[shader_slot].input_data_kind;
   1376 		BeamformerReshapeBakeParameters *rb = &cp->shader_descriptors[shader_slot].bake.Reshape;
   1377 		BeamformerReshapePushConstants pc = {
   1378 			.left_input_buffer  = input_pointer,
   1379 			.right_input_buffer = input_pointer + rb->SizeX * rb->SizeY * rb->SizeZ
   1380 			                                      * beamformer_data_kind_byte_size[input_data_kind],
   1381 			.output_buffer      = output_pointer,
   1382 		};
   1383 
   1384 		gpu_command_push_constants(cmd, 0, sizeof(pc), &pc);
   1385 		gpu_command_dispatch_compute(cmd, dispatch);
   1386 
   1387 		cc->ping_pong_input_index = !cc->ping_pong_input_index;
   1388 	}break;
   1389 
   1390 	// NOTE(rnp): invalid stages should be filtered in planning phase
   1391 	InvalidDefaultCase;
   1392 	}
   1393 
   1394 	#if 0
   1395 	switch (shader) {
   1396 	case BeamformerShaderKind_MinMax:{
   1397 		for (u32 i = 1; i < frame->image.mip_map_levels; i++) {
   1398 			glBindImageTexture(0, frame->texture, i - 1, GL_TRUE, 0, GL_READ_ONLY,  GL_RG32F);
   1399 			glBindImageTexture(1, frame->texture, i - 0, GL_TRUE, 0, GL_WRITE_ONLY, GL_RG32F);
   1400 			glProgramUniform1i(program, MIN_MAX_MIPS_LEVEL_UNIFORM_LOC, i);
   1401 
   1402 			u32 width  = (u32)frame->dim.x >> i;
   1403 			u32 height = (u32)frame->dim.y >> i;
   1404 			u32 depth  = (u32)frame->dim.z >> i;
   1405 			glDispatchCompute(ORONE(width / 32), ORONE(height), ORONE(depth / 32));
   1406 			glMemoryBarrier(GL_SHADER_IMAGE_ACCESS_BARRIER_BIT);
   1407 		}
   1408 	}break;
   1409 	case BeamformerShaderKind_Sum:{
   1410 		u32 aframe_index = ctx->averaged_frame_index % countof(ctx->averaged_frames);
   1411 		BeamformerFrame *aframe = ctx->averaged_frames + aframe_index;
   1412 		aframe->id              = ctx->averaged_frame_index;
   1413 		atomic_store_u32(&aframe->ready_to_present, 0);
   1414 		/* TODO(rnp): hack we need a better way of specifying which frames to sum;
   1415 		 * this is fine for rolling averaging but what if we want to do something else */
   1416 		assert(frame >= ctx->beamform_frames);
   1417 		assert(frame < ctx->beamform_frames + countof(ctx->beamform_frames));
   1418 		u32 base_index   = (u32)(frame - ctx->beamform_frames);
   1419 		u32 to_average   = (u32)cp->average_frames;
   1420 		u32 frame_count  = 0;
   1421 		u32 *in_textures = push_array(&arena, u32, BeamformerMaxBacklogFrames);
   1422 		ComputeFrameIterator cfi = compute_frame_iterator(ctx, 1 + base_index - to_average, to_average);
   1423 		for (BeamformerFrame *it = frame_next(&cfi); it; it = frame_next(&cfi))
   1424 			in_textures[frame_count++] = it->texture;
   1425 
   1426 		assert(to_average == frame_count);
   1427 
   1428 		glProgramUniform1f(program, SUM_PRESCALE_UNIFORM_LOC, 1 / (f32)frame_count);
   1429 		/* NOTE: zero output before summing */
   1430 		glClearTexImage(aframe->texture, 0, GL_RED, GL_FLOAT, 0);
   1431 		glMemoryBarrier(GL_TEXTURE_UPDATE_BARRIER_BIT);
   1432 
   1433 		glBindImageTexture(0, out_texture, 0, GL_TRUE, 0, GL_READ_WRITE, GL_RG32F);
   1434 		for (u32 i = 0; i < in_texture_count; i++) {
   1435 			glBindImageTexture(1, in_textures[i], 0, GL_TRUE, 0, GL_READ_ONLY, GL_RG32F);
   1436 			glDispatchCompute(dispatch.x, dispatch.y, dispatch.z);
   1437 			glMemoryBarrier(GL_SHADER_IMAGE_ACCESS_BARRIER_BIT);
   1438 		}
   1439 
   1440 		memory_copy(aframe->voxel_transform.E,  frame->voxel_transform.E, sizeof(frame->voxel_transform));
   1441 		aframe->compound_count   = frame->compound_count;
   1442 		aframe->acquisition_kind = frame->acquisition_kind;
   1443 	}break;
   1444 	}
   1445 	#endif
   1446 }
   1447 
   1448 function void
   1449 complete_queue(BeamformerCtx *ctx, BeamformWorkQueue *q, Arena *arena)
   1450 {
   1451 	BeamformerComputeContext * cs = &ctx->compute_context;
   1452 	BeamformerSharedMemory *   sm = ctx->shared_memory;
   1453 
   1454 	for (BeamformWork *work = beamform_work_queue_pop(q);
   1455 	     work;
   1456 	     beamform_work_queue_pop_commit(q), work = beamform_work_queue_pop(q))
   1457 	{
   1458 		switch (work->kind) {
   1459 
   1460 		case BeamformerWorkKind_ExportBuffer:{
   1461 			/* TODO(rnp): better way of handling DispatchCompute barrier */
   1462 			post_sync_barrier(ctx->shared_memory, BeamformerSharedMemoryLockKind_DispatchCompute);
   1463 			beamformer_shared_memory_take_lock(ctx->shared_memory, (i32)work->lock, (u32)-1);
   1464 			BeamformerExportContext *ec = &work->export_context;
   1465 			switch (ec->kind) {
   1466 			case BeamformerExportKind_BeamformedData:{
   1467 				BeamformerFrameBacklog *bl = &ctx->compute_context.backlog;
   1468 				u32 req_count = Clamp(ec->count, 1, bl->counter);
   1469 				u32 frame_idx = bl->counter - req_count;
   1470 				u8 *sm_output = beamformer_shared_memory_data_pointer(sm, ctx->shared_memory_size);
   1471 				u64 exported_size = 0;
   1472 				for (u32 export_count = 0; export_count < req_count; export_count++, frame_idx++) {
   1473 					BeamformerFrame *f = bl->frames + frame_idx % countof(bl->frames);
   1474 					u64 frame_size = beamformer_frame_byte_size(f->points, f->data_kind);
   1475 					assert((frame_size & 63) == 0);
   1476 					// NOTE(tkh) we don't want to assume that all req_count frames are the same size,
   1477 					// so we either need to count the total size of all requested frames first or
   1478 					// just fill up as much as possible.
   1479 					if (exported_size + frame_size <= ec->size) {
   1480 						u64 offset = f->gpu_pointer - bl->buffer->gpu_pointer;
   1481 						gpu_host_wait_timeline(GPUTimeline_Compute, f->timeline_valid_value, -1ULL);
   1482 						gpu_buffer_range_download(sm_output + exported_size, bl->buffer, offset, frame_size, 1);
   1483 						exported_size += frame_size;
   1484 					}
   1485 				}
   1486 			}break;
   1487 
   1488 			case BeamformerExportKind_Stats:{
   1489 				ComputeTimingTable *table = ctx->compute_timing_table;
   1490 				/* NOTE(rnp): do a little spin to let this finish updating */
   1491 				spin_wait(table->write_index != atomic_load_u32(&table->read_index));
   1492 				ComputeShaderStats *stats = ctx->compute_shader_stats;
   1493 				if (sizeof(stats->table) <= ec->size)
   1494 					memory_copy(beamformer_shared_memory_data_pointer(sm, ctx->shared_memory_size),
   1495 					         &stats->table, sizeof(stats->table));
   1496 			}break;
   1497 			InvalidDefaultCase;
   1498 			}
   1499 			beamformer_shared_memory_release_lock(ctx->shared_memory, work->lock);
   1500 			post_sync_barrier(ctx->shared_memory, BeamformerSharedMemoryLockKind_ExportSync);
   1501 		}break;
   1502 
   1503 		case BeamformerWorkKind_CreateFilter:{
   1504 			BeamformerCreateFilterContext *fctx = &work->create_filter_context;
   1505 			u32 block = fctx->parameter_block;
   1506 			u32 slot  = fctx->filter_slot;
   1507 			BeamformerComputePlan *cp = beamformer_compute_plan_for_block(cs, block, arena);
   1508 			cp->filter_parameters[slot] = fctx->parameters;
   1509 		}break;
   1510 
   1511 		case BeamformerWorkKind_WaitThenCompute:
   1512 		case BeamformerWorkKind_Compute:
   1513 		{
   1514 			push_compute_timing_info(ctx->compute_timing_table,
   1515 			                         (ComputeTimingInfo){.kind = ComputeTimingInfoKind_ComputeFrameBegin});
   1516 
   1517 			BeamformerComputePlan *cp = beamformer_compute_plan_for_block(cs, work->compute_context.parameter_block, arena);
   1518 			if unlikely(beamformer_parameter_block_dirty(sm, work->compute_context.parameter_block)) {
   1519 				u32 block = work->compute_context.parameter_block;
   1520 				Temp scratch = temp_begin(arena);
   1521 				beamformer_commit_parameter_block(ctx, cp, block, arena);
   1522 				temp_end(scratch);
   1523 			}
   1524 
   1525 			post_sync_barrier(ctx->shared_memory, BeamformerSharedMemoryLockKind_DispatchCompute);
   1526 
   1527 			u32 dirty_programs = atomic_swap_u32(&cp->dirty_programs, 0);
   1528 			static_assert(BeamformerMaxComputeShaderStages <= 32, "");
   1529 			if unlikely(dirty_programs) {
   1530 				for EachBit(dirty_programs, slot) {
   1531 					assert(slot < BeamformerMaxComputeShaderStages);
   1532 					Temp scratch;
   1533 					DeferLoop(scratch = temp_begin(arena), temp_end(scratch))
   1534 					{
   1535 						u64 gpu_heap_pointer = cp->gpu_temp_arena.gpu_pointer;
   1536 						i32 index = beamformer_shader_reloadable_index_by_shader[cp->pipeline.shaders[slot]];
   1537 						BeamformerShaderReloadInfo info = {
   1538 							.shader      = cp->pipeline.shaders[slot],
   1539 							.shader_kind = VulkanShaderKind_Compute,
   1540 							.shader_descriptor = cp->shader_descriptors + slot,
   1541 							.gpu_heap_pointer  = cp->shader_descriptors[slot].uses_heap ? gpu_heap_pointer : 0,
   1542 							.filename_or_data  = BakeShaders ? beamformer_shader_data[index][0]
   1543 							                                 : beamformer_reloadable_shader_files[index][0],
   1544 							.layout            = cp->shader_descriptors[slot].layout,
   1545 						};
   1546 						beamformer_reload_pipeline(cp->vulkan_pipelines + slot, &info, 1, arena);
   1547 					}
   1548 				}
   1549 			}
   1550 
   1551 			atomic_store_u32(&cs->processing_compute, 1);
   1552 
   1553 			start_renderdoc_capture();
   1554 
   1555 			i32 das_index = -1;
   1556 			i32 coherency_weighting = -1;
   1557 			for (u32 i = 0; i < cp->pipeline.shader_count; i++) {
   1558 				if (cp->pipeline.shaders[i] == BeamformerShaderKind_CoherencyWeighting)
   1559 					coherency_weighting = (i32)i;
   1560 
   1561 				if (cp->pipeline.shaders[i] == BeamformerShaderKind_DAS)
   1562 					das_index = (i32)i;
   1563 			}
   1564 
   1565 			BeamformerFrame *frame  = beamformer_frame_next(cs, cp->output_points, cp->iq_pipeline);
   1566 			frame->acquisition_kind = cp->acquisition_kind;
   1567 			frame->contrast_mode    = cp->contrast_mode;
   1568 			frame->compound_count   = cp->acquisition_count;
   1569 			frame->parameter_block  = work->compute_context.parameter_block;
   1570 			frame->view_plane_tag   = work->compute_context.view_plane;
   1571 			memory_copy(frame->voxel_transform.E, cp->voxel_transform.E, sizeof(cp->voxel_transform));
   1572 
   1573 			GPUCommandList cmd = gpu_command_list_begin(GPUTimeline_Compute);
   1574 			gpu_command_timestamp(cmd);
   1575 
   1576 			if (das_index >= 0) {
   1577 				GPUBuffer *backlog = cs->backlog.buffer;
   1578 				u64 frame_size = beamformer_frame_byte_size(frame->points, frame->data_kind);
   1579 				u64 offset     = frame->gpu_pointer - backlog->gpu_pointer;
   1580 				gpu_command_clear_buffer(cmd, backlog, offset, frame_size, 0);
   1581 			}
   1582 
   1583 			if (coherency_weighting >= 0) {
   1584 				BeamformerCoherencyWeightingBakeParameters *cw = &cp->shader_descriptors[coherency_weighting].bake.CoherencyWeighting;
   1585 				GPUBuffer *gpu_arena = &cp->gpu_temp_arena;
   1586 				u64 coherent_size = beamformer_incoherent_frame_byte_size(frame->points, frame->data_kind);
   1587 				gpu_command_clear_buffer(cmd, gpu_arena, cw->IncoherentSum, coherent_size, 0);
   1588 			}
   1589 
   1590 			BeamformerRFBuffer *rf = &cs->rf_buffer;
   1591 			u64 compute_index = rf->compute_index;
   1592 			u64 slot = compute_index % countof(rf->upload_complete_values);
   1593 
   1594 			// NOTE(rnp): the library/ui will ensure that any time a new dataset is uploaded
   1595 			// the next time beamforming is issued it will be tagged as WaitThenCompute.
   1596 			if (work->kind != BeamformerWorkKind_WaitThenCompute)
   1597 				slot = (compute_index - 1) % countof(rf->upload_complete_values);
   1598 
   1599 			// NOTE(rnp): nvidia needs a memory barrier between pipeline stages
   1600 			// for correct output. It doesn't seem to effect performance on nvidia cards.
   1601 			// It does have a very minor performance decrease on AMD, but technically its
   1602 			// against vulkan spec to skip them so for simplicity we will default to for now
   1603 			b32 pipeline_memory_barrier = 1; //gpu_info()->vendor == GPUVendor_NVIDIA;
   1604 
   1605 			for (u32 channel_offset = 0;
   1606 			     channel_offset < cp->channel_count;
   1607 			     channel_offset += BeamformerChunkChannelCount)
   1608 			{
   1609 				u64 rf_pointer = rf->buffer.gpu_pointer + slot * rf->active_rf_size;
   1610 				rf_pointer += cp->raw_channel_byte_stride * channel_offset;
   1611 
   1612 				// NOTE(rnp): handle the case where the channel count is not a multiple of
   1613 				// BeamformerChunkChannelCount. We do not want to have to modify every shader
   1614 				// that might run to care about this so here we run a copy/clear to pad the
   1615 				// data for the last iteration.
   1616 				b32 special_handling = (cp->channel_count - channel_offset) < BeamformerChunkChannelCount;
   1617 				if unlikely(special_handling) {
   1618 					u32 channel_pad_count = BeamformerChunkChannelCount - (cp->channel_count - channel_offset);
   1619 					u32 input_index = cs->ping_pong_input_index;
   1620 					u64 pp_size     = cs->ping_pong_buffer.size / PING_PONG_BUFFER_SLOTS;
   1621 					u64 copy_size   = cp->raw_channel_byte_stride * (cp->channel_count - channel_offset);
   1622 					u64 copy_offset = rf_pointer - rf->buffer.gpu_pointer;
   1623 					u64 clear_size  = cp->raw_channel_byte_stride * channel_pad_count;
   1624 
   1625 					gpu_command_copy_buffer(cmd, &cs->ping_pong_buffer, input_index * pp_size, &rf->buffer, copy_offset, copy_size);
   1626 					gpu_command_clear_buffer(cmd, &cs->ping_pong_buffer, input_index * pp_size + copy_size, clear_size, 0);
   1627 					// TODO(rnp): technically this should be a transfer to compute barrier
   1628 					// but a full memory barrier will also do the trick
   1629 					gpu_command_pipeline_barrier(cmd, 1);
   1630 				}
   1631 
   1632 				for (u32 i = 0; i < cp->first_image_shader_index; i++) {
   1633 					u32 output_index      = !cs->ping_pong_input_index;
   1634 					u32 input_index       =  cs->ping_pong_input_index;
   1635 
   1636 					u64 pp_size           = cs->ping_pong_buffer.size / PING_PONG_BUFFER_SLOTS;
   1637 					u64 pp_input_pointer  = cs->ping_pong_buffer.gpu_pointer + input_index  * pp_size;
   1638 					u64 pp_output_pointer = cs->ping_pong_buffer.gpu_pointer + output_index * pp_size;
   1639 
   1640 					u64 output_pointer = pp_output_pointer;
   1641 					u64 input_pointer  = (i == 0 && !special_handling) ? rf_pointer : pp_input_pointer;
   1642 
   1643 					if (i != 0) gpu_command_pipeline_barrier(cmd, pipeline_memory_barrier);
   1644 					do_compute_shader(cmd, cp, frame, input_pointer, output_pointer, i, channel_offset);
   1645 					gpu_command_timestamp(cmd);
   1646 				}
   1647 			}
   1648 
   1649 			for (u32 i = cp->first_image_shader_index; i < cp->pipeline.shader_count; i++) {
   1650 				gpu_command_pipeline_barrier(cmd, pipeline_memory_barrier);
   1651 				do_compute_shader(cmd, cp, frame, 0, 0, i, 0);
   1652 				gpu_command_timestamp(cmd);
   1653 			}
   1654 
   1655 			u32 wait_info_count = 0;
   1656 			GPUSemaphoreSignalInfo wait_info = {.semaphore = rf->upload_semaphore};
   1657 			if (work->kind == BeamformerWorkKind_WaitThenCompute) {
   1658 				spin_wait(atomic_load_u64(&rf->insertion_index) <= compute_index);
   1659 				wait_info.value = atomic_load_u64(rf->upload_complete_values + slot);
   1660 				wait_info_count = 1;
   1661 			}
   1662 
   1663 			u64 end_timeline_value = gpu_command_list_end(cmd, &wait_info, wait_info_count, 0, 0);
   1664 			atomic_store_u64(rf->compute_complete_values + slot, end_timeline_value);
   1665 
   1666 			if (work->kind == BeamformerWorkKind_WaitThenCompute)
   1667 				atomic_add_u64(&rf->compute_index, 1);
   1668 
   1669 			atomic_store_u64(&frame->timeline_valid_value, end_timeline_value);
   1670 
   1671 			Temp scratch;
   1672 			DeferLoop(scratch = temp_begin(arena), temp_end(scratch))
   1673 			{
   1674 				/* NOTE(rnp): this blocks until work completes */
   1675 				u64  count       = 0;
   1676 				u64 *timestamps  = gpu_read_timestamps(GPUTimeline_Compute, &count, arena);
   1677 
   1678 				i32 steps        = ((i32)cp->channel_count / BeamformerChunkChannelCount) - 1;
   1679 				i32 step         = 0;
   1680 				u32 shader_index = 0;
   1681 				u64 last_time    = count > 0 ? timestamps[0] : 0;
   1682 
   1683 				for (u64 i = 1; i < count; i++) {
   1684 					push_compute_timing_info(ctx->compute_timing_table, (ComputeTimingInfo){
   1685 						.kind        = ComputeTimingInfoKind_Shader,
   1686 						.shader      = cp->pipeline.shaders[shader_index],
   1687 						.shader_slot = shader_index,
   1688 						.timer_count = timestamps[i] - last_time,
   1689 					});
   1690 					last_time = timestamps[i];
   1691 
   1692 					shader_index++;
   1693 					if (shader_index == cp->first_image_shader_index && step < steps) {
   1694 						shader_index = 0;
   1695 						step++;
   1696 					}
   1697 				}
   1698 			}
   1699 
   1700 			cs->processing_progress = 1;
   1701 
   1702 			//if (has_sum) {
   1703 			if (0) {
   1704 				#if 0
   1705 				u32 aframe_index = ((ctx->averaged_frame_index++) % countof(ctx->averaged_frames));
   1706 				ctx->averaged_frames[aframe_index].view_plane_tag  = frame->view_plane_tag;
   1707 				ctx->averaged_frames[aframe_index].ready_to_present = 1;
   1708 				atomic_store_u64((u64 *)&ctx->latest_frame, (u64)(ctx->averaged_frames + aframe_index));
   1709 				#endif
   1710 			} else {
   1711 				atomic_store_u64((u64 *)&ctx->latest_frame, (u64)frame);
   1712 			}
   1713 
   1714 			atomic_store_u32(&cs->processing_compute, 0);
   1715 
   1716 			push_compute_timing_info(ctx->compute_timing_table,
   1717 			                         (ComputeTimingInfo){.kind = ComputeTimingInfoKind_ComputeFrameEnd});
   1718 
   1719 			end_renderdoc_capture();
   1720 		}break;
   1721 		InvalidDefaultCase;
   1722 		}
   1723 	}
   1724 }
   1725 
   1726 function void
   1727 coalesce_timing_table(ComputeTimingTable *t, ComputeShaderStats *stats)
   1728 {
   1729 	/* TODO(rnp): we do not currently do anything to handle the potential for a half written
   1730 	 * info item. this could result in garbage entries but they shouldn't really matter */
   1731 
   1732 	u32 target = atomic_load_u32(&t->write_index);
   1733 	u32 stats_index = stats->latest_frame_index;
   1734 
   1735 	b32 has_rf = 0;
   1736 	f32 gpu_clocks_to_nano = 1.0e-9f * gpu_info()->timestamp_period_ns;
   1737 
   1738 	// NOTE(rnp): not equal (the index may wrap)
   1739 	while (t->read_index != target) {
   1740 		ComputeTimingInfo info = t->buffer[t->read_index % countof(t->buffer)];
   1741 		switch (info.kind) {
   1742 
   1743 		case ComputeTimingInfoKind_ComputeFrameBegin:{
   1744 			assert(t->compute_frame_active == 0);
   1745 			t->compute_frame_active = 1;
   1746 			/* NOTE(rnp): allow multiple instances of same shader to accumulate */
   1747 			t->in_flight_shader_count = 0;
   1748 			memory_clear(t->in_flight_shader_ids, 0, sizeof(t->in_flight_shader_ids));
   1749 			memory_clear(stats->table.times[stats_index], 0, sizeof(stats->table.times[stats_index]));
   1750 		}break;
   1751 
   1752 		case ComputeTimingInfoKind_ComputeFrameEnd:{
   1753 			assert(t->compute_frame_active == 1);
   1754 			t->compute_frame_active = 0;
   1755 			stats_index = stats->latest_frame_index = (stats_index + 1) % countof(stats->table.times);
   1756 			stats->table.shader_count = t->in_flight_shader_count;
   1757 			memory_copy(stats->table.shader_ids, t->in_flight_shader_ids, sizeof(t->in_flight_shader_ids));
   1758 		}break;
   1759 
   1760 		case ComputeTimingInfoKind_Shader:{
   1761 			t->in_flight_shader_count = Max(t->in_flight_shader_count, info.shader_slot + 1u);
   1762 			t->in_flight_shader_ids[info.shader_slot] = info.shader;
   1763 			stats->table.times[stats_index][info.shader_slot] += info.timer_count * gpu_clocks_to_nano;
   1764 		}break;
   1765 
   1766 		case ComputeTimingInfoKind_RF_Data:{
   1767 			stats->latest_rf_index = (stats->latest_rf_index + 1) % countof(stats->table.rf_time_deltas);
   1768 			f32 delta = info.timer_count / (f32)os_system_info()->timer_frequency;
   1769 			stats->table.rf_time_deltas[stats->latest_rf_index] = delta;
   1770 			has_rf = 1;
   1771 		}break;
   1772 		}
   1773 		/* NOTE(rnp): do this at the end so that stats table is always in a consistent state */
   1774 		t->read_index++;
   1775 	}
   1776 
   1777 	for (u32 i = 0; i < stats->table.shader_count; i++) {
   1778 		f32 sum = 0;
   1779 		for EachElement(stats->table.times, it)
   1780 			sum += stats->table.times[it][i];
   1781 		stats->average_times[i] = sum / countof(stats->table.times);
   1782 	}
   1783 
   1784 	if (has_rf) {
   1785 		f32 sum = 0;
   1786 		for EachElement(stats->table.rf_time_deltas, i)
   1787 			sum += stats->table.rf_time_deltas[i];
   1788 		stats->rf_time_delta_average = sum / countof(stats->table.rf_time_deltas);
   1789 	}
   1790 }
   1791 
   1792 DEBUG_EXPORT BEAMFORMER_COMPLETE_COMPUTE_FN(beamformer_complete_compute)
   1793 {
   1794 	BeamformerSharedMemory *sm = ctx->shared_memory;
   1795 	complete_queue(ctx, &sm->external_work_queue, arena);
   1796 	complete_queue(ctx, ctx->beamform_work_queue, arena);
   1797 }
   1798 
   1799 DEBUG_EXPORT BEAMFORMER_RF_UPLOAD_FN(beamformer_rf_upload)
   1800 {
   1801 	BeamformerSharedMemory *sm                  = ctx->shared_memory;
   1802 	BeamformerSharedMemoryLockKind scratch_lock = BeamformerSharedMemoryLockKind_ScratchSpace;
   1803 	BeamformerSharedMemoryLockKind upload_lock  = BeamformerSharedMemoryLockKind_UploadRF;
   1804 
   1805 	u64 rf_block_rf_size;
   1806 	if (atomic_load_u32(sm->locks + upload_lock) &&
   1807 	    (rf_block_rf_size = atomic_swap_u64(&sm->rf_block_rf_size, 0)))
   1808 	{
   1809 		beamformer_shared_memory_take_lock(ctx->shared_memory, (i32)scratch_lock, (u32)-1);
   1810 
   1811 		BeamformerRFBuffer *rf = ctx->rf_buffer;
   1812 
   1813 		rf->active_rf_size = gpu_round_up_to_sync_size(rf_block_rf_size & 0x00FFFFFFFFFFFFFFull, 64);
   1814 		if unlikely((u64)rf->buffer.size < countof(rf->upload_complete_values) * rf->active_rf_size) {
   1815 			gpu_buffer_allocate(&rf->buffer, (GPUBufferAllocateInfo){
   1816 				.size  = countof(rf->upload_complete_values) * rf->active_rf_size,
   1817 				.flags = GPUUsageFlag_HostWrite,
   1818 				.label = str8("RawRFBuffer"),
   1819 				.timelines_used = (GPUTimeline []){GPUTimeline_Compute},
   1820 				.timeline_count = 1,
   1821 			});
   1822 		}
   1823 
   1824 		u64 slot = rf->insertion_index % countof(rf->upload_complete_values);
   1825 
   1826 		/* NOTE(rnp): don't overwrite slot if the compute thread hasn't processed it */
   1827 		spin_wait(rf->insertion_index - atomic_load_u64(&rf->compute_index) >= countof(rf->upload_complete_values));
   1828 		gpu_host_wait_timeline(GPUTimeline_Compute, rf->compute_complete_values[slot], -1ULL);
   1829 
   1830 		assert((ctx->shared_memory_size % os_system_info()->page_size) == 0 &&
   1831 		       (os_system_info()->page_size % gpu_round_up_to_sync_size(1, 64)) == 0);
   1832 
   1833 		u64 wait_value = gpu_semaphore_value(rf->upload_semaphore) + 1;
   1834 		atomic_store_u64(rf->upload_complete_values + slot, wait_value);
   1835 		atomic_add_u64(&rf->insertion_index, 1);
   1836 
   1837 		void *host_pointer = gpu_buffer_host_pointer(&rf->buffer);
   1838 		memory_copy_non_temporal((u8 *)host_pointer + slot * rf->active_rf_size,
   1839 		                         beamformer_shared_memory_data_pointer(sm, ctx->shared_memory_size),
   1840 		                         rf->active_rf_size);
   1841 		store_fence();
   1842 
   1843 		GPUSemaphoreSignalInfo signal_info = {.semaphore = rf->upload_semaphore, .value = wait_value};
   1844 		gpu_buffer_make_visible(&rf->buffer, slot * rf->active_rf_size, rf->active_rf_size, &signal_info, 1);
   1845 
   1846 		beamformer_shared_memory_release_lock(ctx->shared_memory, (i32)scratch_lock);
   1847 		post_sync_barrier(ctx->shared_memory, upload_lock);
   1848 		os_wake_all_waiters(ctx->compute_worker_sync);
   1849 
   1850 		u64 current_time = os_timer_count();
   1851 		push_compute_timing_info(ctx->compute_timing_table, (ComputeTimingInfo){
   1852 			.kind        = ComputeTimingInfoKind_RF_Data,
   1853 			.timer_count = current_time - rf->timestamp,
   1854 		});
   1855 		rf->timestamp = current_time;
   1856 	}
   1857 }
   1858 
   1859 function void
   1860 beamformer_queue_compute(BeamformerCtx *ctx, BeamformerFrame *frame, u32 parameter_block)
   1861 {
   1862 	BeamformerSharedMemory *sm = ctx->shared_memory;
   1863 	BeamformerSharedMemoryLockKind dispatch_lock = BeamformerSharedMemoryLockKind_DispatchCompute;
   1864 	if (!sm->live_imaging_parameters.active && beamformer_shared_memory_take_lock(sm, (i32)dispatch_lock, 0))
   1865 	{
   1866 		BeamformWork *work = beamform_work_queue_push(ctx->beamform_work_queue);
   1867 		if (work) {
   1868 			work->kind = BeamformerWorkKind_Compute;
   1869 			work->compute_context.view_plane      = frame ? frame->view_plane_tag : 0;
   1870 			work->compute_context.parameter_block = parameter_block;
   1871 			beamform_work_queue_push_commit(ctx->beamform_work_queue);
   1872 		}
   1873 	}
   1874 	os_wake_all_waiters(&ctx->compute_worker.sync_variable);
   1875 }
   1876 
   1877 #include "ui.c"
   1878 
   1879 function void
   1880 beamformer_process_input_events(BeamformerCtx *ctx, BeamformerInput *input,
   1881                                 BeamformerInputEvent *events, u32 event_count)
   1882 {
   1883 	for (u32 index = 0; index < event_count; index++) {
   1884 		BeamformerInputEvent *event = events + index;
   1885 		switch (event->kind) {
   1886 
   1887 		// NOTE(rnp): ui will handle these
   1888 		case BeamformerInputEventKind_ButtonPress:
   1889 		case BeamformerInputEventKind_ButtonRelease:
   1890 		case BeamformerInputEventKind_MouseScroll:
   1891 		case BeamformerInputEventKind_WindowResize:
   1892 		{}break;
   1893 
   1894 		case BeamformerInputEventKind_ExecutableReload:{
   1895 			ui_init(ctx, ctx->ui_arena);
   1896 		}break;
   1897 
   1898 		case BeamformerInputEventKind_FileEvent:{
   1899 			BeamformerFileReloadContext *frc = event->file_watch_user_context;
   1900 			switch (frc->kind) {
   1901 			case BeamformerFileReloadKind_ComputeShader:{
   1902 				for EachElement(ctx->compute_context.compute_plans, block) {
   1903 					BeamformerComputePlan *cp = ctx->compute_context.compute_plans[block];
   1904 					for (u32 slot = 0; cp && slot < cp->pipeline.shader_count; slot++) {
   1905 						i32 shader_index = beamformer_shader_reloadable_index_by_shader[cp->pipeline.shaders[slot]];
   1906 						if (beamformer_reloadable_shader_kinds[shader_index] == frc->shader_reload.shader)
   1907 							atomic_or_u32(&cp->dirty_programs, 1 << slot);
   1908 					}
   1909 				}
   1910 
   1911 				// TODO(rnp): track latest parameter block
   1912 				if (ctx->latest_frame)
   1913 					beamformer_queue_compute(ctx, ctx->latest_frame, 0);
   1914 			}break;
   1915 
   1916 			case BeamformerFileReloadKind_RenderShader:{
   1917 				beamformer_reload_render_pipeline(frc->shader_reload.pipeline, frc->shader_reload.shader, ctx->arena);
   1918 				ctx->render_shader_updated = 1;
   1919 			}break;
   1920 
   1921 			InvalidDefaultCase;
   1922 			}
   1923 		}break;
   1924 
   1925 		InvalidDefaultCase;
   1926 		}
   1927 	}
   1928 }
   1929 
   1930 function void
   1931 beamformer_panel_group_insert_at(BeamformerUIPanel *group, BeamformerUIPanel *tab, u64 new_child_index)
   1932 {
   1933 	if (tab->parent) beamformer_ui_panel_unlink(tab);
   1934 	new_child_index = Min(new_child_index, group->child_count);
   1935 
   1936 	tab->parent = group;
   1937 	group->child_count++;
   1938 	if (group->kind == BeamformerPanelKind_TabGroup) group->u.tab_focus = tab;
   1939 
   1940 	BeamformerUIPanel *previous_sibling = new_child_index == 0 ? 0 : group->first_child;
   1941 	for (u64 child_index = 1; child_index < new_child_index; child_index++)
   1942 		previous_sibling = previous_sibling->next_sibling;
   1943 
   1944 	if (previous_sibling) {
   1945 		tab->previous_sibling = previous_sibling;
   1946 		tab->next_sibling     = previous_sibling->next_sibling;
   1947 		if (tab->next_sibling) tab->next_sibling->previous_sibling = tab;
   1948 		previous_sibling->next_sibling = tab;
   1949 		if (previous_sibling == group->last_child) group->last_child = tab;
   1950 	} else {
   1951 		DLLInsertFirst(0, group->first_child, group->last_child, tab, next_sibling, previous_sibling);
   1952 	}
   1953 }
   1954 
   1955 BEAMFORMER_EXPORT void
   1956 beamformer_frame_step(void *memory, BeamformerInput *input)
   1957 {
   1958 	BeamformerCtx *ctx = beamformer_context = memory;
   1959 	beamformer_input = input;
   1960 
   1961 	u64 current_time = os_timer_count();
   1962 	dt_for_frame = (f64)(current_time - ctx->frame_timestamp) / os_system_info()->timer_frequency;
   1963 	ctx->frame_timestamp = current_time;
   1964 	ctx->frame_index++;
   1965 
   1966 	coalesce_timing_table(ctx->compute_timing_table, ctx->compute_shader_stats);
   1967 
   1968 	// NOTE(rnp): reset frame state
   1969 	{
   1970 		ctx->registers = &ctx->base_registers;
   1971 		swap(ctx->command_queues[0], ctx->command_queues[1]);
   1972 		zero_struct(ctx->command_queues + 0);
   1973 		//zero_struct(ctx->registers);
   1974 		arena_clear(beamformer_frame_arena());
   1975 	}
   1976 
   1977 	beamformer_process_input_events(ctx, input, input->event_queue, input->event_count);
   1978 
   1979 	BeamformerSharedMemory *sm = ctx->shared_memory;
   1980 	u32 live_imaging_active = atomic_load_u32(&sm->live_imaging_parameters.active);
   1981 	if (live_imaging_active != ctx->live_imaging_active) {
   1982 		if (ctx->live_imaging_active) {
   1983 			if (ctx->auto_live_control_panel) {
   1984 				BeamformerUIPanel *parent = ctx->auto_live_control_panel->parent;
   1985 				beamformer_command(beamformer_command_infos[BeamformerCommandKind_CloseTab].string, .tree_node = (u64)ctx->auto_live_control_panel);
   1986 				if (parent->child_count == 1)
   1987 					beamformer_command(beamformer_command_infos[BeamformerCommandKind_CloseTab].string, .tree_node = (u64)parent);
   1988 			}
   1989 		} else {
   1990 			if (beamformer_registers()->live_controls) {
   1991 				beamformer_command(beamformer_command_infos[BeamformerCommandKind_FocusTab].string,
   1992 				                   .tree_node = beamformer_registers()->live_controls);
   1993 			} else {
   1994 				ctx->auto_live_control_panel = beamformer_ui_push_panel(0, BeamformerPanelKind_LiveImagingControls);
   1995 				beamformer_command(beamformer_command_infos[BeamformerCommandKind_SplitTree].string,
   1996 				                   .tree_node        = (u64)ctx->auto_live_control_panel,
   1997 				                   .split_axis       = Axis2_X,
   1998 				                   .split_left_tree  = (u64)ui_context->tree,
   1999 				                   .split_right_tree = 0,
   2000 				                   .drop_target_tree = (u64)ui_context->tree);
   2001 			}
   2002 			ctx->live_imaging_active_frame = ctx->frame_index;
   2003 		}
   2004 		ctx->live_imaging_active = live_imaging_active;
   2005 	}
   2006 
   2007 	if (atomic_load_u32(sm->locks + BeamformerSharedMemoryLockKind_UploadRF))
   2008 		os_wake_all_waiters(&ctx->upload_worker.sync_variable);
   2009 	if (atomic_load_u32(sm->locks + BeamformerSharedMemoryLockKind_DispatchCompute))
   2010 		os_wake_all_waiters(&ctx->compute_worker.sync_variable);
   2011 
   2012 	beamformer_registers()->frame = (u64)(ctx->latest_frame - ctx->compute_context.backlog.frames);
   2013 
   2014 	beamformer_ui_frame();
   2015 
   2016 	// NOTE(rnp): execute commands
   2017 	for (BeamformerCommandNode *node = ctx->command_queues[0].first;
   2018 	     node;
   2019 	     node = node == node->next ? 0 : node->next)
   2020 	{
   2021 		BeamformerRegistersScope()
   2022 		{
   2023 			memory_copy(beamformer_registers(), node->command.registers, sizeof(*node->command.registers));
   2024 			BeamformerCommandKind kind = beamformer_command_kind_from_string(node->command.name);
   2025 			switch (kind) {
   2026 			InvalidDefaultCase;
   2027 			case BeamformerCommandKind_CloseTab:{
   2028 				BeamformerUIPanel *tab = (BeamformerUIPanel *)beamformer_registers()->tree_node;
   2029 				ui_kill_panel(tab);
   2030 			}break;
   2031 
   2032 			case BeamformerCommandKind_FocusTab:{
   2033 				BeamformerUIPanel *tab = (BeamformerUIPanel *)beamformer_registers()->tree_node;
   2034 				assert(tab->parent->kind == BeamformerPanelKind_TabGroup);
   2035 				tab->parent->u.tab_focus = tab;
   2036 			}break;
   2037 
   2038 			case BeamformerCommandKind_MoveTab:{
   2039 				BeamformerUIPanel *move   = (BeamformerUIPanel *)beamformer_registers()->tree_node;
   2040 				BeamformerUIPanel *group  = (BeamformerUIPanel *)beamformer_registers()->drop_target_tree;
   2041 				BeamformerUIPanel *parent = move->parent;
   2042 				u64 new_child_index = beamformer_registers()->drop_child_index;
   2043 				beamformer_panel_group_insert_at(group, move, new_child_index);
   2044 
   2045 				if (move->kind == BeamformerPanelKind_LiveImagingControls) {
   2046 					beamformer_context->base_registers.v.live_controls = (u64)move;
   2047 					if (move == ctx->auto_live_control_panel)
   2048 						ctx->auto_live_control_panel = 0;
   2049 				}
   2050 
   2051 				if (parent->child_count == 0)
   2052 					beamformer_command(beamformer_command_infos[BeamformerCommandKind_CloseTab].string, .tree_node = (u64)parent);
   2053 			}break;
   2054 
   2055 			case BeamformerCommandKind_OpenTab:{
   2056 				BeamformerUIPanel *panel = (BeamformerUIPanel *)beamformer_registers()->tree_node;
   2057 				assert(panel->kind == BeamformerPanelKind_TabGroup);
   2058 
   2059 				BeamformerPanelKind new_panel_kind = beamformer_panel_kind_from_string(beamformer_registers()->string);
   2060 				beamformer_ui_push_panel(panel, new_panel_kind);
   2061 			}break;
   2062 
   2063 			case BeamformerCommandKind_SplitTree:{
   2064 				BeamformerUIPanel *drag  = (BeamformerUIPanel *)beamformer_registers()->tree_node;
   2065 				BeamformerUIPanel *left  = (BeamformerUIPanel *)beamformer_registers()->split_left_tree;
   2066 				BeamformerUIPanel *right = (BeamformerUIPanel *)beamformer_registers()->split_right_tree;
   2067 				Axis2 axis = beamformer_registers()->split_axis;
   2068 
   2069 				BeamformerUIPanel *new_split     = beamformer_ui_push_panel(0, BeamformerPanelKind_Split);
   2070 				BeamformerUIPanel *new_tab_group = beamformer_ui_push_panel(0, BeamformerPanelKind_TabGroup);
   2071 				beamformer_panel_group_insert_at(new_tab_group, drag, 0);
   2072 
   2073 				BeamformerUIPanel *target = 0;
   2074 				u32 target_child_index = 0;
   2075 				f32 new_split_pct = 0.5f;
   2076 
   2077 				if (left == 0 || right == 0) {
   2078 					// NOTE(rnp): split on edge of window
   2079 					target             = left ? left : right;
   2080 					target_child_index = left ? 0 : 1;
   2081 
   2082 					if (target->kind == BeamformerPanelKind_TabGroup) {
   2083 						new_split->kind        = BeamformerPanelKind_TabGroup;
   2084 						new_split->u.tab_focus = target->u.tab_focus;
   2085 					}
   2086 
   2087 					for (BeamformerUIPanel *child = target->last_child, *next; child; child = next) {
   2088 						next = child->previous_sibling;
   2089 						beamformer_panel_group_insert_at(new_split, child, 0);
   2090 					}
   2091 
   2092 					beamformer_panel_group_insert_at(target, new_tab_group, 0);
   2093 				} else if (((drag == left)  && right->kind == BeamformerPanelKind_Split) ||
   2094 				           ((drag == right) && left->kind  == BeamformerPanelKind_Split))
   2095 				{
   2096 					// NOTE(rnp): split on internal split
   2097 					target             = left == drag ? right : left;
   2098 					target_child_index = 1;
   2099 					new_split_pct      = 1.f / 3.f;
   2100 					beamformer_panel_group_insert_at(new_split, new_tab_group, 0);
   2101 					beamformer_panel_group_insert_at(new_split, target->last_child, 1);
   2102 				} else {
   2103 					// NOTE(rnp): TabGroup Split
   2104 					target             = left == drag ? right : left;
   2105 					target_child_index = left == drag ? 1 : 0;
   2106 					assert(target->kind == BeamformerPanelKind_TabGroup);
   2107 
   2108 					BeamformerUIPanel *focus = target->u.tab_focus;
   2109 					new_split->kind = BeamformerPanelKind_TabGroup;
   2110 					for (BeamformerUIPanel *child = target->last_child, *next; child; child = next) {
   2111 						next = child->previous_sibling;
   2112 						beamformer_panel_group_insert_at(new_split, child, 0);
   2113 					}
   2114 					new_split->u.tab_focus = focus;
   2115 
   2116 					beamformer_panel_group_insert_at(target, new_tab_group, 0);
   2117 				}
   2118 
   2119 				beamformer_panel_group_insert_at(target, new_split, target_child_index);
   2120 				if (target->kind == BeamformerPanelKind_Split) {
   2121 					new_split->u.split.axis     = target->u.split.axis;
   2122 					new_split->u.split.fraction = target->u.split.fraction;
   2123 				}
   2124 				target->kind             = BeamformerPanelKind_Split;
   2125 				target->u.split.axis     = axis;
   2126 				target->u.split.fraction = new_split_pct;
   2127 			}break;
   2128 
   2129 			}
   2130 		}
   2131 	}
   2132 
   2133 	ctx->render_shader_updated = 0;
   2134 }