vulkan.c (106391B)
1 /* See LICENSE for license details. */ 2 // TODO(rnp) 3 // [ ]: what is needed for HDR? I think it makes sense to just default to it nowadays 4 // [ ]: once opengl is removed switch images to SRGB and/or 16 bit Float 5 6 #include "beamformer_internal.h" 7 #include "vulkan.h" 8 #include "external/glslang/glslang/Include/glslang_c_interface.h" 9 10 #define VulkanDebug BEAMFORMER_DEBUG 11 12 #define ForceSingleQueue (0) 13 #define ForceStagingBuffers (0) 14 #define SupportNonCoherentHostMemory (0) 15 16 #define glslang_info(s) str8("[glslang] " s) 17 #define vulkan_info(s) str8("[vulkan] " s) 18 19 #define ValidVulkanHandle(h) ((h).value[0] != 0) 20 21 #define MaxCommandBuffersInFlight (3) 22 #define MaxCommandBufferTimestamps (1024) 23 24 typedef enum { 25 VulkanQueueKind_Graphics, 26 VulkanQueueKind_Compute, 27 VulkanQueueKind_Transfer, 28 VulkanQueueKind_Count, 29 } VulkanQueueKind; 30 31 typedef enum { 32 VulkanMemoryKind_Device, 33 #if ForceStagingBuffers 34 VulkanMemoryKind_BAR = VulkanMemoryKind_Device, 35 #else 36 VulkanMemoryKind_BAR, 37 #endif 38 VulkanMemoryKind_Host, 39 VulkanMemoryKind_Count, 40 } VulkanMemoryKind; 41 42 typedef struct VulkanEntity VulkanEntity; 43 44 typedef struct { 45 VkDeviceMemory memory; 46 VkBuffer buffer; 47 u64 memory_size; 48 49 void * host_pointer; 50 51 VulkanMemoryKind memory_kind; 52 53 // NOTE: only used when the buffer is backing a VulkanRenderModel. 54 VkIndexType index_type; 55 56 // NOTE(rnp): only valid for buffer that will be written from the CPU and 57 // when the system needs to use a staging buffer rather than BAR access 58 VulkanEntity *next; 59 } VulkanBuffer; 60 61 typedef struct { 62 VkDeviceMemory memory; 63 VkImage image; 64 VkImageView view; 65 } VulkanImage; 66 67 typedef struct { 68 VkPipeline pipeline; 69 VkPipelineLayout layout; 70 VkShaderStageFlags stage_flags; 71 } VulkanPipeline; 72 73 typedef struct { 74 VkSemaphore semaphore; 75 u64 value; 76 } VulkanSemaphore; 77 78 typedef struct { 79 GPUTimeline timeline; 80 u32 buffer_index; 81 // NOTE(rnp): since there may not be QueueKind_Count queues, when putting values into this 82 // array you must be careful to map through the queue_indices array in the vulkan_context. 83 u64 in_flight_wait_values[VulkanQueueKind_Count]; 84 } VulkanCommandBuffer; 85 86 typedef enum { 87 VulkanEntityKind_Buffer, 88 VulkanEntityKind_CommandBuffer, 89 VulkanEntityKind_Image, 90 VulkanEntityKind_Pipeline, 91 VulkanEntityKind_RenderModel, 92 VulkanEntityKind_Semaphore, 93 } VulkanEntityKind; 94 95 struct VulkanEntity { 96 VulkanEntity * next; 97 VulkanEntityKind kind; 98 union { 99 VulkanBuffer buffer; 100 VulkanCommandBuffer command_buffer; 101 VulkanImage image; 102 VulkanPipeline pipeline; 103 VulkanSemaphore semaphore; 104 } as; 105 }; 106 107 typedef alignas(64) struct { 108 i32 lock; 109 110 u16 queue_family; 111 u16 queue_index; 112 VkQueue queue; 113 114 VulkanSemaphore timeline_semaphore; 115 116 VkPipelineStageFlags2 pipeline_stage_flags; 117 } VulkanQueue; 118 static_assert(alignof(VulkanQueue) == 64, "VulkanQueue must be placed on its own cacheline"); 119 120 typedef alignas(64) struct { 121 i32 lock; 122 u32 next_command_buffer_index; 123 124 // NOTE(rnp): small arena for storing temporary data during submission 125 Arena *arena; 126 127 VulkanPipeline *bound_pipeline; 128 129 u64 last_submission_values[MaxCommandBuffersInFlight]; 130 u64 timestamp_counts[MaxCommandBuffersInFlight]; 131 132 VkCommandPool handle; 133 VkQueryPool query_pool; 134 VkCommandBuffer buffers[MaxCommandBuffersInFlight]; 135 } VulkanCommandPool; 136 137 typedef struct { 138 Arena *arena; 139 i32 arena_lock; 140 141 VkInstance handle; 142 VkDevice device; 143 VkPhysicalDevice physical_device; 144 145 // NOTE(rnp): fallback for when a shader fails to compile 146 VulkanPipeline default_compute_pipeline; 147 VulkanPipeline default_graphics_pipeline; 148 149 GPUInfo gpu_info; 150 151 struct { 152 u64 max_allocation_size; 153 u64 non_coherent_atom_size; 154 u64 memory_heap_sizes[VulkanMemoryKind_Count]; 155 u8 gpu_heap_index; 156 i8 memory_type_indices[VulkanMemoryKind_Count]; 157 b8 memory_host_coherent[VulkanMemoryKind_Count]; 158 static_assert(VK_MAX_MEMORY_HEAPS < I8_MAX, ""); 159 static_assert(VK_MAX_MEMORY_TYPES < U8_MAX, ""); 160 } memory_info; 161 162 VulkanCommandPool *command_pools[GPUTimeline_Count]; 163 VulkanQueue *queues[VulkanQueueKind_Count]; 164 // NOTE(rnp): there are a few places in the code where simply going through the queues map 165 // is not sufficient. those places need to know of the unique queues which unique queue 166 // is being referred to. that code uses this map instead. 167 u16 queue_indices[VulkanQueueKind_Count]; 168 u16 unique_queues; 169 170 VkFormat swap_chain_image_format; 171 VkFormat depth_stencil_format; 172 173 174 VulkanEntity *entity_freelist; 175 Arena *entity_arena; 176 i32 entity_lock; 177 } VulkanContext; 178 179 read_only char *vk_required_instance_extensions[] = { 180 }; 181 182 #if OS_WINDOWS 183 #define VK_OS_REQUIRED_DEVICE_EXTENSIONS_LIST \ 184 X("VK_KHR_external_memory_win32") \ 185 X("VK_KHR_external_semaphore_win32") \ 186 187 #else 188 #define VK_OS_REQUIRED_DEVICE_EXTENSIONS_LIST \ 189 X("VK_KHR_external_memory_fd") \ 190 X("VK_KHR_external_semaphore_fd") \ 191 192 #endif 193 194 #define VK_REQUIRED_DEVICE_EXTENSIONS_LIST \ 195 X("VK_KHR_16bit_storage") \ 196 X("VK_KHR_8bit_storage") \ 197 X("VK_KHR_external_memory") \ 198 X("VK_KHR_external_semaphore") \ 199 X("VK_KHR_storage_buffer_storage_class") \ 200 X("VK_KHR_timeline_semaphore") \ 201 VK_OS_REQUIRED_DEVICE_EXTENSIONS_LIST 202 203 #define X(str) str8_comp(str), 204 read_only str8 vk_required_device_extensions[] = {VK_REQUIRED_DEVICE_EXTENSIONS_LIST}; 205 #undef X 206 207 #define VK_OPTIONAL_DEVICE_EXTENSIONS_LIST \ 208 X(VK_KHR, cooperative_matrix) \ 209 210 #define X(p, s, ...) str8_comp(#p "_" #s), 211 read_only str8 vk_optional_device_extensions[] = {VK_OPTIONAL_DEVICE_EXTENSIONS_LIST}; 212 #undef X 213 214 #define VK_REQUIRED_PHYSICAL_FEATURES \ 215 X(shaderInt16) \ 216 X(shaderInt64) \ 217 218 #define VK_REQUIRED_PHYSICAL_11_FEATURES \ 219 X(storageBuffer16BitAccess) \ 220 221 #define VK_REQUIRED_PHYSICAL_12_FEATURES \ 222 X(bufferDeviceAddress) \ 223 X(shaderFloat16) \ 224 X(shaderInt8) \ 225 X(storageBuffer8BitAccess) \ 226 X(timelineSemaphore) \ 227 X(vulkanMemoryModel) \ 228 229 #define VK_REQUIRED_PHYSICAL_13_FEATURES \ 230 X(dynamicRendering) \ 231 X(synchronization2) \ 232 233 #define VK_DEBUG_EXTENSIONS \ 234 X(VK_KHR, shader_non_semantic_info) \ 235 X(VK_KHR, shader_relaxed_extended_instruction) \ 236 237 #define X(p, s, ...) str8_comp(#p "_" #s), 238 read_only str8 vk_debug_extensions[] = {VK_DEBUG_EXTENSIONS}; 239 #undef X 240 241 #define VK_INSTANCE_DEBUG_EXTENSIONS_LIST \ 242 X(VK_EXT, debug_utils) \ 243 244 #define X(p, s, ...) str8_comp(#p "_" #s), 245 read_only str8 vk_instance_debug_extensions[] = {VK_INSTANCE_DEBUG_EXTENSIONS_LIST}; 246 #undef X 247 248 #if VulkanDebug 249 #define VK_VALIDATION_LAYERS_LIST \ 250 X(KHRONOS, validation) \ 251 252 #else 253 #define VK_VALIDATION_LAYERS_LIST 254 #endif 255 256 read_only str8 vk_validation_layers[] = { 257 #define X(vendor, name, ...) str8_comp("VK_LAYER_" #vendor "_" #name), 258 VK_VALIDATION_LAYERS_LIST 259 #undef X 260 }; 261 262 global struct { 263 u32 driver_api_version; 264 union { 265 struct { 266 #define X(_, name, ...) b8 name; 267 VK_OPTIONAL_DEVICE_EXTENSIONS_LIST 268 #undef X 269 }; 270 b8 E[countof(vk_optional_device_extensions)]; 271 } optional; 272 273 union { 274 struct { 275 #define X(_, name, ...) b8 name; 276 VK_DEBUG_EXTENSIONS 277 #undef X 278 }; 279 b8 E[countof(vk_debug_extensions)]; 280 } debug; 281 282 union { 283 struct { 284 #define X(_, name, ...) b8 name; 285 VK_INSTANCE_DEBUG_EXTENSIONS_LIST 286 #undef X 287 }; 288 b8 E[countof(vk_instance_debug_extensions)]; 289 } instance; 290 291 #if VulkanDebug 292 struct { 293 union { 294 struct { 295 #define X(_, name, ...) b8 name; 296 VK_VALIDATION_LAYERS_LIST 297 #undef X 298 }; 299 b8 E[countof(vk_validation_layers)]; 300 } enabled; 301 302 union { 303 struct { 304 #define X(_, name, ...) u32 name; 305 VK_VALIDATION_LAYERS_LIST 306 #undef X 307 }; 308 u32 E[countof(vk_validation_layers)]; 309 } version; 310 } layers; 311 #endif 312 } vulkan_config; 313 314 #define MAX_ENABLED_EXTENSIONS ( countof(vk_required_device_extensions) \ 315 + countof(vk_optional_device_extensions) \ 316 + countof(vk_debug_extensions) \ 317 ) 318 319 global VulkanContext vulkan_context[1]; 320 321 /* NOTE(rnp): the idea here is to set reasonable development constraints. 322 * They should probably not match one to one with the maximums of the dev 323 * machine's hardware. Instead these are here to cause compile time failure 324 * for features which are not expected to work everywhere. */ 325 global glslang_resource_t glslc_resource_constraints[1] = {{ 326 .max_compute_work_group_count_x = 65535, 327 .max_compute_work_group_count_y = 65535, 328 .max_compute_work_group_count_z = 65535, 329 .max_compute_work_group_size_x = 1024, 330 .max_compute_work_group_size_y = 1024, 331 .max_compute_work_group_size_z = 1024, 332 333 // NOTE: taken from glslang defaults 334 .max_lights = 32, 335 .max_clip_planes = 6, 336 .max_texture_units = 32, 337 .max_texture_coords = 32, 338 .max_vertex_attribs = 64, 339 .max_vertex_uniform_components = 4096, 340 .max_varying_floats = 64, 341 .max_vertex_texture_image_units = 32, 342 .max_combined_texture_image_units = 80, 343 .max_texture_image_units = 32, 344 .max_fragment_uniform_components = 4096, 345 .max_draw_buffers = 32, 346 .max_vertex_uniform_vectors = 128, 347 .max_varying_vectors = 8, 348 .max_fragment_uniform_vectors = 16, 349 .max_vertex_output_vectors = 16, 350 .max_fragment_input_vectors = 15, 351 .min_program_texel_offset = -8, 352 .max_program_texel_offset = 7, 353 .max_clip_distances = 8, 354 .max_compute_uniform_components = 1024, 355 .max_compute_texture_image_units = 16, 356 .max_compute_image_uniforms = 8, 357 .max_compute_atomic_counters = 8, 358 .max_compute_atomic_counter_buffers = 1, 359 .max_varying_components = 60, 360 .max_vertex_output_components = 64, 361 .max_fragment_input_components = 128, 362 .max_image_units = 8, 363 .max_combined_image_units_and_fragment_outputs = 8, 364 .max_combined_shader_output_resources = 8, 365 .max_image_samples = 0, 366 .max_vertex_image_uniforms = 0, 367 .max_fragment_image_uniforms = 8, 368 .max_combined_image_uniforms = 8, 369 .max_viewports = 16, 370 .max_vertex_atomic_counters = 0, 371 .max_fragment_atomic_counters = 8, 372 .max_combined_atomic_counters = 8, 373 .max_atomic_counter_bindings = 1, 374 .max_vertex_atomic_counter_buffers = 0, 375 .max_fragment_atomic_counter_buffers = 1, 376 .max_combined_atomic_counter_buffers = 1, 377 .max_atomic_counter_buffer_size = 16384, 378 .max_transform_feedback_buffers = 4, 379 .max_transform_feedback_interleaved_components = 64, 380 .max_cull_distances = 8, 381 .max_combined_clip_and_cull_distances = 8, 382 .max_samples = 4, 383 .max_mesh_output_vertices_ext = 256, 384 .max_mesh_output_primitives_ext = 256, 385 .max_mesh_work_group_size_x_ext = 128, 386 .max_mesh_work_group_size_y_ext = 128, 387 .max_mesh_work_group_size_z_ext = 128, 388 .max_task_work_group_size_x_ext = 128, 389 .max_task_work_group_size_y_ext = 128, 390 .max_task_work_group_size_z_ext = 128, 391 .max_mesh_view_count_ext = 4, 392 .max_dual_source_draw_buffers_ext = 1, 393 394 .limits = { 395 .non_inductive_for_loops = 1, 396 .while_loops = 1, 397 .do_while_loops = 1, 398 .general_uniform_indexing = 1, 399 .general_attribute_matrix_vector_indexing = 1, 400 .general_varying_indexing = 1, 401 .general_sampler_indexing = 1, 402 .general_variable_indexing = 1, 403 .general_constant_matrix_vector_indexing = 1, 404 }, 405 }}; 406 407 #if BEAMFORMER_RENDERDOC_HOOKS 408 DEBUG_IMPORT void * 409 vk_renderdoc_instance_handle(void) 410 { 411 return *((void **)vulkan_context->handle); 412 } 413 #endif 414 415 #if VulkanDebug 416 #define vk_label_object(k, h, label, extra) vk_label_object_(VK_OBJECT_TYPE_##k, (u64)h, label, extra) 417 function void 418 vk_label_object_(VkObjectType kind, u64 handle, str8 label, str8 extra) 419 { 420 local_persist u8 buffer[1024]; 421 Stream sb = stream_from_buffer(buffer, countof(buffer)); 422 if (vulkan_config.instance.debug_utils && label.length > 0) { 423 stream_append_str8s(&sb, label, str8(" ("), extra, str8(")")); 424 stream_append_byte(&sb, 0); 425 if (!sb.errors) { 426 VkDebugUtilsObjectNameInfoEXT object_name_info = { 427 .sType = VK_STRUCTURE_TYPE_DEBUG_UTILS_OBJECT_NAME_INFO_EXT, 428 .objectType = kind, 429 .objectHandle = handle, 430 .pObjectName = (char *)sb.data, 431 }; 432 vkSetDebugUtilsObjectNameEXT(vulkan_context->device, &object_name_info); 433 } 434 } 435 } 436 #else 437 #define vk_label_object(...) 438 #define vk_label_object_(...) 439 #endif 440 441 function VulkanEntity * 442 vk_entity_allocate(VulkanEntityKind kind) 443 { 444 VulkanEntity *result = 0; 445 DeferLoop(take_lock(&vulkan_context->entity_lock, -1), release_lock(&vulkan_context->entity_lock)) 446 { 447 result = SLLPopFreelist(vulkan_context->entity_freelist); 448 if (!result) result = push_struct_no_zero(vulkan_context->entity_arena, VulkanEntity); 449 } 450 451 zero_struct(result); 452 result->kind = kind; 453 return result; 454 } 455 456 function void 457 vk_entity_release(VulkanEntity *entity) 458 { 459 DeferLoop(take_lock(&vulkan_context->entity_lock, -1), release_lock(&vulkan_context->entity_lock)) 460 { 461 SLLStackPush(vulkan_context->entity_freelist, entity, next); 462 } 463 } 464 465 function void * 466 vk_entity_data(u64 handle, VulkanEntityKind kind) 467 { 468 VulkanEntity *e = (VulkanEntity *)handle; 469 assert(handle && e->kind == kind); 470 return &e->as; 471 } 472 473 function VkCommandBuffer 474 vk_command_buffer(GPUCommandList h) 475 { 476 VulkanCommandBuffer *vcb = vk_entity_data(h.value, VulkanEntityKind_CommandBuffer); 477 VulkanCommandPool *vcp = vulkan_context->command_pools[vcb->timeline]; 478 VkCommandBuffer result = vcp->buffers[vcb->buffer_index]; 479 return result; 480 } 481 482 #define glslang_log(a, ...) glslang_log_(a, arg_list(str8, __VA_ARGS__)) 483 function void 484 glslang_log_(Arena *arena, str8 *items, u64 count) 485 { 486 Stream sb = arena_stream(arena); 487 stream_append_str8(&sb, glslang_info("")); 488 stream_append_str8s_(&sb, items, count); 489 if (sb.data[sb.widx - 1] != '\n') stream_append_byte(&sb, '\n'); 490 os_console_log(sb.data, sb.widx); 491 } 492 493 function str8 494 glsl_to_spirv(Arena *arena, u32 kind, str8 shader_text, str8 name) 495 { 496 /* NOTE(rnp): glslang's garbage c interface doesn't expose internal usage of strings with length */ 497 assert(shader_text.data[shader_text.length] == 0); 498 499 glslang_input_t input = { 500 .language = GLSLANG_SOURCE_GLSL, 501 .stage = kind, 502 .client = GLSLANG_CLIENT_VULKAN, 503 .client_version = GLSLANG_TARGET_VULKAN_1_4, 504 .target_language = GLSLANG_TARGET_SPV, 505 .target_language_version = GLSLANG_TARGET_SPV_1_6, 506 .code = (c8 *)shader_text.data, 507 .default_version = 460, 508 .default_profile = GLSLANG_NO_PROFILE, 509 .force_default_version_and_profile = 0, 510 .forward_compatible = 0, 511 .messages = GLSLANG_MSG_DEFAULT_BIT, 512 .resource = glslc_resource_constraints, 513 }; 514 glslang_shader_t *shader = glslang_shader_create(&input); 515 516 str8 error = {0}; 517 if (glslang_shader_preprocess(shader, &input)) { 518 if (!glslang_shader_parse(shader, &input)) 519 error = str8("parsing failed"); 520 } else { 521 error = str8("preprocessing failed"); 522 } 523 524 if (error.length) { 525 glslang_log(arena, name, str8(": "), error, str8("\n"), 526 str8_from_c_str((c8 *)glslang_shader_get_info_log(shader)), 527 str8_from_c_str((c8 *)glslang_shader_get_info_debug_log(shader))); 528 glslang_shader_delete(shader); 529 shader = 0; 530 } 531 532 str8 result = {0}; 533 if (shader) { 534 glslang_program_t *program = glslang_program_create(); 535 glslang_program_add_shader(program, shader); 536 i32 messages = GLSLANG_MSG_DEBUG_INFO_BIT|GLSLANG_MSG_SPV_RULES_BIT|GLSLANG_MSG_VULKAN_RULES_BIT; 537 if (glslang_program_link(program, messages)) { 538 glslang_spv_options_t options = {.validate = 1,}; 539 540 if (vulkan_config.debug.shader_non_semantic_info && 541 vulkan_config.debug.shader_relaxed_extended_instruction) 542 { 543 options.generate_debug_info = 1; 544 options.emit_nonsemantic_shader_debug_info = 1; 545 options.emit_nonsemantic_shader_debug_source = 1; 546 } 547 548 glslang_program_add_source_text(program, kind, (c8 *)shader_text.data, shader_text.length); 549 glslang_program_SPIRV_generate_with_options(program, kind, &options); 550 551 u32 words = glslang_program_SPIRV_get_size(program); 552 result.data = (u8 *)push_array(arena, u32, words); 553 result.length = words * sizeof(u32); 554 glslang_program_SPIRV_get(program, (u32 *)result.data); 555 556 str8 spirv_msg = str8_from_c_str((c8 *)glslang_program_SPIRV_get_messages(program)); 557 if (spirv_msg.length) glslang_log(arena, name, str8(": spirv info: "), spirv_msg); 558 } else { 559 glslang_log(arena, name, str8(": shader linking failed\n"), 560 str8_from_c_str((c8 *)glslang_program_get_info_log(program)), 561 str8_from_c_str((c8 *)glslang_program_get_info_debug_log(program))); 562 } 563 glslang_shader_delete(shader); 564 glslang_program_delete(program); 565 } 566 567 return result; 568 } 569 570 function u32 571 vk_shader_kind_to_glslang_shader_kind(u32 kind) 572 { 573 u32 result = ctz_u64(kind); 574 return result; 575 } 576 577 function VkShaderModule 578 vk_compile_shader_module(Arena *arena, u32 kind, str8 text, str8 name) 579 { 580 VkShaderModule result = {0}; 581 str8 spirv = glsl_to_spirv(arena, vk_shader_kind_to_glslang_shader_kind(kind), text, name); 582 VkShaderModuleCreateInfo create_info = { 583 .sType = VK_STRUCTURE_TYPE_SHADER_MODULE_CREATE_INFO, 584 .codeSize = (u64)spirv.length, 585 .pCode = (u32 *)spirv.data, 586 }; 587 if (spirv.length > 0) vkCreateShaderModule(vulkan_context->device, &create_info, 0, &result); 588 589 return result; 590 } 591 592 function VkShaderStageFlags 593 vk_stage_flags_from_shader_kind(VulkanShaderKind kind) 594 { 595 read_only VkShaderStageFlags map[VulkanShaderKind_Count + 1] = { 596 [VulkanShaderKind_Vertex] = VK_SHADER_STAGE_VERTEX_BIT, 597 [VulkanShaderKind_Mesh] = VK_SHADER_STAGE_MESH_BIT_EXT, 598 [VulkanShaderKind_Fragment] = VK_SHADER_STAGE_FRAGMENT_BIT, 599 [VulkanShaderKind_Compute] = VK_SHADER_STAGE_COMPUTE_BIT, 600 [VulkanShaderKind_Count] = 0, 601 }; 602 VkShaderStageFlags result = map[Clamp((u32)kind, 0, VulkanShaderKind_Count)]; 603 return result; 604 } 605 606 function VkSpecializationMapEntry * 607 vk_specialization_map_from_struct_id(Arena *arena, i32 struct_id) 608 { 609 assert(struct_id >= 0); 610 const MetaStructInfo *si = meta_struct_info_by_id + struct_id; 611 const MetaStructMember *sm = meta_struct_members_by_id[struct_id]; 612 VkSpecializationMapEntry *result = push_array(arena, VkSpecializationMapEntry, si->member_count); 613 for EachIndex(si->member_count, it) { 614 result[it].constantID = it; 615 result[it].offset = sm[it].offset; 616 result[it].size = meta_kind_byte_sizes[sm[it].type_id]; 617 } 618 return result; 619 } 620 621 function VulkanPipeline 622 vk_compute_pipeline_from_info(Arena *arena, VulkanPipelineCreateInfo *info, u32 push_constants_size) 623 { 624 VulkanPipeline result = {.stage_flags = VK_SHADER_STAGE_COMPUTE_BIT}; 625 VkShaderModule module = vk_compile_shader_module(arena, VK_SHADER_STAGE_COMPUTE_BIT, info->text, info->name); 626 if (module) { 627 VkPushConstantRange push_constant_range = { 628 .stageFlags = VK_SHADER_STAGE_COMPUTE_BIT, 629 .offset = 0, 630 .size = push_constants_size, 631 }; 632 633 VkPipelineLayoutCreateInfo pipeline_layout_create_info = { 634 .sType = VK_STRUCTURE_TYPE_PIPELINE_LAYOUT_CREATE_INFO, 635 .pushConstantRangeCount = push_constants_size ? 1 : 0, 636 .pPushConstantRanges = push_constants_size ? &push_constant_range : 0, 637 }; 638 639 vkCreatePipelineLayout(vulkan_context->device, &pipeline_layout_create_info, 0, &result.layout); 640 641 VkComputePipelineCreateInfo pipeline_create_info = { 642 .sType = VK_STRUCTURE_TYPE_COMPUTE_PIPELINE_CREATE_INFO, 643 .layout = result.layout, 644 .stage = { 645 .sType = VK_STRUCTURE_TYPE_PIPELINE_SHADER_STAGE_CREATE_INFO, 646 .stage = VK_SHADER_STAGE_COMPUTE_BIT, 647 .module = module, 648 .pName = "main", 649 }, 650 }; 651 652 VkSpecializationInfo specialization_info = {0}; 653 if (info->specialization_data && info->specialization_struct_id >= 0) { 654 const MetaStructInfo *si = meta_struct_info_by_id + info->specialization_struct_id; 655 pipeline_create_info.stage.pSpecializationInfo = &specialization_info; 656 specialization_info.pMapEntries = vk_specialization_map_from_struct_id(arena, info->specialization_struct_id); 657 specialization_info.mapEntryCount = si->member_count; 658 specialization_info.dataSize = si->size; 659 specialization_info.pData = info->specialization_data; 660 } 661 662 vkCreateComputePipelines(vulkan_context->device, 0, 1, &pipeline_create_info, 0, &result.pipeline); 663 664 vk_label_object(PIPELINE, result.pipeline, info->name, str8("Pipeline")); 665 vk_label_object(PIPELINE_LAYOUT, result.layout, info->name, str8("Pipeline Layout")); 666 vk_label_object(SHADER_MODULE, module, info->name, str8("Module")); 667 668 vkDestroyShaderModule(vulkan_context->device, module, 0); 669 } 670 if (result.pipeline == 0) result = vulkan_context->default_compute_pipeline; 671 672 return result; 673 } 674 675 function VulkanPipeline 676 vk_graphics_pipeline_from_infos(Arena *arena, VulkanPipelineCreateInfo *infos, u32 count, u32 push_constants_size) 677 { 678 assume(count == 2); 679 680 VulkanPipeline result = {0}; 681 VkShaderModule modules[2]; 682 683 modules[0] = vk_compile_shader_module(arena, vk_stage_flags_from_shader_kind(infos[0].kind), 684 infos[0].text, infos[0].name); 685 modules[1] = vk_compile_shader_module(arena, vk_stage_flags_from_shader_kind(infos[1].kind), 686 infos[1].text, infos[1].name); 687 if (modules[0] && modules[1]) { 688 result.stage_flags = vk_stage_flags_from_shader_kind(infos[0].kind) 689 | vk_stage_flags_from_shader_kind(infos[1].kind); 690 691 VkPushConstantRange pcr = { 692 .stageFlags = result.stage_flags, 693 .offset = 0, 694 .size = push_constants_size, 695 }; 696 697 VkPipelineLayoutCreateInfo pipeline_layout_info = { 698 .sType = VK_STRUCTURE_TYPE_PIPELINE_LAYOUT_CREATE_INFO, 699 .pushConstantRangeCount = push_constants_size ? 1 : 0, 700 .pPushConstantRanges = push_constants_size ? &pcr : 0, 701 }; 702 703 vkCreatePipelineLayout(vulkan_context->device, &pipeline_layout_info, 0, &result.layout); 704 705 VkPipelineShaderStageCreateInfo shader_stage_create_infos[2] = { 706 { 707 .sType = VK_STRUCTURE_TYPE_PIPELINE_SHADER_STAGE_CREATE_INFO, 708 .stage = vk_stage_flags_from_shader_kind(infos[0].kind), 709 .module = modules[0], 710 .pName = "main", 711 }, 712 { 713 .sType = VK_STRUCTURE_TYPE_PIPELINE_SHADER_STAGE_CREATE_INFO, 714 .stage = vk_stage_flags_from_shader_kind(infos[1].kind), 715 .module = modules[1], 716 .pName = "main", 717 }, 718 }; 719 720 VkPipelineVertexInputStateCreateInfo vertex_input_info = { 721 .sType = VK_STRUCTURE_TYPE_PIPELINE_VERTEX_INPUT_STATE_CREATE_INFO, 722 }; 723 724 VkPipelineInputAssemblyStateCreateInfo input_assembly_info = { 725 .sType = VK_STRUCTURE_TYPE_PIPELINE_INPUT_ASSEMBLY_STATE_CREATE_INFO, 726 .topology = VK_PRIMITIVE_TOPOLOGY_TRIANGLE_LIST, 727 }; 728 729 VkPipelineViewportStateCreateInfo viewport_info = { 730 .sType = VK_STRUCTURE_TYPE_PIPELINE_VIEWPORT_STATE_CREATE_INFO, 731 .viewportCount = 1, 732 .scissorCount = 1, 733 }; 734 735 VkPipelineRasterizationStateCreateInfo rasterization_info = { 736 .sType = VK_STRUCTURE_TYPE_PIPELINE_RASTERIZATION_STATE_CREATE_INFO, 737 .polygonMode = VK_POLYGON_MODE_FILL, 738 .lineWidth = 1.0f, 739 .cullMode = VK_CULL_MODE_BACK_BIT, 740 .frontFace = VK_FRONT_FACE_CLOCKWISE, 741 }; 742 743 VkPipelineMultisampleStateCreateInfo multisampling_info = { 744 .sType = VK_STRUCTURE_TYPE_PIPELINE_MULTISAMPLE_STATE_CREATE_INFO, 745 .rasterizationSamples = vulkan_context->gpu_info.max_msaa_samples, 746 }; 747 748 VkPipelineDepthStencilStateCreateInfo depth_test_create_info = { 749 .sType = VK_STRUCTURE_TYPE_PIPELINE_DEPTH_STENCIL_STATE_CREATE_INFO, 750 .depthTestEnable = 1, 751 .depthWriteEnable = 1, 752 .depthCompareOp = VK_COMPARE_OP_LESS, 753 .depthBoundsTestEnable = 1, 754 .stencilTestEnable = 0, 755 .front = {0}, 756 .back = {0}, 757 .minDepthBounds = 0.0f, 758 .maxDepthBounds = 1.0f, 759 }; 760 761 u32 colour_mask = VK_COLOR_COMPONENT_R_BIT|VK_COLOR_COMPONENT_G_BIT|VK_COLOR_COMPONENT_B_BIT|VK_COLOR_COMPONENT_A_BIT; 762 VkPipelineColorBlendAttachmentState blend_state = { 763 .colorWriteMask = colour_mask, 764 .blendEnable = 1, 765 .srcColorBlendFactor = VK_BLEND_FACTOR_SRC_ALPHA, 766 .dstColorBlendFactor = VK_BLEND_FACTOR_ONE_MINUS_SRC_ALPHA, 767 .colorBlendOp = VK_BLEND_OP_ADD, 768 .srcAlphaBlendFactor = VK_BLEND_FACTOR_ONE, 769 .dstAlphaBlendFactor = VK_BLEND_FACTOR_ZERO, 770 .alphaBlendOp = VK_BLEND_OP_ADD, 771 }; 772 773 VkPipelineColorBlendStateCreateInfo colour_blend_state_create = { 774 .sType = VK_STRUCTURE_TYPE_PIPELINE_COLOR_BLEND_STATE_CREATE_INFO, 775 .logicOpEnable = 0, 776 .logicOp = VK_LOGIC_OP_COPY, 777 .attachmentCount = 1, 778 .pAttachments = &blend_state, 779 }; 780 781 VkDynamicState dynamic_states[] = { 782 VK_DYNAMIC_STATE_VIEWPORT, 783 VK_DYNAMIC_STATE_SCISSOR, 784 }; 785 786 VkPipelineDynamicStateCreateInfo dynamic_state_info = { 787 .sType = VK_STRUCTURE_TYPE_PIPELINE_DYNAMIC_STATE_CREATE_INFO, 788 .dynamicStateCount = countof(dynamic_states), 789 .pDynamicStates = dynamic_states, 790 }; 791 792 //VkFormat colour_attachment_format = VK_FORMAT_R8G8B8A8_SRGB; 793 VkFormat colour_attachment_format = VK_FORMAT_R8G8B8A8_UNORM; 794 VkPipelineRenderingCreateInfo rendering_create_info = { 795 .sType = VK_STRUCTURE_TYPE_PIPELINE_RENDERING_CREATE_INFO, 796 .colorAttachmentCount = 1, 797 .pColorAttachmentFormats = &colour_attachment_format, 798 .depthAttachmentFormat = vulkan_context->depth_stencil_format, 799 .stencilAttachmentFormat = vulkan_context->depth_stencil_format, 800 }; 801 802 VkGraphicsPipelineCreateInfo pci = { 803 .sType = VK_STRUCTURE_TYPE_GRAPHICS_PIPELINE_CREATE_INFO, 804 .pNext = &rendering_create_info, 805 .stageCount = countof(shader_stage_create_infos), 806 .pStages = shader_stage_create_infos, 807 .pVertexInputState = &vertex_input_info, 808 .pInputAssemblyState = &input_assembly_info, 809 .pViewportState = &viewport_info, 810 .pRasterizationState = &rasterization_info, 811 .pMultisampleState = &multisampling_info, 812 .pDepthStencilState = &depth_test_create_info, 813 .pColorBlendState = &colour_blend_state_create, 814 .pDynamicState = &dynamic_state_info, 815 .layout = result.layout, 816 }; 817 818 vkCreateGraphicsPipelines(vulkan_context->device, 0, 1, &pci,0, &result.pipeline); 819 820 str8 extras[] = { 821 [VulkanShaderKind_Vertex] = str8_comp("Vertex Module"), 822 [VulkanShaderKind_Mesh] = str8_comp("Mesh Module"), 823 [VulkanShaderKind_Fragment] = str8_comp("Fragment Module"), 824 }; 825 assert(infos[0].kind < countof(extras)); 826 assert(infos[1].kind < countof(extras)); 827 828 vk_label_object(PIPELINE, result.pipeline, infos[0].name, str8("Pipeline")); 829 vk_label_object(PIPELINE_LAYOUT, result.layout, infos[0].name, str8("Pipeline Layout")); 830 //vk_label_object_(VK_OBJECT_TYPE_SHADER_MODULE, (u64)modules[0], infos[0].name, extras[infos[0].kind]); 831 //vk_label_object_(VK_OBJECT_TYPE_SHADER_MODULE, (u64)modules[1], infos[1].name, extras[infos[1].kind]); 832 } 833 834 if (modules[0]) vkDestroyShaderModule(vulkan_context->device, modules[0], 0); 835 if (modules[1]) vkDestroyShaderModule(vulkan_context->device, modules[1], 0); 836 837 if (result.pipeline == 0) result = vulkan_context->default_graphics_pipeline; 838 839 return result; 840 } 841 842 function VulkanSemaphore 843 vk_make_semaphore(OSHandle *export) 844 { 845 VulkanContext *vk = vulkan_context; 846 847 VkSemaphoreCreateInfo sci = {.sType = VK_STRUCTURE_TYPE_SEMAPHORE_CREATE_INFO}; 848 VkExportSemaphoreCreateInfo esci = { 849 .sType = VK_STRUCTURE_TYPE_EXPORT_SEMAPHORE_CREATE_INFO, 850 .handleTypes = OS_WINDOWS ? VK_EXTERNAL_SEMAPHORE_HANDLE_TYPE_OPAQUE_WIN32_BIT 851 : VK_EXTERNAL_SEMAPHORE_HANDLE_TYPE_OPAQUE_FD_BIT, 852 }; 853 VkSemaphoreTypeCreateInfo stc = { 854 .sType = VK_STRUCTURE_TYPE_SEMAPHORE_TYPE_CREATE_INFO, 855 .semaphoreType = VK_SEMAPHORE_TYPE_TIMELINE, 856 }; 857 858 if (export) sci.pNext = &esci; 859 else sci.pNext = &stc; 860 861 VulkanSemaphore result = {0}; 862 863 vkCreateSemaphore(vk->device, &sci, 0, &result.semaphore); 864 865 if (export) { 866 if (OS_WINDOWS) { 867 VkSemaphoreGetWin32HandleInfoKHR ghi = { 868 .sType = VK_STRUCTURE_TYPE_SEMAPHORE_GET_WIN32_HANDLE_INFO_KHR, 869 .handleType = VK_EXTERNAL_SEMAPHORE_HANDLE_TYPE_OPAQUE_WIN32_BIT, 870 .semaphore = result.semaphore, 871 }; 872 void *handle; 873 vkGetSemaphoreWin32HandleKHR(vk->device, &ghi, &handle); 874 export->value[0] = (u64)handle; 875 } else { 876 VkSemaphoreGetFdInfoKHR ghi = { 877 .sType = VK_STRUCTURE_TYPE_SEMAPHORE_GET_FD_INFO_KHR, 878 .handleType = VK_EXTERNAL_SEMAPHORE_HANDLE_TYPE_OPAQUE_FD_BIT, 879 .semaphore = result.semaphore, 880 }; 881 i32 handle; 882 vkGetSemaphoreFdKHR(vk->device, &ghi, &handle); 883 export->value[0] = (u64)handle; 884 } 885 } 886 887 return result; 888 } 889 890 function void 891 vk_release_memory(VkDeviceMemory memory, u64 size) 892 { 893 VulkanContext *vk = vulkan_context; 894 vkFreeMemory(vk->device, memory, 0); 895 atomic_add_u64(&vk->gpu_info.gpu_heap_used, -size); 896 } 897 898 function b32 899 vk_allocate_memory(VkDeviceMemory *memory, u64 size, VulkanMemoryKind kind, VkMemoryAllocateFlags flags, 900 VkMemoryDedicatedAllocateInfo *dedicated_allocate_info, OSHandle *export) 901 { 902 VulkanContext *vk = vulkan_context; 903 904 VkExportMemoryAllocateInfo export_info = { 905 .sType = VK_STRUCTURE_TYPE_EXPORT_MEMORY_ALLOCATE_INFO, 906 .handleTypes = OS_WINDOWS ? VK_EXTERNAL_MEMORY_HANDLE_TYPE_OPAQUE_WIN32_BIT 907 : VK_EXTERNAL_MEMORY_HANDLE_TYPE_OPAQUE_FD_BIT, 908 }; 909 910 VkMemoryAllocateFlagsInfo memory_allocate_flags_info = { 911 .sType = VK_STRUCTURE_TYPE_MEMORY_ALLOCATE_FLAGS_INFO, 912 .flags = flags, 913 .pNext = dedicated_allocate_info, 914 }; 915 916 if (export) { 917 export_info.pNext = dedicated_allocate_info; 918 memory_allocate_flags_info.pNext = &export_info; 919 } 920 921 VkMemoryAllocateInfo memory_allocate_info = { 922 .sType = VK_STRUCTURE_TYPE_MEMORY_ALLOCATE_INFO, 923 .allocationSize = size, 924 .memoryTypeIndex = vk->memory_info.memory_type_indices[kind], 925 .pNext = &memory_allocate_flags_info, 926 }; 927 928 b32 result = size <= vk->memory_info.memory_heap_sizes[kind] && 929 vkAllocateMemory(vk->device, &memory_allocate_info, 0, memory) == VK_SUCCESS; 930 if (result) { 931 atomic_add_u64(&vk->gpu_info.gpu_heap_used, memory_allocate_info.allocationSize); 932 933 if (export) { 934 if (OS_WINDOWS) { 935 VkMemoryGetWin32HandleInfoKHR handle_info = { 936 .sType = VK_STRUCTURE_TYPE_MEMORY_GET_WIN32_HANDLE_INFO_KHR, 937 .memory = *memory, 938 .handleType = VK_EXTERNAL_MEMORY_HANDLE_TYPE_OPAQUE_WIN32_BIT, 939 }; 940 void *handle; 941 vkGetMemoryWin32HandleKHR(vk->device, &handle_info, &handle); 942 export->value[0] = (u64)handle; 943 } else { 944 VkMemoryGetFdInfoKHR fd_info = { 945 .sType = VK_STRUCTURE_TYPE_MEMORY_GET_FD_INFO_KHR, 946 .memory = *memory, 947 .handleType = VK_EXTERNAL_MEMORY_HANDLE_TYPE_OPAQUE_FD_BIT, 948 }; 949 i32 fd; 950 vkGetMemoryFdKHR(vk->device, &fd_info, &fd); 951 export->value[0] = (u64)fd; 952 } 953 } 954 } 955 return result; 956 } 957 958 function u32 959 vk_index_size(VkIndexType type) 960 { 961 u32 result = 0; 962 switch (type) { 963 case VK_INDEX_TYPE_UINT16:{ result = 2; }break; 964 case VK_INDEX_TYPE_UINT32:{ result = 4; }break; 965 InvalidDefaultCase; 966 } 967 return result; 968 } 969 970 typedef struct { 971 GPUBuffer *gpu_buffer; 972 u64 size; 973 u64 single_transfer_size; 974 GPUUsageFlags flags; 975 u32 queue_family_count; 976 u32 queue_family_indices[GPUTimeline_Count]; 977 VkIndexType index_type; 978 OSHandle *export; 979 str8 label; 980 } VulkanBufferAllocateInfo; 981 982 function b32 983 vk_buffer_allocate_common_base(VulkanBuffer *vb, VulkanBufferAllocateInfo *ai, VulkanMemoryKind memory_kind) 984 { 985 VulkanContext *vk = vulkan_context; 986 987 // TODO(rnp): this probably should be handled, its usually 4GB. likely 988 // need to chain multiple allocations and handle it in shader code 989 u64 clamp_size = vk->memory_info.max_allocation_size & ~(vk->memory_info.non_coherent_atom_size - 1); 990 991 // NOTE(rnp): renderdoc can't handle buffers that are too close to the allocation size limit 992 if (renderdoc_attached()) 993 clamp_size -= MB(8); 994 995 u64 size = Min(ai->size, clamp_size); 996 997 VkBufferCreateInfo buffer_create_info = { 998 .sType = VK_STRUCTURE_TYPE_BUFFER_CREATE_INFO, 999 .usage = VK_BUFFER_USAGE_STORAGE_BUFFER_BIT, 1000 .size = size, 1001 .queueFamilyIndexCount = ai->queue_family_count, 1002 .pQueueFamilyIndices = ai->queue_family_indices, 1003 .sharingMode = ai->queue_family_count > 1 ? VK_SHARING_MODE_CONCURRENT 1004 : VK_SHARING_MODE_EXCLUSIVE, 1005 }; 1006 1007 if (ai->gpu_buffer) 1008 buffer_create_info.usage |= VK_BUFFER_USAGE_SHADER_DEVICE_ADDRESS_BIT; 1009 1010 if (ai->flags & (GPUUsageFlag_TransferSource|GPUUsageFlag_HostRead)) 1011 buffer_create_info.usage |= VK_BUFFER_USAGE_TRANSFER_SRC_BIT; 1012 1013 if (ai->flags & (GPUUsageFlag_TransferDestination|GPUUsageFlag_HostWrite)) 1014 buffer_create_info.usage |= VK_BUFFER_USAGE_TRANSFER_DST_BIT; 1015 1016 if (ai->index_type != VK_INDEX_TYPE_NONE_KHR) 1017 buffer_create_info.usage |= VK_BUFFER_USAGE_INDEX_BUFFER_BIT; 1018 1019 VkExternalMemoryBufferCreateInfo external_memory_buffer_create_info = { 1020 .sType = VK_STRUCTURE_TYPE_EXTERNAL_MEMORY_BUFFER_CREATE_INFO, 1021 .handleTypes = OS_WINDOWS ? VK_EXTERNAL_MEMORY_HANDLE_TYPE_OPAQUE_WIN32_BIT 1022 : VK_EXTERNAL_MEMORY_HANDLE_TYPE_OPAQUE_FD_BIT, 1023 }; 1024 if (ai->export) buffer_create_info.pNext = &external_memory_buffer_create_info; 1025 1026 vkCreateBuffer(vk->device, &buffer_create_info, 0, &vb->buffer); 1027 vk_label_object(BUFFER, vb->buffer, ai->label, str8("Buffer")); 1028 1029 VkMemoryRequirements memory_requirements; 1030 vkGetBufferMemoryRequirements(vk->device, vb->buffer, &memory_requirements); 1031 1032 assert((u64)size <= memory_requirements.size); 1033 size = memory_requirements.size; 1034 1035 VkMemoryDedicatedAllocateInfo dedicated_allocate_info = { 1036 .sType = VK_STRUCTURE_TYPE_MEMORY_DEDICATED_ALLOCATE_INFO, 1037 .buffer = vb->buffer, 1038 }; 1039 1040 b32 result = vk_allocate_memory(&vb->memory, size, memory_kind, ai->gpu_buffer ? VK_MEMORY_ALLOCATE_DEVICE_ADDRESS_BIT : 0, 1041 &dedicated_allocate_info, ai->export); 1042 if (result) { 1043 vk_label_object(DEVICE_MEMORY, vb->memory, ai->label, str8("Memory")); 1044 1045 vb->memory_size = size; 1046 vb->memory_kind = memory_kind; 1047 vb->index_type = ai->index_type; 1048 1049 if (ai->flags & GPUUsageFlag_HostReadWrite) 1050 vkMapMemory(vk->device, vb->memory, 0, vb->memory_size, 0, &vb->host_pointer); 1051 vkBindBufferMemory(vk->device, vb->buffer, vb->memory, 0); 1052 1053 if (ai->gpu_buffer) { 1054 VkBufferDeviceAddressInfo buffer_device_address_info = { 1055 .sType = VK_STRUCTURE_TYPE_BUFFER_DEVICE_ADDRESS_INFO, 1056 .buffer = vb->buffer, 1057 }; 1058 ai->gpu_buffer->gpu_pointer = vkGetBufferDeviceAddress(vk->device, &buffer_device_address_info); 1059 ai->gpu_buffer->size = size; 1060 } 1061 } else { 1062 vkDestroyBuffer(vk->device, vb->buffer, 0); 1063 vb->buffer = 0; 1064 } 1065 return result; 1066 } 1067 1068 function b32 1069 vk_buffer_allocate_common(VulkanBuffer *vb, VulkanBufferAllocateInfo *ai) 1070 { 1071 VulkanContext *vk = vulkan_context; 1072 1073 /* NOTE(rnp): to create a CPU writable buffer: 1074 * 1. try to allocate and map the entire buffer 1075 * - this may fail if the buffer is bigger than the BAR size 1076 * (unknowable from vulkan), or the memory space has become 1077 * too fragmented (unlikely) 1078 * 2. if allocation or mapping fails we must chain a host buffer 1079 * for staging. If this happens in practice we should add 1080 * the ability to import an existing external allocation 1081 */ 1082 u32 host_rw_flags = (ai->flags & GPUUsageFlag_HostReadWrite); 1083 1084 assert(!host_rw_flags || (host_rw_flags && ai->queue_family_count > 0)); 1085 1086 b32 result = 0; 1087 if ((ForceStagingBuffers && !host_rw_flags) || !ForceStagingBuffers) 1088 result = vk_buffer_allocate_common_base(vb, ai, host_rw_flags ? VulkanMemoryKind_BAR : VulkanMemoryKind_Device); 1089 1090 if (!result && host_rw_flags) { 1091 u32 transfer_queue_family = vk->queues[VulkanQueueKind_Transfer]->queue_family; 1092 1093 ai->flags &= ~GPUUsageFlag_HostReadWrite; 1094 ai->flags |= GPUUsageFlag_TransferDestination; 1095 1096 b32 found = 0; 1097 for EachElement(ai->queue_family_indices, it) { 1098 if (ai->queue_family_indices[it] == transfer_queue_family) { 1099 found = 1; 1100 break; 1101 } 1102 } 1103 if (!found) ai->queue_family_indices[ai->queue_family_count++] = transfer_queue_family; 1104 1105 if (vk_buffer_allocate_common_base(vb, ai, VulkanMemoryKind_Device)) { 1106 VulkanEntity *e = vk_entity_allocate(VulkanEntityKind_Buffer); 1107 VulkanBuffer *vsb = &e->as.buffer; 1108 1109 u64 host_size = ai->single_transfer_size > 0 ? ai->single_transfer_size : vb->memory_size; 1110 VulkanBufferAllocateInfo asi = { 1111 .size = AlignUpPowerOfTwo(host_size, vk->memory_info.non_coherent_atom_size), 1112 .index_type = VK_INDEX_TYPE_NONE_KHR, 1113 .flags = host_rw_flags|GPUUsageFlag_TransferSource, 1114 .queue_family_count = 1, 1115 .queue_family_indices[0] = vk->queues[VulkanQueueKind_Transfer]->queue_family, 1116 }; 1117 Temp scratch; 1118 DeferLoop(take_lock(&vk->arena_lock, -1), release_lock(&vk->arena_lock)) 1119 DeferLoop(scratch = temp_begin(vk->arena), temp_end(scratch)) 1120 { 1121 asi.label = push_str8_from_parts(vk->arena, str8("_"), ai->label, str8("Staging")); 1122 result = vk_buffer_allocate_common_base(vsb, &asi, VulkanMemoryKind_Host); 1123 } 1124 1125 if (result) vb->next = e; 1126 else vk_entity_release(e); 1127 } 1128 } 1129 return result; 1130 } 1131 1132 function void 1133 vk_load_instance(Arena *arena, Stream *err) 1134 { 1135 Temp scratch = temp_begin(arena); 1136 #define X(name, ...) name = (name##_fn *)vkGetInstanceProcAddr(0, #name); 1137 VkBaseProcedureList 1138 #undef X 1139 1140 u32 enabled_validation_layers_count = 0; 1141 const char *enabled_validation_layers[countof(vk_validation_layers)]; 1142 1143 u32 enabled_instance_extensions_count = 0; 1144 const char *enabled_instance_extensions[countof(vk_required_instance_extensions) + countof(vk_instance_debug_extensions)]; 1145 1146 static_assert(countof(vk_required_instance_extensions) == 0, ""); 1147 //for EachElement(vk_required_instance_extensions, it) 1148 // enabled_instance_extensions[enabled_instance_extensions_count++] = vk_required_instance_extensions[it]; 1149 1150 #if VulkanDebug 1151 { 1152 u32 layer_count = 0; 1153 vkEnumerateInstanceLayerProperties(&layer_count, 0); 1154 1155 VkLayerProperties *layers = push_array(arena, VkLayerProperties, layer_count); 1156 str8 *layer_str8s = push_array(arena, str8, layer_count); 1157 vkEnumerateInstanceLayerProperties(&layer_count, layers); 1158 1159 for (u32 i = 0; i < layer_count; i++) 1160 layer_str8s[i] = str8_from_c_str(layers[i].layerName); 1161 1162 for EachElement(vk_validation_layers, it) { 1163 for(u32 i = 0; i < layer_count; i++) { 1164 if (str8_equal(vk_validation_layers[it], layer_str8s[i])) { 1165 u32 index = enabled_validation_layers_count++; 1166 enabled_validation_layers[index] = (char *)vk_validation_layers[it].data; 1167 vulkan_config.layers.enabled.E[it] = 1; 1168 vulkan_config.layers.version.E[it] = layers[i].specVersion; 1169 break; 1170 } 1171 } 1172 } 1173 1174 if (countof(vk_validation_layers) != enabled_validation_layers_count) { 1175 i32 missing_count = countof(vk_validation_layers) - enabled_validation_layers_count; 1176 stream_append_str8s(err, vulkan_info("missing validation layer"), 1177 missing_count > 1 ? str8("s:") : str8(":"), str8("\n")); 1178 1179 for EachElement(vk_validation_layers, it) 1180 if (vulkan_config.layers.enabled.E[it] == 0) 1181 stream_append_str8s(err, str8(" "), vk_validation_layers[it], str8("\n")); 1182 } 1183 1184 u32 instance_extension_count = 0; 1185 vkEnumerateInstanceExtensionProperties(0, &instance_extension_count, 0); 1186 1187 VkExtensionProperties *instance_extensions = push_array(arena, VkExtensionProperties, instance_extension_count); 1188 str8 *instance_ext_str8s = push_array(arena, str8, instance_extension_count); 1189 vkEnumerateInstanceExtensionProperties(0, &instance_extension_count, instance_extensions); 1190 for EachIndex(instance_extension_count, it) 1191 instance_ext_str8s[it] = str8_from_c_str(instance_extensions[it].extensionName); 1192 1193 for EachElement(vk_instance_debug_extensions, it) { 1194 for EachIndex(instance_extension_count, i) { 1195 if (str8_equal(vk_instance_debug_extensions[it], instance_ext_str8s[i])) { 1196 u32 index = enabled_instance_extensions_count++; 1197 enabled_instance_extensions[index] = (char *)vk_instance_debug_extensions[it].data; 1198 vulkan_config.instance.E[it] = 1; 1199 break; 1200 } 1201 } 1202 } 1203 } 1204 #endif 1205 1206 VkApplicationInfo app_info = { 1207 .sType = VK_STRUCTURE_TYPE_APPLICATION_INFO, 1208 .pApplicationName = BEAMFORMER_NAME_STRING, 1209 .applicationVersion = 0, 1210 .pEngineName = "No Engine", 1211 .engineVersion = 0, 1212 .apiVersion = VK_MAKE_API_VERSION(1, 3, 0, 0), 1213 }; 1214 1215 VkInstanceCreateInfo instance_create_info = { 1216 .sType = VK_STRUCTURE_TYPE_INSTANCE_CREATE_INFO, 1217 .pApplicationInfo = &app_info, 1218 .ppEnabledExtensionNames = enabled_instance_extensions, 1219 .enabledExtensionCount = enabled_instance_extensions_count, 1220 .ppEnabledLayerNames = enabled_validation_layers, 1221 .enabledLayerCount = enabled_validation_layers_count, 1222 }; 1223 1224 #if 0 && VulkanDebug 1225 VkValidationFeatureEnableEXT validation_feature_enables[] = { 1226 VK_VALIDATION_FEATURE_ENABLE_GPU_ASSISTED_EXT, 1227 VK_VALIDATION_FEATURE_ENABLE_BEST_PRACTICES_EXT, 1228 VK_VALIDATION_FEATURE_ENABLE_DEBUG_PRINTF_EXT, 1229 VK_VALIDATION_FEATURE_ENABLE_SYNCHRONIZATION_VALIDATION_EXT, 1230 }; 1231 1232 VkValidationFeaturesEXT validation_features = { 1233 .sType = VK_STRUCTURE_TYPE_VALIDATION_FEATURES_EXT, 1234 .enabledValidationFeatureCount = countof(validation_feature_enables), 1235 .pEnabledValidationFeatures = validation_feature_enables, 1236 }; 1237 1238 instance_create_info.pNext = &validation_features; 1239 #endif 1240 1241 vkCreateInstance(&instance_create_info, 0, &vulkan_context->handle); 1242 1243 #define X(name, ...) name = (name##_fn *)vkGetInstanceProcAddr(vulkan_context->handle, #name); 1244 VkInstanceProcedureList 1245 #undef X 1246 temp_end(scratch); 1247 } 1248 1249 function void 1250 vk_load_physical_device(Arena *arena, Stream *err) 1251 { 1252 Temp scratch = temp_begin(arena); 1253 VulkanContext *vk = vulkan_context; 1254 1255 u32 device_count; 1256 vkEnumeratePhysicalDevices(vk->handle, &device_count, 0); 1257 1258 VkPhysicalDevice *devices = push_array(arena, typeof(*devices), device_count); 1259 vkEnumeratePhysicalDevices(vk->handle, &device_count, devices); 1260 1261 i32 best_index = -1, best_score = -1; 1262 for (u32 i = 0; i < device_count; i++) { 1263 VkPhysicalDeviceProperties2 *dp = push_struct(arena, typeof(*dp)); 1264 dp->sType = VK_STRUCTURE_TYPE_PHYSICAL_DEVICE_PROPERTIES_2; 1265 vkGetPhysicalDeviceProperties2(devices[i], dp); 1266 1267 i32 score = 0; 1268 if (dp->properties.deviceType == VK_PHYSICAL_DEVICE_TYPE_DISCRETE_GPU) 1269 score++; 1270 1271 if (score > best_score) { 1272 best_score = score; 1273 best_index = (i32)i; 1274 } 1275 } 1276 1277 vk->physical_device = best_index >= 0 ? devices[best_index] : 0; 1278 if (!vk->physical_device) 1279 fatal(vulkan_info("failed to find a suitable GPU\n")); 1280 1281 VkPhysicalDeviceProperties2 dp = {.sType = VK_STRUCTURE_TYPE_PHYSICAL_DEVICE_PROPERTIES_2}; 1282 VkPhysicalDeviceVulkan11Properties v11p = {.sType = VK_STRUCTURE_TYPE_PHYSICAL_DEVICE_VULKAN_1_1_PROPERTIES}; 1283 dp.pNext = &v11p; 1284 1285 vkGetPhysicalDeviceProperties2(vk->physical_device, &dp); 1286 1287 stream_append_str8s(err, vulkan_info("selecting device: "), str8_from_c_str(dp.properties.deviceName), str8("\n")); 1288 stream_append_str8(err, vulkan_info("Vulkan Version: ")); 1289 { 1290 u32 dv = dp.properties.apiVersion; 1291 stream_appendf(err, "%u.%u.%u\n", VK_API_VERSION_MAJOR(dv), VK_API_VERSION_MINOR(dv), VK_API_VERSION_PATCH(dv)); 1292 } 1293 1294 { 1295 u32 extension_count = 0; 1296 vkEnumerateDeviceExtensionProperties(vk->physical_device, 0, &extension_count, 0); 1297 VkExtensionProperties *extensions = push_array(arena, VkExtensionProperties, extension_count); 1298 vkEnumerateDeviceExtensionProperties(vk->physical_device, 0, &extension_count, extensions); 1299 1300 str8 *ext_str8s = push_array(arena, str8, extension_count); 1301 for (u32 index = 0; index < extension_count; index++) 1302 ext_str8s[index] = str8_from_c_str(extensions[index].extensionName); 1303 1304 b8 *supported = push_array(arena, b8, countof(vk_required_device_extensions)); 1305 for EachIndex(extension_count, index) 1306 for EachElement(vk_required_device_extensions, it) 1307 supported[it] |= str8_equal(vk_required_device_extensions[it], ext_str8s[index]); 1308 1309 u32 supported_count = 0; 1310 for EachElement(vk_required_device_extensions, it) 1311 supported_count += supported[it]; 1312 1313 u32 missing_count = countof(vk_required_device_extensions) - supported_count; 1314 if (missing_count) { 1315 stream_append_str8s(err, vulkan_info("fatal error: missing required device extension"), 1316 missing_count > 1 ? str8("s") : str8(""), str8(":\n")); 1317 for EachElement(vk_required_device_extensions, it) { 1318 if (!supported[it]) { 1319 str8 name = vk_required_device_extensions[it]; 1320 stream_append_str8s(err, vulkan_info(" "), name, str8("\n")); 1321 } 1322 } 1323 fatal(stream_to_str8(err)); 1324 } 1325 1326 for EachIndex(extension_count, index) 1327 for EachElement(vk_optional_device_extensions, it) 1328 vulkan_config.optional.E[it] |= str8_equal(vk_optional_device_extensions[it], ext_str8s[index]); 1329 1330 #if VulkanDebug 1331 for EachIndex(extension_count, index) 1332 for EachElement(vk_debug_extensions, it) 1333 vulkan_config.debug.E[it] |= str8_equal(vk_debug_extensions[it], ext_str8s[index]); 1334 #endif 1335 } 1336 1337 { 1338 VkPhysicalDeviceFeatures2 df = {.sType = VK_STRUCTURE_TYPE_PHYSICAL_DEVICE_FEATURES_2}; 1339 VkPhysicalDeviceVulkan11Features v11f = {.sType = VK_STRUCTURE_TYPE_PHYSICAL_DEVICE_VULKAN_1_1_FEATURES}; 1340 VkPhysicalDeviceVulkan12Features v12f = {.sType = VK_STRUCTURE_TYPE_PHYSICAL_DEVICE_VULKAN_1_2_FEATURES}; 1341 VkPhysicalDeviceVulkan13Features v13f = {.sType = VK_STRUCTURE_TYPE_PHYSICAL_DEVICE_VULKAN_1_3_FEATURES}; 1342 df.pNext = &v11f; 1343 v11f.pNext = &v12f; 1344 v12f.pNext = &v13f; 1345 vkGetPhysicalDeviceFeatures2(vk->physical_device, &df); 1346 1347 { 1348 b32 all_supported = 1; 1349 #define X(name, ...) all_supported &= df.features.name; 1350 VK_REQUIRED_PHYSICAL_FEATURES 1351 #undef X 1352 1353 if (!all_supported) { 1354 stream_append_str8(err, vulkan_info("fatal error: missing physical device features:\n")); 1355 #define X(name, ...) if (!df.features.name) stream_append_str8(err, str8(" " #name "\n")); 1356 VK_REQUIRED_PHYSICAL_FEATURES 1357 #undef X 1358 fatal(stream_to_str8(err)); 1359 } 1360 } 1361 1362 { 1363 b32 all_supported = 1; 1364 #define X(name, ...) all_supported &= v11f.name; 1365 VK_REQUIRED_PHYSICAL_11_FEATURES 1366 #undef X 1367 1368 if (!all_supported) { 1369 stream_append_str8(err, vulkan_info("fatal error: missing physical device features:\n")); 1370 #define X(name, ...) if (!v11f.name) stream_append_str8(err, str8(" " #name "\n")); 1371 VK_REQUIRED_PHYSICAL_11_FEATURES 1372 #undef X 1373 fatal(stream_to_str8(err)); 1374 } 1375 } 1376 1377 { 1378 b32 all_supported = 1; 1379 #define X(name, ...) all_supported &= v12f.name; 1380 VK_REQUIRED_PHYSICAL_12_FEATURES 1381 #undef X 1382 1383 if (!all_supported) { 1384 stream_append_str8(err, vulkan_info("fatal error: missing physical device features:\n")); 1385 #define X(name, ...) if (!v12f.name) stream_append_str8(err, str8(" " #name "\n")); 1386 VK_REQUIRED_PHYSICAL_12_FEATURES 1387 #undef X 1388 fatal(stream_to_str8(err)); 1389 } 1390 } 1391 1392 { 1393 b32 all_supported = 1; 1394 #define X(name, ...) all_supported &= v13f.name; 1395 VK_REQUIRED_PHYSICAL_13_FEATURES 1396 #undef X 1397 1398 if (!all_supported) { 1399 stream_append_str8(err, vulkan_info("fatal error: missing physical device features:\n")); 1400 #define X(name, ...) if (!v13f.name) stream_append_str8(err, str8(" " #name "\n")); 1401 VK_REQUIRED_PHYSICAL_13_FEATURES 1402 #undef X 1403 fatal(stream_to_str8(err)); 1404 } 1405 } 1406 1407 if (vulkan_config.optional.cooperative_matrix) { 1408 u32 property_count = 0; 1409 vkGetPhysicalDeviceCooperativeMatrixPropertiesKHR(vk->physical_device, &property_count, 0); 1410 1411 VkCooperativeMatrixPropertiesKHR *mat = push_array(arena, VkCooperativeMatrixPropertiesKHR, property_count); 1412 1413 // NOTE(rnp): validation layer stupidity 1414 for EachIndex(property_count, it) 1415 mat[it].sType = VK_STRUCTURE_TYPE_COOPERATIVE_MATRIX_PROPERTIES_KHR; 1416 1417 vkGetPhysicalDeviceCooperativeMatrixPropertiesKHR(vk->physical_device, &property_count, mat); 1418 b32 supported = 0; 1419 // TODO(rnp): for now the requirements are hardcoded, it is possible to support a couple 1420 // variations if needed. 1421 for EachIndex(property_count, it) { 1422 b32 match = 1; 1423 match &= mat[it].scope == VK_SCOPE_SUBGROUP_KHR; 1424 1425 match &= mat[it].MSize == 16; 1426 match &= mat[it].NSize == 16; 1427 match &= mat[it].KSize == 16; 1428 1429 match &= mat[it].AType == VK_COMPONENT_TYPE_FLOAT16_KHR; 1430 match &= mat[it].BType == VK_COMPONENT_TYPE_FLOAT16_KHR; 1431 match &= mat[it].CType == VK_COMPONENT_TYPE_FLOAT32_KHR; 1432 match &= mat[it].ResultType == VK_COMPONENT_TYPE_FLOAT32_KHR; 1433 1434 supported |= match; 1435 } 1436 vk->gpu_info.cooperative_matrix = supported; 1437 } 1438 } 1439 1440 VkPhysicalDeviceMemoryProperties2 mp = {.sType = VK_STRUCTURE_TYPE_PHYSICAL_DEVICE_MEMORY_PROPERTIES_2}; 1441 vkGetPhysicalDeviceMemoryProperties2(vk->physical_device, &mp); 1442 1443 VkPhysicalDeviceMemoryProperties *bmp = &mp.memoryProperties; 1444 1445 // NOTE(rnp): vulkan spec says that highest performance memory types must 1446 // come first. just take the first one found. 1447 1448 for (u32 i = 0; i < bmp->memoryHeapCount; i++) { 1449 if (bmp->memoryHeaps[i].flags & VK_MEMORY_HEAP_DEVICE_LOCAL_BIT) { 1450 vk->memory_info.gpu_heap_index = i; 1451 break; 1452 } 1453 } 1454 1455 for (u32 i = 0; i < bmp->memoryTypeCount; i++) { 1456 if (bmp->memoryTypes[i].propertyFlags & VK_MEMORY_PROPERTY_DEVICE_LOCAL_BIT) { 1457 u32 heap_index = bmp->memoryTypes[i].heapIndex; 1458 assert(heap_index == vk->memory_info.gpu_heap_index); 1459 vk->memory_info.memory_type_indices[VulkanMemoryKind_Device] = i; 1460 vk->memory_info.memory_heap_sizes[VulkanMemoryKind_Device] = bmp->memoryHeaps[heap_index].size; 1461 break; 1462 } 1463 } 1464 1465 i32 bar_index = -1; 1466 1467 #if !ForceStagingBuffers 1468 u32 bar_flags = VK_MEMORY_PROPERTY_DEVICE_LOCAL_BIT|VK_MEMORY_PROPERTY_HOST_VISIBLE_BIT; 1469 for (u32 i = 0; i < bmp->memoryTypeCount; i++) { 1470 if ((bmp->memoryTypes[i].propertyFlags & bar_flags) == bar_flags) { 1471 u32 heap_index = bmp->memoryTypes[i].heapIndex; 1472 vk->memory_info.memory_heap_sizes[VulkanMemoryKind_BAR] = bmp->memoryHeaps[heap_index].size; 1473 bar_index = (i32)i; 1474 break; 1475 } 1476 } 1477 1478 vk->memory_info.memory_type_indices[VulkanMemoryKind_BAR] = bar_index; 1479 #endif 1480 1481 vk->memory_info.memory_type_indices[VulkanMemoryKind_Host] = -1; 1482 for (u32 i = 0; i < bmp->memoryTypeCount; i++) { 1483 if ((bmp->memoryTypes[i].propertyFlags & VK_MEMORY_PROPERTY_DEVICE_LOCAL_BIT) == 0) { 1484 if (bmp->memoryTypes[i].propertyFlags & VK_MEMORY_PROPERTY_HOST_VISIBLE_BIT) { 1485 u32 heap_index = bmp->memoryTypes[i].heapIndex; 1486 vk->memory_info.memory_type_indices[VulkanMemoryKind_Host] = (i8)i; 1487 vk->memory_info.memory_heap_sizes[VulkanMemoryKind_Host] = bmp->memoryHeaps[heap_index].size; 1488 break; 1489 } 1490 } 1491 } 1492 1493 // NOTE(rnp): some devices are fully unified so the only memory type is BAR memory 1494 if (vk->memory_info.memory_type_indices[VulkanMemoryKind_Host] == -1 && bar_index != -1) { 1495 vk->memory_info.memory_type_indices[VulkanMemoryKind_Host] = bar_index; 1496 vk->memory_info.memory_heap_sizes[VulkanMemoryKind_Host] = vk->memory_info.memory_heap_sizes[VulkanMemoryKind_BAR]; 1497 } 1498 1499 if (vk->memory_info.memory_type_indices[VulkanMemoryKind_Host] == -1) { 1500 stream_append_str8(err, vulkan_info("fatal error: vulkan driver does not provide host visible memory\n")); 1501 fatal(stream_to_str8(err)); 1502 } 1503 1504 for EachElement(vk->memory_info.memory_type_indices, it) { 1505 u32 ti = vk->memory_info.memory_type_indices[it]; 1506 u32 flags = bmp->memoryTypes[ti].propertyFlags; 1507 vk->memory_info.memory_host_coherent[it] = (flags & VK_MEMORY_PROPERTY_HOST_COHERENT_BIT) != 0; 1508 } 1509 1510 if (!SupportNonCoherentHostMemory) { 1511 if (!vk->memory_info.memory_host_coherent[VulkanMemoryKind_Host]) 1512 stream_append_str8(err, vulkan_info("fatal error: host visible memory is not coherent\n")); 1513 if (!vk->memory_info.memory_host_coherent[VulkanMemoryKind_BAR] && !ForceStagingBuffers) 1514 stream_append_str8(err, vulkan_info("fatal error: BAR memory is not coherent\n")); 1515 if (!vk->memory_info.memory_host_coherent[VulkanMemoryKind_Host] || 1516 (!vk->memory_info.memory_host_coherent[VulkanMemoryKind_BAR] && !ForceStagingBuffers)) 1517 { 1518 fatal(stream_to_str8(err)); 1519 } 1520 } 1521 1522 vulkan_config.driver_api_version = dp.properties.apiVersion; 1523 vk->memory_info.max_allocation_size = v11p.maxMemoryAllocationSize; 1524 vk->memory_info.non_coherent_atom_size = dp.properties.limits.nonCoherentAtomSize; 1525 vk->gpu_info.vendor = dp.properties.vendorID; 1526 vk->gpu_info.gpu_heap_size = bmp->memoryHeaps[vk->memory_info.gpu_heap_index].size; 1527 vk->gpu_info.timestamp_period_ns = dp.properties.limits.timestampPeriod; 1528 vk->gpu_info.max_image_dimension_2D = dp.properties.limits.maxImageDimension2D; 1529 vk->gpu_info.max_image_dimension_3D = dp.properties.limits.maxImageDimension3D; 1530 vk->gpu_info.max_msaa_samples = round_down_power_of_two(dp.properties.limits.framebufferColorSampleCounts); 1531 vk->gpu_info.subgroup_size = v11p.subgroupSize; 1532 vk->gpu_info.max_compute_shared_memory_size = dp.properties.limits.maxComputeSharedMemorySize; 1533 1534 temp_end(scratch); 1535 // IMPORTANT(rnp): memory must only be pushed at the end of the function 1536 vk->gpu_info.name = push_str8(vk->arena, str8_from_c_str(dp.properties.deviceName)); 1537 1538 #if VulkanDebug 1539 { 1540 b32 mismatch = 0; 1541 for EachElement(vk_validation_layers, it) { 1542 u32 lv = vulkan_config.layers.version.E[it]; 1543 u32 dv = vulkan_config.driver_api_version; 1544 if (lv < dv) { 1545 mismatch = 1; 1546 stream_append_str8s(err, vulkan_info("warning: validaton layer \""), 1547 vk_validation_layers[it], str8("\" version: ")); 1548 stream_appendf(err, "%u.%u.%u", VK_API_VERSION_MAJOR(lv), VK_API_VERSION_MINOR(lv), VK_API_VERSION_PATCH(lv)); 1549 stream_append_str8(err, str8(" lower than driver API version: ")); 1550 stream_appendf(err, "%u.%u.%u\n", VK_API_VERSION_MAJOR(dv), VK_API_VERSION_MINOR(dv), VK_API_VERSION_PATCH(dv)); 1551 } 1552 } 1553 1554 if (mismatch) 1555 stream_append_str8(err, vulkan_info("DO NOT report any bugs without updating your validation layers!\n")); 1556 } 1557 #endif 1558 } 1559 1560 function void 1561 vk_load_queues(Arena *arena, Stream *err) 1562 { 1563 /////////////////////////////////////////////////////// 1564 // NOTE(rnp): try to allocate an appropriate queue for 1565 // each of the following tasks: 1566 // * UI Rendering (Graphics) 1567 // * Beamforming (Compute) 1568 // * Upload (Transfer) 1569 // Then create a logical device ready for use 1570 1571 VulkanContext *vk = vulkan_context; 1572 1573 u32 queue_family_count; 1574 vkGetPhysicalDeviceQueueFamilyProperties(vk->physical_device, &queue_family_count, 0); 1575 1576 Temp scratch = temp_begin(arena); 1577 VkQueueFamilyProperties *queues = push_array(arena, typeof(*queues), queue_family_count); 1578 vkGetPhysicalDeviceQueueFamilyProperties(vk->physical_device, &queue_family_count, queues); 1579 1580 i32 queue_indices[VulkanQueueKind_Count]; 1581 for EachElement(queue_indices, it) queue_indices[it] = -1; 1582 1583 /////////////////////////////////////////////////////////////// 1584 // NOTE(rnp): start by assigning queue families for each queue 1585 1586 /* NOTE(rnp): try for exclusive transfer queue */ 1587 #if !ForceSingleQueue 1588 { 1589 u32 mask = VK_QUEUE_GRAPHICS_BIT|VK_QUEUE_COMPUTE_BIT|VK_QUEUE_TRANSFER_BIT; 1590 u32 max_timestamp_bits = 0; 1591 for (u32 index = 0; index < queue_family_count; index++) { 1592 if ((queues[index].queueFlags & mask) == VK_QUEUE_TRANSFER_BIT) { 1593 if (queues[index].timestampValidBits > max_timestamp_bits) { 1594 max_timestamp_bits = queues[index].timestampValidBits; 1595 queue_indices[VulkanQueueKind_Transfer] = (i32)index; 1596 } 1597 } 1598 } 1599 } 1600 1601 /* NOTE(rnp): try for compute separate from graphics */ 1602 for (u32 index = 0; index < queue_family_count; index++) { 1603 if ((queues[index].queueFlags & VK_QUEUE_COMPUTE_BIT) != 0 && 1604 (queues[index].queueFlags & VK_QUEUE_GRAPHICS_BIT) == 0) 1605 { 1606 queue_indices[VulkanQueueKind_Compute] = (i32)index; 1607 break; 1608 } 1609 } 1610 #endif /* !ForceSingleQueue */ 1611 1612 /* NOTE(rnp): find graphics family and verify it is exclusive */ 1613 b32 multi_graphics = 0; 1614 for (u32 index = 0; index < queue_family_count; index++) { 1615 if ((queues[index].queueFlags & VK_QUEUE_GRAPHICS_BIT) != 0) { 1616 // TODO(rnp): check for presentation support 1617 multi_graphics = queue_indices[VulkanQueueKind_Graphics] != -1; 1618 queue_indices[VulkanQueueKind_Graphics] = (i32)index; 1619 } 1620 } 1621 1622 if (multi_graphics) 1623 stream_append_str8(err, vulkan_info("warning: multiple queue families reported graphics support\n")); 1624 1625 if (queue_indices[VulkanQueueKind_Graphics] == -1) { 1626 stream_append_str8(err, vulkan_info("fatal error: GPU does not support graphics presentation\n")); 1627 fatal(stream_to_str8(err)); 1628 } 1629 1630 if (queue_indices[VulkanQueueKind_Compute] == -1) 1631 if ((queues[queue_indices[VulkanQueueKind_Graphics]].queueFlags & VK_QUEUE_COMPUTE_BIT) != 0) 1632 queue_indices[VulkanQueueKind_Compute] = queue_indices[VulkanQueueKind_Graphics]; 1633 1634 if (queue_indices[VulkanQueueKind_Compute] == -1) { 1635 stream_append_str8(err, vulkan_info("fatal error: GPU does not support compute\n")); 1636 fatal(stream_to_str8(err)); 1637 } 1638 1639 if (queue_indices[VulkanQueueKind_Transfer] == -1) { 1640 if ((queues[queue_indices[VulkanQueueKind_Compute]].queueFlags & VK_QUEUE_TRANSFER_BIT) != 0) 1641 queue_indices[VulkanQueueKind_Transfer] = queue_indices[VulkanQueueKind_Compute]; 1642 else if ((queues[queue_indices[VulkanQueueKind_Graphics]].queueFlags & VK_QUEUE_TRANSFER_BIT) != 0) 1643 queue_indices[VulkanQueueKind_Transfer] = queue_indices[VulkanQueueKind_Graphics]; 1644 } 1645 1646 if (queue_indices[VulkanQueueKind_Transfer] == -1) { 1647 stream_append_str8(err, vulkan_info("fatal error: GPU does not support data transfer\n")); 1648 fatal(stream_to_str8(err)); 1649 } 1650 1651 ///////////////////////////////////////////////////////////////// 1652 // NOTE(rnp): if queues share families try to allocate subqueues 1653 1654 u32 assigned_subindices[VulkanQueueKind_Count] = {0}; 1655 i32 queue_subindices[VulkanQueueKind_Count] = {0}; 1656 1657 assigned_subindices[VulkanQueueKind_Graphics] += 1; 1658 1659 if (queue_indices[VulkanQueueKind_Compute] == queue_indices[VulkanQueueKind_Graphics]) { 1660 if (assigned_subindices[VulkanQueueKind_Graphics] < queues[queue_indices[VulkanQueueKind_Graphics]].queueCount) 1661 queue_subindices[VulkanQueueKind_Compute] = assigned_subindices[VulkanQueueKind_Graphics]++; 1662 } else { 1663 assigned_subindices[VulkanQueueKind_Compute] += 1; 1664 } 1665 1666 if (queue_indices[VulkanQueueKind_Transfer] == queue_indices[VulkanQueueKind_Graphics]) { 1667 if (assigned_subindices[VulkanQueueKind_Graphics] < queues[queue_indices[VulkanQueueKind_Graphics]].queueCount) 1668 queue_subindices[VulkanQueueKind_Transfer] = assigned_subindices[VulkanQueueKind_Graphics]++; 1669 } else if (queue_indices[VulkanQueueKind_Transfer] == queue_indices[VulkanQueueKind_Compute]) { 1670 if (assigned_subindices[VulkanQueueKind_Compute] < queues[queue_indices[VulkanQueueKind_Compute]].queueCount) 1671 queue_subindices[VulkanQueueKind_Transfer] = assigned_subindices[VulkanQueueKind_Compute]++; 1672 } else { 1673 assigned_subindices[VulkanQueueKind_Transfer] += 1; 1674 } 1675 1676 for EachElement(assigned_subindices, it) 1677 vk->unique_queues += assigned_subindices[it]; 1678 1679 temp_end(scratch); 1680 1681 ///////////////////////////////////////////// 1682 // NOTE(rnp): fill in info and create device 1683 1684 u32 unique_queue_count = 0; 1685 for EachElement(vk->queues, i) { 1686 VulkanQueue *qp = 0; 1687 1688 u32 unique_queue_index = unique_queue_count; 1689 for EachElement(vk->queues, j) { 1690 if (vk->queues[j] && 1691 vk->queues[j]->queue_family == queue_indices[i] && 1692 vk->queues[j]->queue_index == queue_subindices[i]) 1693 { 1694 qp = vk->queues[j]; 1695 unique_queue_index = j; 1696 break; 1697 } 1698 } 1699 1700 if (!qp) { 1701 qp = push_struct(vk->arena, VulkanQueue); 1702 qp->queue_family = queue_indices[i]; 1703 qp->queue_index = queue_subindices[i]; 1704 unique_queue_count++; 1705 } 1706 1707 vk->queues[i] = qp; 1708 vk->queue_indices[i] = unique_queue_index; 1709 } 1710 1711 for EachElement(vk->command_pools, it) 1712 vk->command_pools[it] = push_struct(vk->arena, VulkanCommandPool); 1713 1714 VkDeviceQueueCreateInfo queue_create_infos[VulkanQueueKind_Count]; 1715 1716 f32 queue_priorities[VulkanQueueKind_Count][VulkanQueueKind_Count]; 1717 for (u32 i = 0; i < VulkanQueueKind_Count; i++) 1718 for (u32 j = 0; j < VulkanQueueKind_Count; j++) 1719 queue_priorities[i][j] = 1.0f; 1720 queue_priorities[vk->queue_indices[VulkanQueueKind_Compute]][queue_subindices[VulkanQueueKind_Compute]] = 0.5f; 1721 1722 u32 queue_create_index = 0; 1723 b32 queue_info_filled[VulkanQueueKind_Count] = {0}; 1724 for EachElement(vk->queue_indices, q) { 1725 u32 base_q = vk->queue_indices[q]; 1726 if (!queue_info_filled[base_q] && assigned_subindices[q] > 0) { 1727 queue_create_infos[queue_create_index++] = (VkDeviceQueueCreateInfo){ 1728 .sType = VK_STRUCTURE_TYPE_DEVICE_QUEUE_CREATE_INFO, 1729 .queueFamilyIndex = vk->queues[base_q]->queue_family, 1730 .queueCount = assigned_subindices[q], 1731 .pQueuePriorities = queue_priorities[q], 1732 }; 1733 } 1734 queue_info_filled[base_q] = 1; 1735 } 1736 1737 u32 enabled_count = 0; 1738 const char *enabled_extensions[MAX_ENABLED_EXTENSIONS]; 1739 1740 for EachElement(vk_required_device_extensions, it) 1741 enabled_extensions[enabled_count++] = (char *)vk_required_device_extensions[it].data; 1742 1743 for EachElement(vk_optional_device_extensions, it) 1744 if (vulkan_config.optional.E[it]) 1745 enabled_extensions[enabled_count++] = (char *)vk_optional_device_extensions[it].data; 1746 1747 for EachElement(vk_debug_extensions, it) 1748 if (vulkan_config.debug.E[it]) 1749 enabled_extensions[enabled_count++] = (char *)vk_debug_extensions[it].data; 1750 1751 VkDeviceCreateInfo device_create_info = { 1752 .sType = VK_STRUCTURE_TYPE_DEVICE_CREATE_INFO, 1753 .pQueueCreateInfos = queue_create_infos, 1754 .queueCreateInfoCount = queue_create_index, 1755 .ppEnabledExtensionNames = enabled_extensions, 1756 .enabledExtensionCount = enabled_count, 1757 }; 1758 1759 VkPhysicalDeviceShaderRelaxedExtendedInstructionFeaturesKHR pdsre = { 1760 .sType = VK_STRUCTURE_TYPE_PHYSICAL_DEVICE_SHADER_RELAXED_EXTENDED_INSTRUCTION_FEATURES_KHR, 1761 .shaderRelaxedExtendedInstruction = 1, 1762 }; 1763 if (vulkan_config.debug.shader_relaxed_extended_instruction) { 1764 pdsre.pNext = (void *)device_create_info.pNext; 1765 device_create_info.pNext = &pdsre; 1766 } 1767 1768 VkPhysicalDeviceCooperativeMatrixFeaturesKHR coop_mat_features = { 1769 .sType = VK_STRUCTURE_TYPE_PHYSICAL_DEVICE_COOPERATIVE_MATRIX_FEATURES_KHR, 1770 .cooperativeMatrix = 1, 1771 .cooperativeMatrixRobustBufferAccess = 0, 1772 }; 1773 if (vk->gpu_info.cooperative_matrix) { 1774 coop_mat_features.pNext = (void *)device_create_info.pNext; 1775 device_create_info.pNext = &coop_mat_features; 1776 } 1777 1778 VkPhysicalDeviceVulkan13Features v13f = { 1779 .sType = VK_STRUCTURE_TYPE_PHYSICAL_DEVICE_VULKAN_1_3_FEATURES, 1780 .pNext = (void *)device_create_info.pNext, 1781 #define X(name, ...) .name = 1, 1782 VK_REQUIRED_PHYSICAL_13_FEATURES 1783 #undef X 1784 }; 1785 device_create_info.pNext = &v13f; 1786 1787 VkPhysicalDeviceVulkan12Features v12f = { 1788 .sType = VK_STRUCTURE_TYPE_PHYSICAL_DEVICE_VULKAN_1_2_FEATURES, 1789 .pNext = (void *)device_create_info.pNext, 1790 #define X(name, ...) .name = 1, 1791 VK_REQUIRED_PHYSICAL_12_FEATURES 1792 #undef X 1793 }; 1794 device_create_info.pNext = &v12f; 1795 1796 VkPhysicalDeviceVulkan11Features v11f = { 1797 .sType = VK_STRUCTURE_TYPE_PHYSICAL_DEVICE_VULKAN_1_1_FEATURES, 1798 .pNext = (void *)device_create_info.pNext, 1799 #define X(name, ...) .name = 1, 1800 VK_REQUIRED_PHYSICAL_11_FEATURES 1801 #undef X 1802 }; 1803 device_create_info.pNext = &v11f; 1804 1805 VkPhysicalDeviceFeatures2 device_features = { 1806 .sType = VK_STRUCTURE_TYPE_PHYSICAL_DEVICE_FEATURES_2, 1807 .pNext = (void *)device_create_info.pNext, 1808 .features = { 1809 #define X(name, ...) .name = 1, 1810 VK_REQUIRED_PHYSICAL_FEATURES 1811 #undef X 1812 }, 1813 }; 1814 device_create_info.pNext = &device_features; 1815 1816 vkCreateDevice(vk->physical_device, &device_create_info, 0, &vk->device); 1817 1818 #define X(name, ...) name = (name##_fn *)vkGetDeviceProcAddr(vk->device, #name); 1819 VkDeviceProcedureList 1820 #undef X 1821 1822 for (u32 q = 0; q < vk->unique_queues; q++) { 1823 VulkanQueue *qp = vk->queues[vk->queue_indices[q]]; 1824 vkGetDeviceQueue(vk->device, qp->queue_family, qp->queue_index, &qp->queue); 1825 qp->timeline_semaphore = vk_make_semaphore(0); 1826 } 1827 1828 vk->queues[VulkanQueueKind_Graphics]->pipeline_stage_flags |= VK_PIPELINE_STAGE_2_ALL_GRAPHICS_BIT; 1829 vk->queues[VulkanQueueKind_Compute]->pipeline_stage_flags |= VK_PIPELINE_STAGE_2_COMPUTE_SHADER_BIT; 1830 1831 for EachElement(vk->command_pools, it) { 1832 VulkanCommandPool *vcp = vk->command_pools[it]; 1833 1834 vcp->arena = arena_create(.reserve_size = KB(64)); 1835 1836 VkCommandPoolCreateInfo command_pool_create_info = { 1837 .sType = VK_STRUCTURE_TYPE_COMMAND_POOL_CREATE_INFO, 1838 .flags = VK_COMMAND_POOL_CREATE_RESET_COMMAND_BUFFER_BIT, 1839 .queueFamilyIndex = vk->queues[it]->queue_family, 1840 }; 1841 1842 vkCreateCommandPool(vk->device, &command_pool_create_info, 0, &vcp->handle); 1843 1844 VkCommandBufferAllocateInfo command_buffer_allocate_info = { 1845 .sType = VK_STRUCTURE_TYPE_COMMAND_BUFFER_ALLOCATE_INFO, 1846 .commandPool = vcp->handle, 1847 .level = VK_COMMAND_BUFFER_LEVEL_PRIMARY, 1848 .commandBufferCount = countof(vcp->buffers), 1849 }; 1850 vkAllocateCommandBuffers(vk->device, &command_buffer_allocate_info, vcp->buffers); 1851 1852 VkQueryPoolCreateInfo query_pool_create_info = { 1853 .sType = VK_STRUCTURE_TYPE_QUERY_POOL_CREATE_INFO, 1854 .queryType = VK_QUERY_TYPE_TIMESTAMP, 1855 .queryCount = MaxCommandBuffersInFlight * MaxCommandBufferTimestamps, 1856 }; 1857 vkCreateQueryPool(vk->device, &query_pool_create_info, 0, &vcp->query_pool); 1858 } 1859 } 1860 1861 function void 1862 vk_load_graphics(void) 1863 { 1864 VulkanContext *vk = vulkan_context; 1865 1866 // NOTE: swap chain image format 1867 { 1868 } 1869 1870 // NOTE: depth/stencil format 1871 { 1872 VkFormat depth_formats[] = { 1873 VK_FORMAT_D32_SFLOAT_S8_UINT, 1874 VK_FORMAT_D24_UNORM_S8_UINT, 1875 VK_FORMAT_D16_UNORM_S8_UINT, 1876 }; 1877 1878 vk->depth_stencil_format = VK_FORMAT_UNDEFINED; 1879 for EachElement(depth_formats, it) { 1880 VkFormatProperties3 format_properties3 = {.sType = VK_STRUCTURE_TYPE_FORMAT_PROPERTIES_3}; 1881 VkFormatProperties2 format_properties2 = { 1882 .sType = VK_STRUCTURE_TYPE_FORMAT_PROPERTIES_2, 1883 .pNext = &format_properties3, 1884 }; 1885 vkGetPhysicalDeviceFormatProperties2(vk->physical_device, depth_formats[it], &format_properties2); 1886 if (format_properties3.optimalTilingFeatures & VK_FORMAT_FEATURE_2_DEPTH_STENCIL_ATTACHMENT_BIT) { 1887 vk->depth_stencil_format = depth_formats[it]; 1888 break; 1889 } 1890 } 1891 } 1892 } 1893 1894 /////////////////////// 1895 // NOTE(rnp): User API 1896 1897 DEBUG_IMPORT void 1898 vk_load(OSLibrary vulkan_library_handle, Stream *err) 1899 { 1900 #define X(name, ...) name = (name##_fn *)os_lookup_symbol(vulkan_library_handle, #name); 1901 VkLoaderProcedureList 1902 #undef X 1903 1904 if (!vkGetInstanceProcAddr) { 1905 stream_append_str8(err, vulkan_info("fatal error: failed to find \"vkGetInstanceProcAddr\"\n")); 1906 fatal(stream_to_str8(err)); 1907 } 1908 1909 VulkanContext *vk = vulkan_context; 1910 vk->arena = arena_create(.name = "Vulkan Arena"); 1911 vk->entity_arena = arena_create(.name = "Vulkan Entity Arena"); 1912 1913 vk_load_instance(vk->arena, err); 1914 vk_load_physical_device(vk->arena, err); 1915 vk_load_queues(vk->arena, err); 1916 vk_load_graphics(); 1917 1918 read_only str8 default_compute_shader = str8("" 1919 "#version 430 core\n" 1920 "layout(push_constant) uniform pc { uint data[256 / 4]; };\n" 1921 "void main() {}\n" 1922 "\n"); 1923 VulkanPipelineCreateInfo compute_create_info = {.text = default_compute_shader, .name = str8("error_compute_shader")}; 1924 vk->default_compute_pipeline = vk_compute_pipeline_from_info(vk->arena, &compute_create_info, 256); 1925 1926 read_only str8 default_vertex_shader = str8("" 1927 "#version 430 core\n" 1928 "layout(push_constant) uniform pc { uint data[256 / 4]; };\n" 1929 "void main() {gl_Position = vec4(0);}\n" 1930 "\n"); 1931 read_only str8 default_fragment_shader = str8("" 1932 "#version 430 core\n" 1933 "layout(location = 0) out vec4 out_colour;" 1934 "layout(push_constant) uniform pc { uint data[256 / 4]; };\n" 1935 "void main() {out_colour = vec4(0.5f, 0.0f, 0.5f, 1.0f);}\n" 1936 "\n"); 1937 1938 VulkanPipelineCreateInfo pipeline_create_infos[2] = { 1939 { 1940 .kind = VulkanShaderKind_Vertex, 1941 .text = default_vertex_shader, 1942 .name = str8("error_vertex_shader"), 1943 }, 1944 { 1945 .kind = VulkanShaderKind_Fragment, 1946 .text = default_fragment_shader, 1947 .name = str8("error_fragment_shader"), 1948 }, 1949 }; 1950 vk->default_graphics_pipeline = vk_graphics_pipeline_from_infos(vk->arena, pipeline_create_infos, 2, 256); 1951 1952 // TODO: setup ui render pipeline 1953 1954 if (err->widx > 0) { 1955 os_console_log(err->data, err->widx); 1956 stream_reset(err, 0); 1957 } 1958 } 1959 1960 DEBUG_IMPORT GPUInfo * 1961 gpu_info(void) 1962 { 1963 return &vulkan_context->gpu_info; 1964 } 1965 1966 function void 1967 vk_vulkan_buffer_release(VulkanBuffer *vb) 1968 { 1969 VulkanContext *vk = vulkan_context; 1970 VulkanEntity *e = (VulkanEntity *)((u8 *)vb - offsetof(VulkanEntity, as)); 1971 // TODO(rnp): this happens implicitly, probably just delete this if block 1972 if (vb->host_pointer) 1973 vkUnmapMemory(vk->device, vb->memory); 1974 1975 if (vb->buffer) 1976 vkDestroyBuffer(vk->device, vb->buffer, 0); 1977 1978 vk_release_memory(vb->memory, vb->memory_kind != VulkanMemoryKind_Host ? vb->memory_size : 0); 1979 vk_entity_release(e); 1980 } 1981 1982 DEBUG_IMPORT void 1983 gpu_buffer_release(GPUBuffer *b) 1984 { 1985 if (b->handle.value) { 1986 VulkanBuffer *vb = vk_entity_data(b->handle.value, VulkanEntityKind_Buffer); 1987 if (vb->next) 1988 vk_vulkan_buffer_release(vk_entity_data((u64)vb->next, VulkanEntityKind_Buffer)); 1989 vk_vulkan_buffer_release(vb); 1990 } 1991 zero_struct(b); 1992 } 1993 1994 DEBUG_IMPORT void 1995 gpu_buffer_allocate(GPUBuffer *b, GPUBufferAllocateInfo info) 1996 { 1997 VulkanContext *vk = vulkan_context; 1998 1999 gpu_buffer_release(b); 2000 2001 assert(info.size >= 0); 2002 2003 if (info.size > 0) { 2004 VulkanEntity *e = vk_entity_allocate(VulkanEntityKind_Buffer); 2005 VulkanBufferAllocateInfo vulkan_buffer_allocate_info = { 2006 .gpu_buffer = b, 2007 .size = (u64)info.size, 2008 .flags = info.flags, 2009 .index_type = VK_INDEX_TYPE_NONE_KHR, 2010 .label = info.label, 2011 .export = info.export, 2012 }; 2013 2014 u32 queue_index_hit_count[VulkanQueueKind_Count] = {0}; 2015 for (u32 it = 0; it < info.timeline_count; it++) 2016 queue_index_hit_count[vk->queue_indices[info.timelines_used[it]]]++; 2017 2018 for EachElement(queue_index_hit_count, it) { 2019 if (queue_index_hit_count[it] > 0) { 2020 u32 index = vulkan_buffer_allocate_info.queue_family_count++; 2021 vulkan_buffer_allocate_info.queue_family_indices[index] = vk->queues[vk->queue_indices[it]]->queue_family; 2022 } 2023 } 2024 2025 if (vk_buffer_allocate_common(&e->as.buffer, &vulkan_buffer_allocate_info)) { 2026 b->handle.value = (u64)e; 2027 } else { 2028 vk_entity_release(e); 2029 } 2030 } 2031 } 2032 2033 function void 2034 vk_command_copy_buffer(VkCommandBuffer cb, VkBuffer db, u64 destination_offset, VkBuffer sb, u64 source_offset, u64 size) 2035 { 2036 VkBufferCopy2 buffer_copy = { 2037 .sType = VK_STRUCTURE_TYPE_BUFFER_COPY_2, 2038 .srcOffset = source_offset, 2039 .dstOffset = destination_offset, 2040 .size = size, 2041 }; 2042 2043 VkCopyBufferInfo2 copy_buffer_info = { 2044 .sType = VK_STRUCTURE_TYPE_COPY_BUFFER_INFO_2, 2045 .srcBuffer = sb, 2046 .dstBuffer = db, 2047 .regionCount = 1, 2048 .pRegions = &buffer_copy, 2049 }; 2050 2051 vkCmdCopyBuffer2(cb, ©_buffer_info); 2052 } 2053 2054 DEBUG_IMPORT u64 2055 gpu_semaphore_value(GPUSemaphore semaphore) 2056 { 2057 VulkanSemaphore *fs = vk_entity_data(semaphore.value, VulkanEntityKind_Semaphore); 2058 u64 result = fs->value; 2059 return result; 2060 } 2061 2062 DEBUG_IMPORT void 2063 gpu_host_signal_semaphore(GPUSemaphore semaphore, u64 value) 2064 { 2065 VulkanSemaphore *vs = vk_entity_data(semaphore.value, VulkanEntityKind_Semaphore); 2066 assert(vs->value < value); 2067 VkSemaphoreSignalInfo ssi = { 2068 .sType = VK_STRUCTURE_TYPE_SEMAPHORE_SIGNAL_INFO, 2069 .semaphore = vs->semaphore, 2070 .value = value, 2071 }; 2072 vs->value = value; 2073 vkSignalSemaphore(vulkan_context->device, &ssi); 2074 } 2075 2076 DEBUG_IMPORT void * 2077 gpu_buffer_host_pointer(GPUBuffer *b) 2078 { 2079 void *result = 0; 2080 if (b->handle.value) { 2081 VulkanBuffer *vb = vk_entity_data(b->handle.value, VulkanEntityKind_Buffer); 2082 result = vb->next ? vb->next->as.buffer.host_pointer : vb->host_pointer; 2083 } 2084 return result; 2085 } 2086 2087 DEBUG_IMPORT u64 2088 gpu_buffer_make_visible(GPUBuffer *b, u64 offset, u64 size, 2089 GPUSemaphoreSignalInfo *signal_infos, u64 signal_info_count) 2090 { 2091 u64 result = 0; 2092 if (b->handle.value) { 2093 VulkanBuffer *vb = vk_entity_data(b->handle.value, VulkanEntityKind_Buffer); 2094 if (vb->memory_kind == VulkanMemoryKind_Device && vb->next) { 2095 GPUCommandList cb = gpu_command_list_begin(GPUTimeline_Transfer); 2096 vk_command_copy_buffer(vk_command_buffer(cb), vb->buffer, offset, vb->next->as.buffer.buffer, offset, size); 2097 result = gpu_command_list_end(cb, 0, 0, signal_infos, signal_info_count); 2098 } else { 2099 for EachIndex(signal_info_count, index) 2100 gpu_host_signal_semaphore(signal_infos[index].semaphore, signal_infos[index].value); 2101 } 2102 } 2103 return result; 2104 } 2105 2106 DEBUG_IMPORT b32 2107 gpu_buffer_needs_sync(GPUBuffer *b) 2108 { 2109 b32 result = 0; 2110 if (b->handle.value) { 2111 VulkanBuffer *vb = vk_entity_data(b->handle.value, VulkanEntityKind_Buffer); 2112 result = vb->next != 0; 2113 } 2114 return result; 2115 } 2116 2117 DEBUG_IMPORT u64 2118 gpu_round_up_to_sync_size(u64 size, u64 min) 2119 { 2120 i64 round = (i64)Max(min, vulkan_context->memory_info.non_coherent_atom_size); 2121 u64 result = (u64)round_up_to((i64)size, round); 2122 return result; 2123 } 2124 2125 function force_inline u64 2126 vk_buffer_buffer_copy(VulkanBuffer *destination, VulkanBuffer *source, u64 destination_offset, u64 source_offset, u64 size, b32 non_temporal) 2127 { 2128 VulkanContext *vk = vulkan_context; 2129 (void)vk; 2130 2131 u64 result = 0; 2132 switch (source->memory_kind) { 2133 #if !ForceStagingBuffers 2134 case VulkanMemoryKind_BAR: 2135 { 2136 switch (destination->memory_kind) { 2137 case VulkanMemoryKind_Host:{ 2138 if (destination->memory) { 2139 // TODO(rnp): there is likely a more efficient way of doing this in this case 2140 InvalidCodePath; 2141 } else { 2142 assert(source->host_pointer); 2143 #if SupportNonCoherentHostMemory 2144 b32 coherent = vk->memory_info.memory_host_coherent[source->memory_kind]; 2145 if (!coherent) { 2146 u64 nca_size = vk->memory_info.non_coherent_atom_size; 2147 VkMappedMemoryRange mrs[1] = {{ 2148 .sType = VK_STRUCTURE_TYPE_MAPPED_MEMORY_RANGE, 2149 .memory = source->memory, 2150 .offset = source_offset - (source_offset % nca_size), 2151 .size = gpu_round_up_to_sync_size(size, nca_size), 2152 }}; 2153 vkInvalidateMappedMemoryRanges(vk->device, countof(mrs), mrs); 2154 } 2155 #else 2156 assert(vk->memory_info.memory_host_coherent[source->memory_kind]); 2157 #endif 2158 2159 void *dest = (u8 *)destination->host_pointer + destination_offset; 2160 void *src = (u8 *)source->host_pointer + source_offset; 2161 2162 // NOTE(rnp): don't trash the CPU cache for large data stores 2163 if (non_temporal) memory_copy_non_temporal(dest, src, size); 2164 else memory_copy(dest, src, size); 2165 } 2166 }break; 2167 InvalidDefaultCase; 2168 } 2169 }break; 2170 #endif 2171 2172 case VulkanMemoryKind_Host:{ 2173 switch (destination->memory_kind) { 2174 #if !ForceStagingBuffers 2175 case VulkanMemoryKind_BAR:{ 2176 assert(destination->host_pointer); 2177 2178 void *dest = (u8 *)destination->host_pointer + destination_offset; 2179 void *src = (u8 *)source->host_pointer + source_offset; 2180 2181 // NOTE(rnp): don't trash the CPU cache for large data stores 2182 if (non_temporal) memory_copy_non_temporal(dest, src, size); 2183 else memory_copy(dest, src, size); 2184 2185 #if SupportNonCoherentHostMemory 2186 b32 coherent = vk->memory_info.memory_host_coherent[destination->memory_kind]; 2187 if (!coherent) { 2188 u64 nca_size = vk->memory_info.non_coherent_atom_size; 2189 VkMappedMemoryRange mrs[1] = {{ 2190 .sType = VK_STRUCTURE_TYPE_MAPPED_MEMORY_RANGE, 2191 .memory = destination->memory, 2192 .offset = destination_offset - (destination_offset % nca_size), 2193 .size = gpu_round_up_to_sync_size(size, nca_size), 2194 }}; 2195 vkFlushMappedMemoryRanges(vk->device, countof(mrs), mrs); 2196 } 2197 #else 2198 assert(vk->memory_info.memory_host_coherent[destination->memory_kind]); 2199 #endif 2200 }break; 2201 #endif 2202 2203 case VulkanMemoryKind_Device:{ 2204 VulkanBuffer *db = vk_entity_data((u64)destination->next, VulkanEntityKind_Buffer); 2205 assert(size <= db->memory_size); 2206 void *dest = (u8 *)db->host_pointer + destination_offset; 2207 void *src = (u8 *)source->host_pointer + source_offset; 2208 // NOTE(rnp): don't trash the CPU cache for large data stores 2209 if (non_temporal) memory_copy_non_temporal(dest, src, size); 2210 else memory_copy(dest, src, size); 2211 store_fence(); 2212 2213 GPUCommandList cb = gpu_command_list_begin(GPUTimeline_Transfer); 2214 vk_command_copy_buffer(vk_command_buffer(cb), destination->buffer, destination_offset, db->buffer, destination_offset, size); 2215 result = gpu_command_list_end(cb, 0, 0, 0, 0); 2216 }break; 2217 2218 InvalidDefaultCase; 2219 2220 } 2221 }break; 2222 2223 case VulkanMemoryKind_Device:{ 2224 switch (destination->memory_kind) { 2225 // NOTE(rnp): only host memory is allowed for destination here 2226 InvalidDefaultCase; 2227 case VulkanMemoryKind_Host:{ 2228 VulkanBuffer *sb = vk_entity_data((u64)source->next, VulkanEntityKind_Buffer); 2229 2230 assert(size <= sb->memory_size); 2231 2232 GPUCommandList cb = gpu_command_list_begin(GPUTimeline_Transfer); 2233 vk_command_copy_buffer(vk_command_buffer(cb), sb->buffer, 0, source->buffer, source_offset, size); 2234 u64 wait_value = gpu_command_list_end(cb, 0, 0, 0, 0); 2235 // TODO(rnp): asynchronous transfers 2236 gpu_host_wait_timeline(GPUTimeline_Transfer, wait_value, -1ULL); 2237 2238 void *dest = (u8 *)destination->host_pointer + destination_offset; 2239 void *src = (u8 *)sb->host_pointer; 2240 // NOTE(rnp): don't trash the CPU cache for large data stores 2241 if (non_temporal) memory_copy_non_temporal(dest, src, size); 2242 else memory_copy(dest, src, size); 2243 }break; 2244 } 2245 }break; 2246 2247 InvalidDefaultCase; 2248 } 2249 2250 return result; 2251 } 2252 2253 DEBUG_IMPORT u64 2254 gpu_buffer_range_upload(GPUBuffer *b, void *data, u64 offset, u64 size, b32 non_temporal) 2255 { 2256 VulkanBuffer *db = vk_entity_data(b->handle.value, VulkanEntityKind_Buffer); 2257 VulkanBuffer sb = { 2258 .host_pointer = data, 2259 .memory_kind = VulkanMemoryKind_Host, 2260 }; 2261 u64 result = vk_buffer_buffer_copy(db, &sb, offset, 0, size, non_temporal); 2262 return result; 2263 } 2264 2265 DEBUG_IMPORT void 2266 gpu_buffer_range_download(void *destination, GPUBuffer *source, u64 offset, u64 size, b32 non_temporal) 2267 { 2268 VulkanBuffer *sb = vk_entity_data(source->handle.value, VulkanEntityKind_Buffer); 2269 VulkanBuffer db = { 2270 .host_pointer = destination, 2271 .memory_kind = VulkanMemoryKind_Host, 2272 }; 2273 vk_buffer_buffer_copy(&db, sb, 0, offset, size, non_temporal); 2274 } 2275 2276 DEBUG_IMPORT void 2277 vk_render_model_release(GPUBuffer *model) 2278 { 2279 if (model->handle.value) 2280 vk_vulkan_buffer_release(vk_entity_data(model->handle.value, VulkanEntityKind_RenderModel)); 2281 zero_struct(model); 2282 } 2283 2284 DEBUG_IMPORT void 2285 vk_render_model_allocate(GPUBuffer *model, void *indices, u64 index_count, u64 model_size, str8 label) 2286 { 2287 vk_render_model_release(model); 2288 2289 VulkanEntity *e = vk_entity_allocate(VulkanEntityKind_RenderModel); 2290 2291 assert(index_count <= U32_MAX); 2292 VkIndexType index_type; 2293 if (index_count <= U16_MAX) index_type = VK_INDEX_TYPE_UINT16; 2294 else index_type = VK_INDEX_TYPE_UINT32; 2295 2296 i64 indices_size = round_up_to(vk_index_size(index_type) * index_count, 64); 2297 2298 i64 size = round_up_to(model_size + indices_size, 64); 2299 assert(size > 0); 2300 2301 VulkanBufferAllocateInfo vulkan_buffer_allocate_info = { 2302 .gpu_buffer = model, 2303 .size = (u64)size, 2304 .flags = GPUUsageFlag_HostWrite, 2305 .index_type = index_type, 2306 .label = label, 2307 .queue_family_count = 1, 2308 .queue_family_indices[0] = vulkan_context->queues[VulkanQueueKind_Graphics]->queue_family, 2309 }; 2310 if (vk_buffer_allocate_common(&e->as.buffer, &vulkan_buffer_allocate_info)) { 2311 model->handle.value = (u64)e; 2312 model->index_count = index_count; 2313 model->gpu_pointer += indices_size; 2314 2315 VulkanBuffer sb = { 2316 .host_pointer = indices, 2317 .memory_kind = VulkanMemoryKind_Host, 2318 }; 2319 2320 vk_buffer_buffer_copy(&e->as.buffer, &sb, 0, 0, vk_index_size(index_type) * index_count, 0); 2321 } else { 2322 vk_entity_release(e); 2323 } 2324 } 2325 2326 DEBUG_IMPORT void 2327 vk_render_model_range_upload(GPUBuffer *model, void *data, u64 offset, u64 size, b32 non_temporal) 2328 { 2329 VulkanBuffer *db = vk_entity_data(model->handle.value, VulkanEntityKind_RenderModel); 2330 VulkanBuffer sb = { 2331 .host_pointer = data, 2332 .memory_kind = VulkanMemoryKind_Host, 2333 }; 2334 2335 offset += round_up_to(vk_index_size(db->index_type) * model->index_count, 64); 2336 2337 vk_buffer_buffer_copy(db, &sb, offset, 0, size, non_temporal); 2338 } 2339 2340 DEBUG_IMPORT void 2341 vk_image_release(GPUImage *image) 2342 { 2343 if ValidVulkanHandle(image->image) { 2344 VulkanContext *vk = vulkan_context; 2345 VulkanImage *vi = vk_entity_data(image->image.value[0], VulkanEntityKind_Image); 2346 2347 vkDestroyImageView(vk->device, vi->view, 0); 2348 vkDestroyImage(vk->device, vi->image, 0); 2349 vk_release_memory(vi->memory, image->memory_size); 2350 2351 vk_entity_release((VulkanEntity *)image->image.value[0]); 2352 } 2353 zero_struct(image); 2354 } 2355 2356 DEBUG_IMPORT void 2357 vk_image_allocate(GPUImage *image, u32 width, u32 height, u32 mips, u32 samples, 2358 VulkanImageUsage usage, GPUUsageFlags flags, OSHandle *export, str8 label) 2359 { 2360 assert(IsPowerOfTwo(samples)); 2361 2362 vk_image_release(image); 2363 2364 VulkanContext *vk = vulkan_context; 2365 VulkanEntity *e = vk_entity_allocate(VulkanEntityKind_Image); 2366 VulkanImage *vi = &e->as.image; 2367 2368 image->image.value[0] = (u64)e; 2369 image->width = Min(width, vk->gpu_info.max_image_dimension_2D); 2370 image->height = Min(height, vk->gpu_info.max_image_dimension_2D); 2371 image->mip_map_levels = Max(mips, 1); 2372 image->samples = Min(samples, vk->gpu_info.max_msaa_samples); 2373 2374 VkFormat usage_format_map[VulkanImageUsage_Count + 1] = { 2375 [VulkanImageUsage_None] = VK_FORMAT_UNDEFINED, 2376 //[VulkanImageUsage_Colour] = VK_FORMAT_R8G8B8A8_SRGB, 2377 [VulkanImageUsage_Colour] = VK_FORMAT_R8G8B8A8_UNORM, 2378 [VulkanImageUsage_DepthStencil] = vk->depth_stencil_format, 2379 [VulkanImageUsage_Count] = VK_FORMAT_UNDEFINED, 2380 }; 2381 2382 read_only VkImageUsageFlagBits usage_extra_bit_map[VulkanImageUsage_Count + 1] = { 2383 [VulkanImageUsage_None] = 0, 2384 [VulkanImageUsage_Colour] = VK_IMAGE_USAGE_COLOR_ATTACHMENT_BIT, 2385 [VulkanImageUsage_DepthStencil] = VK_IMAGE_USAGE_DEPTH_STENCIL_ATTACHMENT_BIT, 2386 [VulkanImageUsage_Count] = 0, 2387 }; 2388 2389 read_only VkImageAspectFlags usage_image_aspect_map[VulkanImageUsage_Count + 1] = { 2390 [VulkanImageUsage_None] = 0, 2391 [VulkanImageUsage_Colour] = VK_IMAGE_ASPECT_COLOR_BIT, 2392 [VulkanImageUsage_DepthStencil] = VK_IMAGE_ASPECT_DEPTH_BIT|VK_IMAGE_ASPECT_STENCIL_BIT, 2393 [VulkanImageUsage_Count] = 0, 2394 }; 2395 2396 usage = Clamp((u32)usage, 0, VulkanImageUsage_Count); 2397 VkImageUsageFlagBits usage_flags = usage_extra_bit_map[usage]; 2398 2399 if (flags & GPUUsageFlag_ImageSampling) usage_flags |= VK_IMAGE_USAGE_SAMPLED_BIT; 2400 if (flags & GPUUsageFlag_TransferSource) usage_flags |= VK_IMAGE_USAGE_TRANSFER_SRC_BIT; 2401 if (flags & GPUUsageFlag_TransferDestination) usage_flags |= VK_IMAGE_USAGE_TRANSFER_DST_BIT; 2402 2403 u32 queue_family = vk->queues[VulkanQueueKind_Graphics]->queue_family; 2404 VkImageCreateInfo image_create_info = { 2405 .sType = VK_STRUCTURE_TYPE_IMAGE_CREATE_INFO, 2406 .flags = export ? VK_IMAGE_CREATE_MUTABLE_FORMAT_BIT : 0, 2407 .imageType = VK_IMAGE_TYPE_2D, 2408 .format = usage_format_map[usage], 2409 .extent = {image->width, image->height, 1}, 2410 .mipLevels = image->mip_map_levels, 2411 .arrayLayers = 1, 2412 .samples = image->samples, 2413 .tiling = VK_IMAGE_TILING_OPTIMAL, 2414 .usage = usage_flags, 2415 // NOTE(rnp): needed if multiple queue families are accessed 2416 .sharingMode = VK_SHARING_MODE_EXCLUSIVE, 2417 .queueFamilyIndexCount = 1, 2418 .pQueueFamilyIndices = &queue_family, 2419 .initialLayout = VK_IMAGE_LAYOUT_UNDEFINED, 2420 }; 2421 2422 VkExternalMemoryImageCreateInfo external_memory_image_create_info = { 2423 .sType = VK_STRUCTURE_TYPE_EXTERNAL_MEMORY_IMAGE_CREATE_INFO, 2424 .handleTypes = OS_WINDOWS ? VK_EXTERNAL_MEMORY_HANDLE_TYPE_OPAQUE_WIN32_BIT 2425 : VK_EXTERNAL_MEMORY_HANDLE_TYPE_OPAQUE_FD_BIT, 2426 }; 2427 2428 if (export) image_create_info.pNext = &external_memory_image_create_info; 2429 2430 vkCreateImage(vk->device, &image_create_info, 0, &vi->image); 2431 2432 VkMemoryRequirements memory_requirements; 2433 vkGetImageMemoryRequirements(vk->device, vi->image, &memory_requirements); 2434 2435 VkMemoryDedicatedAllocateInfo dedicated_allocate_info = { 2436 .sType = VK_STRUCTURE_TYPE_MEMORY_DEDICATED_ALLOCATE_INFO, 2437 .image = vi->image, 2438 }; 2439 2440 if (vk_allocate_memory(&vi->memory, memory_requirements.size, VulkanMemoryKind_Device, 0, &dedicated_allocate_info, export)) { 2441 image->memory_size = memory_requirements.size; 2442 vkBindImageMemory(vk->device, vi->image, vi->memory, 0); 2443 2444 VkImageViewCreateInfo image_view_info = { 2445 .sType = VK_STRUCTURE_TYPE_IMAGE_VIEW_CREATE_INFO, 2446 .image = vi->image, 2447 .viewType = VK_IMAGE_VIEW_TYPE_2D, 2448 .format = usage_format_map[usage], 2449 .subresourceRange = { 2450 .aspectMask = usage_image_aspect_map[usage], 2451 .baseMipLevel = 0, 2452 .levelCount = 1, 2453 .baseArrayLayer = 0, 2454 .layerCount = 1, 2455 }, 2456 }; 2457 vkCreateImageView(vk->device, &image_view_info, 0, &vi->view); 2458 2459 vk_label_object(IMAGE, vi->image, label, str8("Image")); 2460 vk_label_object(IMAGE_VIEW, vi->view, label, str8("Image View")); 2461 vk_label_object(DEVICE_MEMORY, vi->memory, label, str8("Memory")); 2462 } else { 2463 vkDestroyImage(vk->device, vi->image, 0); 2464 vk_entity_release(e); 2465 zero_struct(image); 2466 } 2467 } 2468 2469 DEBUG_IMPORT GPUSemaphore 2470 gpu_semaphore_create(OSHandle *export) 2471 { 2472 VulkanEntity *e = vk_entity_allocate(VulkanEntityKind_Semaphore); 2473 e->as.semaphore = vk_make_semaphore(export); 2474 GPUSemaphore result = {(u64)e}; 2475 return result; 2476 } 2477 2478 DEBUG_IMPORT b32 2479 gpu_host_wait_timeline(GPUTimeline timeline, u64 value, u64 timeout_ns) 2480 { 2481 b32 result = 0; 2482 if Between(timeline, 0, GPUTimeline_Count - 1) { 2483 VulkanContext *vk = vulkan_context; 2484 VulkanQueue *vq = vk->queues[timeline]; 2485 VkSemaphoreWaitInfo semaphore_wait_info = { 2486 .sType = VK_STRUCTURE_TYPE_SEMAPHORE_WAIT_INFO, 2487 .pSemaphores = &vq->timeline_semaphore.semaphore, 2488 .semaphoreCount = 1, 2489 .pValues = &value, 2490 }; 2491 result = vkWaitSemaphores(vk->device, &semaphore_wait_info, timeout_ns) == VK_SUCCESS; 2492 } 2493 return result; 2494 } 2495 2496 DEBUG_IMPORT u64 2497 gpu_host_signal_timeline(GPUTimeline timeline) 2498 { 2499 u64 result = -1; 2500 if Between(timeline, 0, GPUTimeline_Count - 1) { 2501 VulkanContext *vk = vulkan_context; 2502 VulkanQueue *vq = vk->queues[timeline]; 2503 VulkanSemaphore *vs = &vq->timeline_semaphore; 2504 result = ++vs->value; 2505 VkSemaphoreSignalInfo ssi = { 2506 .sType = VK_STRUCTURE_TYPE_SEMAPHORE_SIGNAL_INFO, 2507 .semaphore = vs->semaphore, 2508 .value = result, 2509 }; 2510 vkSignalSemaphore(vk->device, &ssi); 2511 } 2512 return result; 2513 } 2514 2515 DEBUG_IMPORT VulkanHandle 2516 vk_pipeline(VulkanPipelineCreateInfo *infos, u32 count, u32 push_constants_size) 2517 { 2518 assert(Between(count, 1, 2)); 2519 assert(count == 2 || infos[0].kind == VulkanShaderKind_Compute); 2520 2521 VulkanHandle result = {0}; 2522 Temp scratch; 2523 DeferLoop(take_lock(&vulkan_context->arena_lock, -1), release_lock(&vulkan_context->arena_lock)) 2524 DeferLoop(scratch = temp_begin(vulkan_context->arena), temp_end(scratch)) 2525 { 2526 VulkanEntity *e = vk_entity_allocate(VulkanEntityKind_Pipeline); 2527 result = (VulkanHandle){(u64)e}; 2528 2529 if (count == 2) e->as.pipeline = vk_graphics_pipeline_from_infos(scratch.arena, infos, count, push_constants_size); 2530 else e->as.pipeline = vk_compute_pipeline_from_info(scratch.arena, infos, push_constants_size); 2531 } 2532 return result; 2533 } 2534 2535 DEBUG_IMPORT b32 2536 vk_pipeline_valid(VulkanHandle h) 2537 { 2538 b32 result = 0; 2539 if ValidVulkanHandle(h) { 2540 VulkanPipeline *vp = vk_entity_data(h.value[0], VulkanEntityKind_Pipeline); 2541 if (vp->stage_flags == VK_SHADER_STAGE_COMPUTE_BIT) 2542 result = vp->pipeline != vulkan_context->default_compute_pipeline.pipeline; 2543 else 2544 result = vp->pipeline != vulkan_context->default_graphics_pipeline.pipeline; 2545 } 2546 return result; 2547 } 2548 2549 DEBUG_IMPORT void 2550 vk_pipeline_release(VulkanHandle h) 2551 { 2552 if (vk_pipeline_valid(h)) { 2553 VulkanEntity *e = (VulkanEntity *)h.value[0]; 2554 GPUTimeline timeline; 2555 // TODO(rnp): this is not correct, compute shaders can also appear on graphics timeline 2556 if (e->as.pipeline.stage_flags == VK_SHADER_STAGE_COMPUTE_BIT) timeline = GPUTimeline_Compute; 2557 else timeline = GPUTimeline_Graphics; 2558 2559 // NOTE(rnp): block more command buffers from being recorded 2560 VulkanCommandPool *vcp = vulkan_context->command_pools[timeline]; 2561 DeferLoop(take_lock(&vcp->lock, -1), release_lock(&vcp->lock)) 2562 { 2563 u32 index = (vcp->next_command_buffer_index - 1) % MaxCommandBuffersInFlight; 2564 gpu_host_wait_timeline(timeline, vcp->last_submission_values[index], -1ULL); 2565 vkDestroyPipeline(vulkan_context->device, e->as.pipeline.pipeline, 0); 2566 vkDestroyPipelineLayout(vulkan_context->device, e->as.pipeline.layout, 0); 2567 2568 if (&e->as.pipeline == vcp->bound_pipeline) 2569 vcp->bound_pipeline = 0; 2570 } 2571 vk_entity_release(e); 2572 } 2573 } 2574 2575 DEBUG_IMPORT GPUCommandList 2576 gpu_command_list_begin(GPUTimeline timeline) 2577 { 2578 GPUCommandList result = {0}; 2579 if Between(timeline, 0, GPUTimeline_Count - 1) { 2580 VulkanContext *vk = vulkan_context; 2581 VulkanCommandPool *vcp = vk->command_pools[timeline]; 2582 2583 take_lock(&vcp->lock, -1); 2584 VulkanEntity *e = vk_entity_allocate(VulkanEntityKind_CommandBuffer); 2585 result.value = (u64)e; 2586 2587 VulkanCommandBuffer *vcb = &e->as.command_buffer; 2588 vcb->timeline = timeline; 2589 vcb->buffer_index = (vcp->next_command_buffer_index++) % MaxCommandBuffersInFlight; 2590 2591 u32 index = vcb->buffer_index; 2592 // TODO(rnp): probably not the best to have this here but it will likely not be hit 2593 b32 wait_result = gpu_host_wait_timeline(timeline, vcp->last_submission_values[index], -1ULL); 2594 assert(wait_result); 2595 2596 vcp->timestamp_counts[index] = 0; 2597 2598 VkCommandBufferBeginInfo buffer_begin_info = { 2599 .sType = VK_STRUCTURE_TYPE_COMMAND_BUFFER_BEGIN_INFO, 2600 .flags = VK_COMMAND_BUFFER_USAGE_ONE_TIME_SUBMIT_BIT, 2601 }; 2602 2603 vkBeginCommandBuffer(vcp->buffers[index], &buffer_begin_info); 2604 vkCmdResetQueryPool(vcp->buffers[index], vcp->query_pool, index * MaxCommandBufferTimestamps, 2605 MaxCommandBufferTimestamps); 2606 } 2607 return result; 2608 } 2609 2610 DEBUG_IMPORT void 2611 gpu_command_bind_pipeline(GPUCommandList command, VulkanHandle pipeline) 2612 { 2613 if (command.value) { 2614 VulkanContext *vk = vulkan_context; 2615 VulkanCommandBuffer *vcb = vk_entity_data(command.value, VulkanEntityKind_CommandBuffer); 2616 VulkanCommandPool *vcp = vk->command_pools[vcb->timeline]; 2617 2618 VulkanPipeline *vp = 0; 2619 if ValidVulkanHandle(pipeline) { 2620 vp = vk_entity_data(pipeline.value[0], VulkanEntityKind_Pipeline); 2621 } else if (vcb->timeline == GPUTimeline_Compute) { 2622 vp = &vk->default_compute_pipeline; 2623 } else if (vcb->timeline == GPUTimeline_Graphics) { 2624 vp = &vk->default_graphics_pipeline; 2625 } else { 2626 InvalidCodePath; 2627 } 2628 2629 read_only VkPipelineBindPoint bind_point_lut[GPUTimeline_Count] = { 2630 [GPUTimeline_Graphics] = VK_PIPELINE_BIND_POINT_GRAPHICS, 2631 [GPUTimeline_Compute] = VK_PIPELINE_BIND_POINT_COMPUTE, 2632 [GPUTimeline_Transfer] = -1, 2633 }; 2634 2635 VkPipelineBindPoint bind_point = bind_point_lut[vcb->timeline]; 2636 assert(bind_point != (VkPipelineBindPoint)-1); 2637 2638 VkCommandBuffer cmd = vk_command_buffer(command); 2639 vkCmdBindPipeline(cmd, bind_point, vp->pipeline); 2640 vcp->bound_pipeline = vp; 2641 } 2642 } 2643 2644 DEBUG_IMPORT void 2645 gpu_command_pipeline_barrier(GPUCommandList command, b32 memory) 2646 { 2647 if (command.value) { 2648 VulkanContext *vk = vulkan_context; 2649 VulkanCommandBuffer *vcb = vk_entity_data(command.value, VulkanEntityKind_CommandBuffer); 2650 VulkanQueue *vq = vk->queues[vcb->timeline]; 2651 2652 // TODO(rnp): this is not really specific enough 2653 b32 needs_memory = memory || (vq->pipeline_stage_flags != VK_PIPELINE_STAGE_2_COMPUTE_SHADER_BIT); 2654 2655 VkMemoryBarrier2 memory_barrier = { 2656 .sType = VK_STRUCTURE_TYPE_MEMORY_BARRIER_2, 2657 .srcStageMask = vq->pipeline_stage_flags, 2658 .dstStageMask = vq->pipeline_stage_flags, 2659 }; 2660 2661 // NOTE(rnp): COMPUTE->COMPUTE does not need memory guards. LLC is coherent on modern GPUs. 2662 if (needs_memory) { 2663 memory_barrier.srcAccessMask = VK_ACCESS_2_MEMORY_WRITE_BIT; 2664 memory_barrier.dstAccessMask = VK_ACCESS_2_MEMORY_READ_BIT; 2665 } 2666 2667 VkDependencyInfo dependency_info = { 2668 .sType = VK_STRUCTURE_TYPE_DEPENDENCY_INFO, 2669 .pMemoryBarriers = &memory_barrier, 2670 .memoryBarrierCount = 1, 2671 }; 2672 vkCmdPipelineBarrier2(vk_command_buffer(command), &dependency_info); 2673 } 2674 } 2675 2676 DEBUG_IMPORT void 2677 gpu_command_clear_buffer(GPUCommandList command, GPUBuffer *buffer, u64 offset, u64 size, u32 clear_word) 2678 { 2679 assert((offset % 4) == 0); 2680 assert((size % 4) == 0); 2681 if (command.value) { 2682 VulkanBuffer *vb = vk_entity_data(buffer->handle.value, VulkanEntityKind_Buffer); 2683 VkCommandBuffer cmd = vk_command_buffer(command); 2684 vkCmdFillBuffer(cmd, vb->buffer, offset, size, clear_word); 2685 } 2686 } 2687 2688 DEBUG_IMPORT void 2689 gpu_command_dispatch_compute(GPUCommandList command, uv3 dispatch) 2690 { 2691 assert(dispatch.x <= U16_MAX); 2692 assert(dispatch.y <= U16_MAX); 2693 assert(dispatch.z <= U16_MAX); 2694 if (command.value) { 2695 VkCommandBuffer cmd = vk_command_buffer(command); 2696 vkCmdDispatch(cmd, dispatch.x, dispatch.y, dispatch.z); 2697 } 2698 } 2699 2700 DEBUG_IMPORT void 2701 gpu_command_push_constants(GPUCommandList command, u32 offset, u32 size, void *values) 2702 { 2703 if (command.value) { 2704 VulkanCommandBuffer *vcb = vk_entity_data(command.value, VulkanEntityKind_CommandBuffer); 2705 VulkanCommandPool *vcp = vulkan_context->command_pools[vcb->timeline]; 2706 VulkanPipeline *vp = vcp->bound_pipeline; 2707 2708 assert(vp); 2709 2710 vkCmdPushConstants(vk_command_buffer(command), vp->layout, vp->stage_flags, offset, size, values); 2711 } 2712 } 2713 2714 DEBUG_IMPORT void 2715 gpu_command_timestamp(GPUCommandList command) 2716 { 2717 if (command.value) { 2718 VulkanContext *vk = vulkan_context; 2719 VulkanCommandBuffer *vcb = vk_entity_data(command.value, VulkanEntityKind_CommandBuffer); 2720 VulkanCommandPool *vcp = vk->command_pools[vcb->timeline]; 2721 2722 read_only VkPipelineStageFlags2 stage_lut[GPUTimeline_Count] = { 2723 [GPUTimeline_Graphics] = VK_PIPELINE_STAGE_2_ALL_GRAPHICS_BIT, 2724 [GPUTimeline_Compute] = VK_PIPELINE_STAGE_2_COMPUTE_SHADER_BIT, 2725 [GPUTimeline_Transfer] = -1, 2726 }; 2727 2728 VkPipelineStageFlags2 stage = stage_lut[vcb->timeline]; 2729 assert(stage != (VkPipelineStageFlags2)-1); 2730 2731 if (vcp->timestamp_counts[vcb->buffer_index] < MaxCommandBufferTimestamps) { 2732 u64 query_index = vcp->timestamp_counts[vcb->buffer_index]++; 2733 vkCmdWriteTimestamp2(vk_command_buffer(command), stage, vcp->query_pool, 2734 vcb->buffer_index * MaxCommandBufferTimestamps + query_index); 2735 } 2736 } 2737 } 2738 2739 DEBUG_IMPORT void 2740 gpu_command_wait_timeline(GPUCommandList command, GPUTimeline timeline, u64 value) 2741 { 2742 if (command.value && Between(timeline, 0, GPUTimeline_Count - 1)) { 2743 VulkanContext *vk = vulkan_context; 2744 VulkanCommandBuffer *vcb = vk_entity_data(command.value, VulkanEntityKind_CommandBuffer); 2745 2746 u32 wait_index = vk->queue_indices[timeline]; 2747 vcb->in_flight_wait_values[wait_index] = Max(value, vcb->in_flight_wait_values[wait_index]); 2748 } 2749 } 2750 2751 DEBUG_IMPORT u64 2752 gpu_command_list_end(GPUCommandList command, GPUSemaphoreSignalInfo *wait_infos, u64 wait_info_count, 2753 GPUSemaphoreSignalInfo *signal_infos, u64 signal_info_count) 2754 { 2755 u64 result = -1; 2756 if (command.value) { 2757 VulkanContext *vk = vulkan_context; 2758 VulkanCommandBuffer *vcb = vk_entity_data(command.value, VulkanEntityKind_CommandBuffer); 2759 VulkanCommandPool *vcp = vk->command_pools[vcb->timeline]; 2760 VulkanQueue *vq = vk->queues[vcb->timeline]; 2761 VulkanSemaphore *vs = &vq->timeline_semaphore; 2762 2763 vkEndCommandBuffer(vcp->buffers[vcb->buffer_index]); 2764 2765 arena_clear(vcp->arena); 2766 2767 DeferLoop(take_lock(&vq->lock, -1), release_lock(&vq->lock)) { 2768 VkCommandBufferSubmitInfo command_buffer_submit_info = { 2769 .sType = VK_STRUCTURE_TYPE_COMMAND_BUFFER_SUBMIT_INFO, 2770 .commandBuffer = vcp->buffers[vcb->buffer_index], 2771 }; 2772 2773 result = ++vs->value; 2774 2775 VkSemaphoreSubmitInfo *signal_submit_infos = push_array(vcp->arena, VkSemaphoreSubmitInfo, signal_info_count + 1); 2776 signal_submit_infos[0] = (VkSemaphoreSubmitInfo){ 2777 .sType = VK_STRUCTURE_TYPE_SEMAPHORE_SUBMIT_INFO, 2778 .semaphore = vs->semaphore, 2779 .value = result, 2780 .stageMask = vq->pipeline_stage_flags, 2781 }; 2782 2783 for EachIndex(signal_info_count, index) { 2784 VulkanSemaphore *fs = vk_entity_data(signal_infos[index].semaphore.value, VulkanEntityKind_Semaphore); 2785 signal_submit_infos[index + 1] = (VkSemaphoreSubmitInfo){ 2786 .sType = VK_STRUCTURE_TYPE_SEMAPHORE_SUBMIT_INFO, 2787 .semaphore = fs->semaphore, 2788 .value = signal_infos[index].value, 2789 .stageMask = vq->pipeline_stage_flags, 2790 }; 2791 fs->value = signal_infos[index].value; 2792 } 2793 2794 u32 wait_submit_info_count = 0; 2795 VkSemaphoreSubmitInfo *wait_submit_infos = push_array(vcp->arena, VkSemaphoreSubmitInfo, 2796 wait_info_count + VulkanQueueKind_Count + 1); 2797 for EachIndex(vk->unique_queues, index) { 2798 u32 queue_index = vk->queue_indices[index]; 2799 if (vcb->in_flight_wait_values[queue_index] > 0) { 2800 VulkanQueue *q = vk->queues[queue_index]; 2801 VkSemaphoreSubmitInfo wait_ssi = { 2802 .sType = VK_STRUCTURE_TYPE_SEMAPHORE_SUBMIT_INFO, 2803 .semaphore = q->timeline_semaphore.semaphore, 2804 .value = vcb->in_flight_wait_values[queue_index], 2805 .stageMask = q->pipeline_stage_flags, 2806 }; 2807 wait_submit_infos[wait_submit_info_count++] = wait_ssi; 2808 } 2809 } 2810 2811 for EachIndex(wait_info_count, index) { 2812 VulkanSemaphore *fs = vk_entity_data(wait_infos[index].semaphore.value, VulkanEntityKind_Semaphore); 2813 wait_submit_infos[wait_submit_info_count++] = (VkSemaphoreSubmitInfo){ 2814 .sType = VK_STRUCTURE_TYPE_SEMAPHORE_SUBMIT_INFO, 2815 .semaphore = fs->semaphore, 2816 .value = wait_infos[index].value, 2817 .stageMask = vq->pipeline_stage_flags, 2818 }; 2819 } 2820 2821 VkSubmitInfo2 submit_info = { 2822 .sType = VK_STRUCTURE_TYPE_SUBMIT_INFO_2, 2823 .commandBufferInfoCount = 1, 2824 .pCommandBufferInfos = &command_buffer_submit_info, 2825 .waitSemaphoreInfoCount = wait_submit_info_count, 2826 .pWaitSemaphoreInfos = wait_submit_infos, 2827 .signalSemaphoreInfoCount = signal_info_count + 1, 2828 .pSignalSemaphoreInfos = signal_submit_infos, 2829 }; 2830 2831 vkQueueSubmit2(vq->queue, 1, &submit_info, 0); 2832 2833 vcp->bound_pipeline = 0; 2834 atomic_store_u64(vcp->last_submission_values + vcb->buffer_index, result); 2835 } 2836 2837 release_lock(&vcp->lock); 2838 2839 vk_entity_release((VulkanEntity *)command.value); 2840 } 2841 return result; 2842 } 2843 2844 DEBUG_IMPORT void 2845 gpu_command_begin_rendering(GPUCommandList command, GPUImage *colour, GPUImage *depth, GPUImage *resolve) 2846 { 2847 if (command.value) { 2848 VkCommandBuffer cmd = vk_command_buffer(command); 2849 2850 assert((colour->width == depth->width) && (colour->height == depth->height)); 2851 2852 VulkanImage *ci = vk_entity_data(colour->image.value[0], VulkanEntityKind_Image); 2853 VulkanImage *di = vk_entity_data(depth->image.value[0], VulkanEntityKind_Image); 2854 VulkanImage *ri = 0; 2855 if (resolve) ri = vk_entity_data(resolve->image.value[0], VulkanEntityKind_Image); 2856 2857 // NOTE: Layout Transitions 2858 { 2859 u32 image_memory_barrier_count = 2; 2860 VkImageMemoryBarrier2 image_memory_barriers[3] = { 2861 { 2862 .sType = VK_STRUCTURE_TYPE_IMAGE_MEMORY_BARRIER_2, 2863 .srcStageMask = VK_PIPELINE_STAGE_2_TOP_OF_PIPE_BIT, 2864 .srcAccessMask = 0, 2865 .dstStageMask = VK_PIPELINE_STAGE_2_COLOR_ATTACHMENT_OUTPUT_BIT, 2866 .dstAccessMask = VK_ACCESS_2_COLOR_ATTACHMENT_READ_BIT|VK_ACCESS_2_COLOR_ATTACHMENT_WRITE_BIT, 2867 .oldLayout = VK_IMAGE_LAYOUT_UNDEFINED, 2868 .newLayout = VK_IMAGE_LAYOUT_COLOR_ATTACHMENT_OPTIMAL, 2869 .image = ci->image, 2870 .subresourceRange = { 2871 .aspectMask = VK_IMAGE_ASPECT_COLOR_BIT, 2872 .baseMipLevel = 0, 2873 .levelCount = 1, 2874 .baseArrayLayer = 0, 2875 .layerCount = 1, 2876 }, 2877 }, 2878 { 2879 .sType = VK_STRUCTURE_TYPE_IMAGE_MEMORY_BARRIER_2, 2880 .srcStageMask = VK_PIPELINE_STAGE_2_EARLY_FRAGMENT_TESTS_BIT|VK_PIPELINE_STAGE_2_LATE_FRAGMENT_TESTS_BIT, 2881 .srcAccessMask = 0, 2882 .dstStageMask = VK_PIPELINE_STAGE_2_EARLY_FRAGMENT_TESTS_BIT|VK_PIPELINE_STAGE_2_LATE_FRAGMENT_TESTS_BIT, 2883 .dstAccessMask = VK_ACCESS_2_DEPTH_STENCIL_ATTACHMENT_WRITE_BIT, 2884 .oldLayout = VK_IMAGE_LAYOUT_UNDEFINED, 2885 .newLayout = VK_IMAGE_LAYOUT_DEPTH_STENCIL_ATTACHMENT_OPTIMAL, 2886 .image = di->image, 2887 .subresourceRange = { 2888 .aspectMask = VK_IMAGE_ASPECT_DEPTH_BIT|VK_IMAGE_ASPECT_STENCIL_BIT, 2889 .baseMipLevel = 0, 2890 .levelCount = 1, 2891 .baseArrayLayer = 0, 2892 .layerCount = 1, 2893 }, 2894 }, 2895 }; 2896 2897 if (resolve) image_memory_barriers[image_memory_barrier_count++] = (VkImageMemoryBarrier2){ 2898 .sType = VK_STRUCTURE_TYPE_IMAGE_MEMORY_BARRIER_2, 2899 .srcStageMask = VK_PIPELINE_STAGE_2_TOP_OF_PIPE_BIT, 2900 .srcAccessMask = 0, 2901 .dstStageMask = VK_PIPELINE_STAGE_2_COLOR_ATTACHMENT_OUTPUT_BIT|VK_PIPELINE_STAGE_2_RESOLVE_BIT, 2902 .dstAccessMask = VK_ACCESS_2_COLOR_ATTACHMENT_READ_BIT|VK_ACCESS_2_COLOR_ATTACHMENT_WRITE_BIT, 2903 .oldLayout = VK_IMAGE_LAYOUT_UNDEFINED, 2904 .newLayout = VK_IMAGE_LAYOUT_COLOR_ATTACHMENT_OPTIMAL, 2905 .image = ri->image, 2906 .subresourceRange = { 2907 .aspectMask = VK_IMAGE_ASPECT_COLOR_BIT, 2908 .baseMipLevel = 0, 2909 .levelCount = 1, 2910 .baseArrayLayer = 0, 2911 .layerCount = 1, 2912 }, 2913 }; 2914 2915 VkDependencyInfo dependency_info = { 2916 .sType = VK_STRUCTURE_TYPE_DEPENDENCY_INFO, 2917 .imageMemoryBarrierCount = image_memory_barrier_count, 2918 .pImageMemoryBarriers = image_memory_barriers, 2919 }; 2920 2921 vkCmdPipelineBarrier2(cmd, &dependency_info); 2922 } 2923 2924 VkRenderingAttachmentInfo colour_attachment = { 2925 .sType = VK_STRUCTURE_TYPE_RENDERING_ATTACHMENT_INFO, 2926 .imageView = ci->view, 2927 .imageLayout = VK_IMAGE_LAYOUT_COLOR_ATTACHMENT_OPTIMAL, 2928 .resolveMode = ri ? VK_RESOLVE_MODE_AVERAGE_BIT : 0, 2929 .resolveImageView = ri ? ri->view : 0, 2930 .resolveImageLayout = ri ? VK_IMAGE_LAYOUT_COLOR_ATTACHMENT_OPTIMAL : 0, 2931 .loadOp = VK_ATTACHMENT_LOAD_OP_CLEAR, 2932 .storeOp = VK_ATTACHMENT_STORE_OP_STORE, 2933 .clearValue = {.color = {{0.0f, 0.0f, 0.0f, 0.0f}}}, 2934 }; 2935 2936 VkRenderingAttachmentInfo depth_stencil_attachment = { 2937 .sType = VK_STRUCTURE_TYPE_RENDERING_ATTACHMENT_INFO, 2938 .imageView = di->view, 2939 .imageLayout = VK_IMAGE_LAYOUT_DEPTH_STENCIL_ATTACHMENT_OPTIMAL, 2940 .loadOp = VK_ATTACHMENT_LOAD_OP_CLEAR, 2941 .storeOp = VK_ATTACHMENT_STORE_OP_STORE, 2942 .clearValue = {.depthStencil = {1.0f, 0}}, 2943 }; 2944 2945 VkRenderingInfo rendering_info = { 2946 .sType = VK_STRUCTURE_TYPE_RENDERING_INFO, 2947 .renderArea = {.offset = {0}, .extent = {colour->width, colour->height}}, 2948 .layerCount = 1, 2949 .colorAttachmentCount = 1, 2950 .pColorAttachments = &colour_attachment, 2951 .pDepthAttachment = &depth_stencil_attachment, 2952 .pStencilAttachment = &depth_stencil_attachment, 2953 }; 2954 2955 vkCmdBeginRendering(cmd, &rendering_info); 2956 } 2957 } 2958 2959 DEBUG_IMPORT void 2960 gpu_command_draw(GPUCommandList command, GPUBuffer *model) 2961 { 2962 if (command.value && model->handle.value) { 2963 VkCommandBuffer cmd = vk_command_buffer(command); 2964 VulkanBuffer *vb = vk_entity_data(model->handle.value, VulkanEntityKind_RenderModel); 2965 vkCmdBindIndexBuffer2(cmd, vb->buffer, 0, vk_index_size(vb->index_type) * model->index_count, vb->index_type); 2966 vkCmdDrawIndexed(cmd, model->index_count, 1, 0, 0, 0); 2967 } 2968 } 2969 2970 DEBUG_IMPORT void 2971 gpu_command_scissor(GPUCommandList command, u32 width, u32 height, u32 x_offset, u32 y_offset) 2972 { 2973 if (command.value) { 2974 VkCommandBuffer cmd = vk_command_buffer(command); 2975 VkRect2D scissor = {.offset = {x_offset, y_offset}, .extent = {width, height}}; 2976 vkCmdSetScissor(cmd, 0, 1, &scissor); 2977 } 2978 } 2979 2980 DEBUG_IMPORT void 2981 gpu_command_viewport(GPUCommandList command, f32 width, f32 height, f32 x_offset, f32 y_offset, f32 min_depth, f32 max_depth) 2982 { 2983 if (command.value) { 2984 VkCommandBuffer cmd = vk_command_buffer(command); 2985 VkViewport viewport = {x_offset, y_offset, width, height, min_depth, max_depth}; 2986 vkCmdSetViewport(cmd, 0, 1, &viewport); 2987 } 2988 } 2989 2990 DEBUG_IMPORT void 2991 gpu_command_end_rendering(GPUCommandList command) 2992 { 2993 if (command.value) vkCmdEndRendering(vk_command_buffer(command)); 2994 } 2995 2996 DEBUG_IMPORT void 2997 gpu_command_copy_buffer(GPUCommandList command, 2998 GPUBuffer *restrict destination, u64 destination_offset, 2999 GPUBuffer *restrict source, u64 source_offset, 3000 u64 size) 3001 { 3002 if (command.value && destination->handle.value && source->handle.value) { 3003 VulkanBuffer *db = vk_entity_data(destination->handle.value, VulkanEntityKind_Buffer); 3004 VulkanBuffer *sb = vk_entity_data(source->handle.value, VulkanEntityKind_Buffer); 3005 vk_command_copy_buffer(vk_command_buffer(command), db->buffer, destination_offset, sb->buffer, source_offset, size); 3006 } 3007 } 3008 3009 DEBUG_IMPORT u64 * 3010 gpu_read_timestamps(GPUTimeline timeline, u64 *count, Arena *arena) 3011 { 3012 u64 *result = 0; 3013 if Between(timeline, 0, GPUTimeline_Count - 1) { 3014 VulkanContext *vk = vulkan_context; 3015 VulkanCommandPool *vcp = vk->command_pools[timeline]; 3016 DeferLoop(take_lock(&vcp->lock, -1), release_lock(&vcp->lock)) 3017 { 3018 u32 index = (vcp->next_command_buffer_index - 1) % MaxCommandBuffersInFlight; 3019 *count = vcp->timestamp_counts[index]; 3020 if (*count > 0) { 3021 result = push_array(arena, u64, *count); 3022 gpu_host_wait_timeline(timeline, vcp->last_submission_values[index], -1ULL); 3023 3024 vkGetQueryPoolResults(vk->device, vcp->query_pool, index * MaxCommandBufferTimestamps, *count, 3025 *count * sizeof(u64), result, 8, VK_QUERY_RESULT_64_BIT|VK_QUERY_RESULT_WAIT_BIT); 3026 } 3027 } 3028 } 3029 return result; 3030 }