#include #include #include #include #include #include #import #import #define GPU_FRAMEBUFFER_COUNT 1 id getMetalLayer(void); id getMetalDevice(void); id getMetalQueue(void); bool gpu_transpose_mat = true; static id command_buffer = nil; static id command_encoder = nil; static id argument_encoder = nil; static id argument_buffer = nil; static id drawable; static id linear_sampler; static int argument_buffer_step; static gpu_buffer_t *current_vb; static gpu_buffer_t *current_ib; static gpu_pipeline_t *current_pipeline; static MTLViewport current_viewport; static MTLScissorRect current_scissor; static MTLRenderPassDescriptor *render_pass_desc; static bool resized = false; static MTLBlendFactor convert_blending_factor(gpu_blending_factor_t factor) { switch (factor) { case GPU_BLEND_ONE: return MTLBlendFactorOne; case GPU_BLEND_ZERO: return MTLBlendFactorZero; case GPU_BLEND_SOURCE_ALPHA: return MTLBlendFactorSourceAlpha; case GPU_BLEND_DEST_ALPHA: return MTLBlendFactorDestinationAlpha; case GPU_BLEND_INV_SOURCE_ALPHA: return MTLBlendFactorOneMinusSourceAlpha; case GPU_BLEND_INV_DEST_ALPHA: return MTLBlendFactorOneMinusDestinationAlpha; } } static MTLBlendOperation convert_blending_operation(gpu_blending_operation_t op) { switch (op) { case GPU_BLENDOP_ADD: return MTLBlendOperationAdd; } } static MTLCompareFunction convert_compare_mode(gpu_compare_mode_t compare) { switch (compare) { case GPU_COMPARE_MODE_ALWAYS: return MTLCompareFunctionAlways; case GPU_COMPARE_MODE_NEVER: return MTLCompareFunctionNever; case GPU_COMPARE_MODE_LESS: return MTLCompareFunctionLess; } } static MTLCullMode convert_cull_mode(gpu_cull_mode_t cull) { switch (cull) { case GPU_CULL_MODE_CLOCKWISE: return MTLCullModeFront; case GPU_CULL_MODE_COUNTERCLOCKWISE: return MTLCullModeBack; case GPU_CULL_MODE_NEVER: return MTLCullModeNone; } } static MTLPixelFormat convert_texture_format(gpu_texture_format_t format) { switch (format) { case GPU_TEXTURE_FORMAT_RGBA128: return MTLPixelFormatRGBA32Float; case GPU_TEXTURE_FORMAT_RGBA64: return MTLPixelFormatRGBA16Float; case GPU_TEXTURE_FORMAT_R32: return MTLPixelFormatR32Float; case GPU_TEXTURE_FORMAT_R16: return MTLPixelFormatR16Float; case GPU_TEXTURE_FORMAT_R8: return MTLPixelFormatR8Unorm; case GPU_TEXTURE_FORMAT_D32: return MTLPixelFormatDepth32Float; case GPU_TEXTURE_FORMAT_RGBA32: default: return MTLPixelFormatBGRA8Unorm; } } static int format_byte_size(gpu_texture_format_t format) { switch (format) { case GPU_TEXTURE_FORMAT_RGBA128: return 16; case GPU_TEXTURE_FORMAT_RGBA64: return 8; case GPU_TEXTURE_FORMAT_R16: return 2; case GPU_TEXTURE_FORMAT_R8: return 1; case GPU_TEXTURE_FORMAT_RGBA32: case GPU_TEXTURE_FORMAT_R32: default: return 4; } } void gpu_render_target_init2(gpu_texture_t *target, int width, int height, gpu_texture_format_t format, int framebuffer_index) { memset(target, 0, sizeof(gpu_texture_t)); target->width = width; target->height = height; target->state = GPU_TEXTURE_STATE_RENDER_TARGET; target->impl._readback = NULL; if (framebuffer_index < 0) { id device = getMetalDevice(); MTLTextureDescriptor *descriptor = [MTLTextureDescriptor new]; descriptor.textureType = MTLTextureType2D; descriptor.width = width; descriptor.height = height; descriptor.depth = 1; descriptor.pixelFormat = convert_texture_format(format); descriptor.arrayLength = 1; descriptor.mipmapLevelCount = 1; descriptor.usage = MTLTextureUsageRenderTarget | MTLTextureUsageShaderRead | MTLTextureUsageShaderWrite; descriptor.resourceOptions = MTLResourceStorageModePrivate; target->impl._tex = (__bridge_retained void *)[device newTextureWithDescriptor:descriptor]; } } void gpu_destroy(void) { } void gpu_resize_internal(int width, int height) { resized = true; } static void next_drawable() { CAMetalLayer *layer = getMetalLayer(); drawable = [layer nextDrawable]; framebuffers[framebuffer_index].impl._tex = (__bridge void *)drawable.texture; } void gpu_init_internal(int depth_buffer_bits, bool vsync) { id device = getMetalDevice(); MTLSamplerDescriptor *linear_desc = [MTLSamplerDescriptor new]; linear_desc.minFilter = MTLSamplerMinMagFilterLinear; linear_desc.magFilter = MTLSamplerMinMagFilterLinear; linear_desc.mipFilter = MTLSamplerMipFilterLinear; linear_desc.sAddressMode = MTLSamplerAddressModeRepeat; linear_desc.tAddressMode = MTLSamplerAddressModeRepeat; linear_desc.supportArgumentBuffers = true; linear_sampler = [device newSamplerStateWithDescriptor:linear_desc]; MTLArgumentDescriptor *constantsDesc = [MTLArgumentDescriptor argumentDescriptor]; constantsDesc.dataType = MTLDataTypePointer; constantsDesc.index = 0; MTLArgumentDescriptor *samplerDesc = [MTLArgumentDescriptor argumentDescriptor]; samplerDesc.dataType = MTLDataTypeSampler; samplerDesc.index = 1; MTLArgumentDescriptor *textureDesc[16]; for (int i = 0; i < 16; ++i) { textureDesc[i] = [MTLArgumentDescriptor argumentDescriptor]; textureDesc[i].dataType = MTLDataTypeTexture; textureDesc[i].index = i + 2; textureDesc[i].textureType = MTLTextureType2D; } NSArray *arguments = [NSArray arrayWithObjects:constantsDesc, samplerDesc, textureDesc[0], textureDesc[1], textureDesc[2], textureDesc[3], textureDesc[4], textureDesc[5], textureDesc[6], textureDesc[7], textureDesc[8], textureDesc[9], textureDesc[10], textureDesc[11], textureDesc[12], textureDesc[13], textureDesc[14], textureDesc[15], nil]; argument_encoder = [device newArgumentEncoderWithArguments:arguments]; argument_buffer_step = [argument_encoder encodedLength]; argument_buffer = [device newBufferWithLength:(argument_buffer_step * 2048) options:MTLResourceStorageModeShared]; gpu_create_framebuffers(depth_buffer_bits); next_drawable(); } void gpu_begin_internal(gpu_texture_t **targets, int count, gpu_texture_t *depth_buffer, unsigned flags, unsigned color, float depth) { render_pass_desc = [MTLRenderPassDescriptor renderPassDescriptor]; for (int i = 0; i < current_render_targets_count; ++i) { render_pass_desc.colorAttachments[i].texture = (__bridge id)current_render_targets[i]->impl._tex; if (flags & GPU_CLEAR_COLOR) { float red, green, blue, alpha; iron_color_components(color, &red, &green, &blue, &alpha); render_pass_desc.colorAttachments[i].loadAction = MTLLoadActionClear; render_pass_desc.colorAttachments[i].storeAction = MTLStoreActionStore; render_pass_desc.colorAttachments[i].clearColor = MTLClearColorMake(red, green, blue, alpha); } else { render_pass_desc.colorAttachments[i].loadAction = MTLLoadActionLoad; render_pass_desc.colorAttachments[i].storeAction = MTLStoreActionStore; render_pass_desc.colorAttachments[i].clearColor = MTLClearColorMake(0.0, 0.0, 0.0, 1.0); } } if (depth_buffer != NULL) { render_pass_desc.depthAttachment.texture = (__bridge id)depth_buffer->impl._tex; } if (flags & GPU_CLEAR_DEPTH) { render_pass_desc.depthAttachment.clearDepth = depth; render_pass_desc.depthAttachment.loadAction = MTLLoadActionClear; render_pass_desc.depthAttachment.storeAction = MTLStoreActionStore; } else { render_pass_desc.depthAttachment.clearDepth = 1; render_pass_desc.depthAttachment.loadAction = MTLLoadActionLoad; render_pass_desc.depthAttachment.storeAction = MTLStoreActionStore; } id queue = getMetalQueue(); if (command_buffer == nil) { command_buffer = [queue commandBuffer]; } command_encoder = [command_buffer renderCommandEncoderWithDescriptor:render_pass_desc]; } void gpu_end_internal() { [command_encoder endEncoding]; current_render_targets_count = 0; } void gpu_wait() { [command_buffer waitUntilCompleted]; } void gpu_flush() { [command_buffer commit]; gpu_wait(); id queue = getMetalQueue(); command_buffer = [queue commandBuffer]; if (gpu_in_use) { command_encoder = [command_buffer renderCommandEncoderWithDescriptor:render_pass_desc]; id pipe = (__bridge id)current_pipeline->impl._pipeline; [command_encoder setRenderPipelineState:pipe]; id depth_state = (__bridge id)current_pipeline->impl._depth; [command_encoder setDepthStencilState:depth_state]; [command_encoder setFrontFacingWinding:MTLWindingClockwise]; [command_encoder setCullMode:convert_cull_mode(current_pipeline->cull_mode)]; id vb = (__bridge id)current_vb->impl.metal_buffer; [command_encoder setVertexBuffer:vb offset:0 atIndex:0]; [command_encoder setViewport:current_viewport]; [command_encoder setScissorRect:current_scissor]; } } void gpu_present_internal() { [command_buffer presentDrawable:drawable]; [command_buffer commit]; [command_buffer waitUntilCompleted]; drawable = nil; command_buffer = nil; command_encoder = nil; if (resized) { CAMetalLayer *layer = getMetalLayer(); layer.drawableSize = CGSizeMake(iron_window_width(), iron_window_height()); for (int i = 0; i < GPU_FRAMEBUFFER_COUNT; ++i) { // gpu_texture_destroy(&framebuffers[i]); gpu_render_target_init2(&framebuffers[i], iron_window_width(), iron_window_height(), GPU_TEXTURE_FORMAT_RGBA32, i); } resized = false; } next_drawable(); } void gpu_barrier(gpu_texture_t *render_target, gpu_texture_state_t state_after) { } int gpu_max_bound_textures(void) { return 16; } void gpu_draw_internal() { id index_buffer = (__bridge id)current_ib->impl.metal_buffer; [command_encoder drawIndexedPrimitives:MTLPrimitiveTypeTriangle indexCount:current_ib->count indexType:MTLIndexTypeUInt32 indexBuffer:index_buffer indexBufferOffset:0]; } void gpu_viewport(int x, int y, int width, int height) { current_viewport.originX = x; current_viewport.originY = y; current_viewport.width = width; current_viewport.height = height; current_viewport.znear = 0.1; current_viewport.zfar = 100.0; [command_encoder setViewport:current_viewport]; } void gpu_scissor(int x, int y, int width, int height) { current_scissor.x = x; current_scissor.y = y; int target_w = current_render_targets[0]->width; int target_h = current_render_targets[0]->height; current_scissor.width = (x + width <= target_w) ? width : target_w - x; current_scissor.height = (y + height <= target_h) ? height : target_h - y; [command_encoder setScissorRect:current_scissor]; } void gpu_disable_scissor() { current_scissor.x = 0; current_scissor.y = 0; current_scissor.width = current_render_targets[0]->width; current_scissor.height = current_render_targets[0]->height; [command_encoder setScissorRect:current_scissor]; } void gpu_set_pipeline(gpu_pipeline_t *pipeline) { current_pipeline = pipeline; id pipe = (__bridge id)pipeline->impl._pipeline; [command_encoder setRenderPipelineState:pipe]; id depth_state = (__bridge id)pipeline->impl._depth; [command_encoder setDepthStencilState:depth_state]; [command_encoder setFrontFacingWinding:MTLWindingClockwise]; [command_encoder setCullMode:convert_cull_mode(pipeline->cull_mode)]; } void gpu_set_vertex_buffer(gpu_buffer_t *buffer) { current_vb = buffer; id buf = (__bridge id)buffer->impl.metal_buffer; [command_encoder setVertexBuffer:buf offset:0 atIndex:0]; } void gpu_set_index_buffer(gpu_buffer_t *buffer) { current_ib = buffer; } void gpu_get_render_target_pixels(gpu_texture_t *render_target, uint8_t *data) { gpu_flush(); // Create readback buffer if (render_target->impl._readback == NULL) { id device = getMetalDevice(); MTLTextureDescriptor *descriptor = [MTLTextureDescriptor new]; descriptor.textureType = MTLTextureType2D; descriptor.width = render_target->width; descriptor.height = render_target->height; descriptor.depth = 1; descriptor.pixelFormat = [(__bridge id)render_target->impl._tex pixelFormat]; descriptor.arrayLength = 1; descriptor.mipmapLevelCount = 1; descriptor.usage = MTLTextureUsageUnknown; descriptor.resourceOptions = MTLResourceStorageModeShared; render_target->impl._readback = (__bridge_retained void *)[device newTextureWithDescriptor:descriptor]; } // Copy render target to readback buffer id queue = getMetalQueue(); id commandBuffer = [queue commandBuffer]; id commandEncoder = [commandBuffer blitCommandEncoder]; [commandEncoder copyFromTexture:(__bridge id)render_target->impl._tex sourceSlice:0 sourceLevel:0 sourceOrigin:MTLOriginMake(0, 0, 0) sourceSize:MTLSizeMake(render_target->width, render_target->height, 1) toTexture:(__bridge id)render_target->impl._readback destinationSlice:0 destinationLevel:0 destinationOrigin:MTLOriginMake(0, 0, 0)]; [commandEncoder endEncoding]; [commandBuffer commit]; [commandBuffer waitUntilCompleted]; // Read buffer id tex = (__bridge id)render_target->impl._readback; int byte_size = format_byte_size(render_target->format); MTLRegion region = MTLRegionMake2D(0, 0, render_target->width, render_target->height); [tex getBytes:data bytesPerRow:byte_size * render_target->width fromRegion:region mipmapLevel:0]; } void gpu_set_constant_buffer(gpu_buffer_t *buffer, int offset, size_t size) { id buf = (__bridge id)buffer->impl.metal_buffer; int i = constant_buffer_index; [argument_encoder setArgumentBuffer:argument_buffer offset:argument_buffer_step * i]; [argument_encoder setBuffer:buf offset:offset atIndex:0]; [argument_encoder setSamplerState:linear_sampler atIndex:1]; [command_encoder setVertexBuffer:argument_buffer offset:argument_buffer_step * i atIndex:1]; [command_encoder setFragmentBuffer:argument_buffer offset:argument_buffer_step * i atIndex:1]; [command_encoder useResource:buf usage:MTLResourceUsageRead stages:MTLRenderStageVertex|MTLRenderStageFragment]; } void gpu_set_texture(int unit, gpu_texture_t *texture) { id tex = (__bridge id)texture->impl._tex; int i = constant_buffer_index; [argument_encoder setArgumentBuffer:argument_buffer offset:argument_buffer_step * i]; [argument_encoder setTexture:tex atIndex:unit + 2]; [command_encoder useResource:tex usage:MTLResourceUsageRead stages:MTLRenderStageVertex|MTLRenderStageFragment]; } void gpu_pipeline_init(gpu_pipeline_t *pipeline) { memset(&pipeline->impl, 0, sizeof(pipeline->impl)); gpu_internal_pipeline_init(pipeline); } void gpu_pipeline_destroy(gpu_pipeline_t *pipeline) { id pipe = (__bridge_transfer id)pipeline->impl._pipeline; pipe = nil; pipeline->impl._pipeline = NULL; id depth_state = (__bridge_transfer id)pipeline->impl._depth; depth_state = nil; pipeline->impl._depth = NULL; } void gpu_pipeline_compile(gpu_pipeline_t *pipeline) { id device = getMetalDevice(); NSError *error = nil; id library = [device newLibraryWithSource:[[NSString alloc] initWithBytes:pipeline->vertex_shader->impl.source length:pipeline->vertex_shader->impl.length encoding:NSUTF8StringEncoding] options:nil error:&error]; if (library == nil) { iron_error("%s", error.localizedDescription.UTF8String); } pipeline->vertex_shader->impl.mtl_function = (__bridge_retained void *)[library newFunctionWithName:[NSString stringWithCString:pipeline->vertex_shader->impl.name encoding:NSUTF8StringEncoding]]; assert(pipeline->vertex_shader->impl.mtl_function); pipeline->fragment_shader->impl.mtl_function = (__bridge_retained void *)[library newFunctionWithName:[NSString stringWithCString:pipeline->fragment_shader->impl.name encoding:NSUTF8StringEncoding]]; assert(pipeline->fragment_shader->impl.mtl_function); MTLRenderPipelineDescriptor *renderPipelineDesc = [[MTLRenderPipelineDescriptor alloc] init]; renderPipelineDesc.vertexFunction = (__bridge id)pipeline->vertex_shader->impl.mtl_function; renderPipelineDesc.fragmentFunction = (__bridge id)pipeline->fragment_shader->impl.mtl_function; for (int i = 0; i < pipeline->color_attachment_count; ++i) { renderPipelineDesc.colorAttachments[i].pixelFormat = convert_texture_format(pipeline->color_attachment[i]); renderPipelineDesc.colorAttachments[i].blendingEnabled = pipeline->blend_source != GPU_BLEND_ONE || pipeline->blend_destination != GPU_BLEND_ZERO || pipeline->alpha_blend_source != GPU_BLEND_ONE || pipeline->alpha_blend_destination != GPU_BLEND_ZERO; renderPipelineDesc.colorAttachments[i].sourceRGBBlendFactor = convert_blending_factor(pipeline->blend_source); renderPipelineDesc.colorAttachments[i].destinationRGBBlendFactor = convert_blending_factor(pipeline->blend_destination); renderPipelineDesc.colorAttachments[i].rgbBlendOperation = convert_blending_operation(pipeline->blend_operation); renderPipelineDesc.colorAttachments[i].sourceAlphaBlendFactor = convert_blending_factor(pipeline->alpha_blend_source); renderPipelineDesc.colorAttachments[i].destinationAlphaBlendFactor = convert_blending_factor(pipeline->alpha_blend_destination); renderPipelineDesc.colorAttachments[i].alphaBlendOperation = convert_blending_operation(pipeline->alpha_blend_operation); renderPipelineDesc.colorAttachments[i].writeMask = (pipeline->color_write_mask_red[i] ? MTLColorWriteMaskRed : 0) | (pipeline->color_write_mask_green[i] ? MTLColorWriteMaskGreen : 0) | (pipeline->color_write_mask_blue[i] ? MTLColorWriteMaskBlue : 0) | (pipeline->color_write_mask_alpha[i] ? MTLColorWriteMaskAlpha : 0); } renderPipelineDesc.depthAttachmentPixelFormat = pipeline->depth_attachment_bits > 0 ? MTLPixelFormatDepth32Float : MTLPixelFormatInvalid; float offset = 0; MTLVertexDescriptor *vertexDescriptor = [[MTLVertexDescriptor alloc] init]; for (int i = 0; i < pipeline->input_layout->size; ++i) { vertexDescriptor.attributes[i].bufferIndex = 0; vertexDescriptor.attributes[i].offset = offset; offset += gpu_vertex_data_size(pipeline->input_layout->elements[i].data); switch (pipeline->input_layout->elements[i].data) { case GPU_VERTEX_DATA_F32_1X: vertexDescriptor.attributes[i].format = MTLVertexFormatFloat; break; case GPU_VERTEX_DATA_F32_2X: vertexDescriptor.attributes[i].format = MTLVertexFormatFloat2; break; case GPU_VERTEX_DATA_F32_3X: vertexDescriptor.attributes[i].format = MTLVertexFormatFloat3; break; case GPU_VERTEX_DATA_F32_4X: vertexDescriptor.attributes[i].format = MTLVertexFormatFloat4; break; case GPU_VERTEX_DATA_I16_2X_NORM: vertexDescriptor.attributes[i].format = MTLVertexFormatShort2Normalized; break; case GPU_VERTEX_DATA_I16_4X_NORM: vertexDescriptor.attributes[i].format = MTLVertexFormatShort4Normalized; break; } } vertexDescriptor.layouts[0].stride = offset; vertexDescriptor.layouts[0].stepFunction = MTLVertexStepFunctionPerVertex; renderPipelineDesc.vertexDescriptor = vertexDescriptor; NSError *errors = nil; MTLRenderPipelineReflection *reflection = nil; pipeline->impl._pipeline = (__bridge_retained void *)[ device newRenderPipelineStateWithDescriptor:renderPipelineDesc options:MTLPipelineOptionBufferTypeInfo reflection:&reflection error:&errors]; MTLDepthStencilDescriptor *depthStencilDescriptor = [MTLDepthStencilDescriptor new]; depthStencilDescriptor.depthCompareFunction = convert_compare_mode(pipeline->depth_mode); depthStencilDescriptor.depthWriteEnabled = pipeline->depth_write; pipeline->impl._depth = (__bridge_retained void *)[device newDepthStencilStateWithDescriptor:depthStencilDescriptor]; } void gpu_shader_destroy(gpu_shader_t *shader) { id function = (__bridge_transfer id)shader->impl.mtl_function; function = nil; shader->impl.mtl_function = NULL; } void gpu_shader_init(gpu_shader_t *shader, const void *data, size_t length, gpu_shader_type_t type) { shader->impl.name[0] = 0; const char *source = data; for (int i = 3; i < length; ++i) { // //> if (source[i] == '\n') { shader->impl.name[i - 3] = 0; break; } shader->impl.name[i - 3] = source[i]; } shader->impl.source = data; shader->impl.length = length; } void gpu_texture_init_from_bytes(gpu_texture_t *texture, void *data, int width, int height, gpu_texture_format_t format) { texture->width = width; texture->height = height; texture->format = format; texture->state = GPU_TEXTURE_STATE_SHADER_RESOURCE; MTLPixelFormat mtlformat = convert_texture_format(format); if (mtlformat == MTLPixelFormatBGRA8Unorm) { mtlformat = MTLPixelFormatRGBA8Unorm; } MTLTextureDescriptor *descriptor = [MTLTextureDescriptor texture2DDescriptorWithPixelFormat:mtlformat width:width height:height mipmapped:NO]; descriptor.textureType = MTLTextureType2D; descriptor.width = width; descriptor.height = height; descriptor.depth = 1; descriptor.pixelFormat = mtlformat; descriptor.arrayLength = 1; descriptor.mipmapLevelCount = 1; descriptor.usage = MTLTextureUsageShaderRead; // MTLTextureUsageShaderWrite id device = getMetalDevice(); id tex = [device newTextureWithDescriptor:descriptor]; texture->impl._tex = (__bridge_retained void *)tex; [tex replaceRegion:MTLRegionMake2D(0, 0, width, height) mipmapLevel:0 slice:0 withBytes:data bytesPerRow:width * format_byte_size(format) bytesPerImage:width * format_byte_size(format) * height]; } void gpu_texture_destroy(gpu_texture_t *target) { id tex = (__bridge_transfer id)target->impl._tex; tex = nil; target->impl._tex = NULL; id readback = (__bridge_transfer id)target->impl._readback; readback = nil; target->impl._readback = NULL; } void gpu_render_target_init(gpu_texture_t *target, int width, int height, gpu_texture_format_t format) { gpu_render_target_init2(target, width, height, format, -1); target->width = width; target->height = height; target->state = GPU_TEXTURE_STATE_RENDER_TARGET; } void gpu_vertex_buffer_init(gpu_buffer_t *buffer, int count, gpu_vertex_structure_t *structure) { memset(&buffer->impl, 0, sizeof(buffer->impl)); buffer->count = count; for (int i = 0; i < structure->size; ++i) { gpu_vertex_element_t element = structure->elements[i]; buffer->stride += gpu_vertex_data_size(element.data); } id device = getMetalDevice(); MTLResourceOptions options = MTLResourceCPUCacheModeWriteCombined; options |= MTLResourceStorageModeShared; id buf = [device newBufferWithLength:count * buffer->stride options:options]; buffer->impl.metal_buffer = (__bridge_retained void *)buf; } float *gpu_vertex_buffer_lock(gpu_buffer_t *buf) { id buffer = (__bridge id)buf->impl.metal_buffer; float *floats = (float *)[buffer contents]; return floats; } void gpu_vertex_buffer_unlock(gpu_buffer_t *buf) { } void gpu_constant_buffer_init(gpu_buffer_t *buffer, int size) { buffer->count = size; buffer->data = NULL; buffer->impl.metal_buffer = (__bridge_retained void *)[getMetalDevice() newBufferWithLength:size options:MTLResourceOptionCPUCacheModeDefault]; } void gpu_constant_buffer_destroy(gpu_buffer_t *buffer) { id buf = (__bridge_transfer id)buffer->impl.metal_buffer; buf = nil; buffer->impl.metal_buffer = NULL; } void gpu_constant_buffer_lock(gpu_buffer_t *buffer, int start, int count) { id buf = (__bridge id)buffer->impl.metal_buffer; uint8_t *data = (uint8_t *)[buf contents]; buffer->data = &data[start]; } void gpu_constant_buffer_unlock(gpu_buffer_t *buffer) { } void gpu_index_buffer_init(gpu_buffer_t *buffer, int indexCount) { buffer->count = indexCount; id device = getMetalDevice(); MTLResourceOptions options = MTLResourceCPUCacheModeWriteCombined; options |= MTLResourceStorageModeShared; buffer->impl.metal_buffer = (__bridge_retained void *)[device newBufferWithLength:sizeof(uint32_t) * indexCount options:options]; } void gpu_buffer_destroy(gpu_buffer_t *buffer) { id buf = (__bridge_transfer id)buffer->impl.metal_buffer; buf = nil; buffer->impl.metal_buffer = NULL; } void *gpu_index_buffer_lock(gpu_buffer_t *buffer) { id metal_buffer = (__bridge id)buffer->impl.metal_buffer; uint8_t *data = (uint8_t *)[metal_buffer contents]; return data; } void gpu_index_buffer_unlock(gpu_buffer_t *buffer) { } typedef struct inst { iron_matrix4x4_t m; int i; } inst_t; static gpu_raytrace_acceleration_structure_t *accel; static gpu_raytrace_pipeline_t *pipeline; static gpu_texture_t *output = NULL; static gpu_buffer_t *constant_buf; static id _raytracing_pipeline; static NSMutableArray *_primitive_accels; static id _instance_accel; static dispatch_semaphore_t _semaphore; static gpu_texture_t *_texpaint0; static gpu_texture_t *_texpaint1; static gpu_texture_t *_texpaint2; static gpu_texture_t *_texenv; static gpu_texture_t *_texsobol; static gpu_texture_t *_texscramble; static gpu_texture_t *_texrank; static gpu_buffer_t *vb[16]; static gpu_buffer_t *vb_last[16]; static gpu_buffer_t *ib[16]; static int vb_count = 0; static int vb_count_last = 0; static inst_t instances[1024]; static int instances_count = 0; void gpu_raytrace_pipeline_init(gpu_raytrace_pipeline_t *pipeline, void *ray_shader, int ray_shader_size, gpu_buffer_t *constant_buffer) { id device = getMetalDevice(); if (!device.supportsRaytracing) return; constant_buf = constant_buffer; NSError *error = nil; id library = [device newLibraryWithSource:[[NSString alloc] initWithBytes:ray_shader length:ray_shader_size encoding:NSUTF8StringEncoding] options:nil error:&error]; if (library == nil) { iron_error("%s", error.localizedDescription.UTF8String); } MTLComputePipelineDescriptor *descriptor = [[MTLComputePipelineDescriptor alloc] init]; descriptor.computeFunction = [library newFunctionWithName:@"raytracingKernel"]; descriptor.threadGroupSizeIsMultipleOfThreadExecutionWidth = YES; _raytracing_pipeline = [device newComputePipelineStateWithDescriptor:descriptor options:0 reflection:nil error:&error]; _semaphore = dispatch_semaphore_create(2); } void gpu_raytrace_pipeline_destroy(gpu_raytrace_pipeline_t *pipeline) { } bool gpu_raytrace_supported() { id device = getMetalDevice(); return device.supportsRaytracing; } id create_acceleration_sctructure(MTLAccelerationStructureDescriptor *descriptor) { id device = getMetalDevice(); id queue = getMetalQueue(); MTLAccelerationStructureSizes accel_sizes = [device accelerationStructureSizesWithDescriptor:descriptor]; id acceleration_structure = [device newAccelerationStructureWithSize:accel_sizes.accelerationStructureSize]; id scratch_buffer = [device newBufferWithLength:accel_sizes.buildScratchBufferSize options:MTLResourceStorageModePrivate]; id command_buffer = [queue commandBuffer]; id command_encoder = [command_buffer accelerationStructureCommandEncoder]; id compacteds_size_buffer = [device newBufferWithLength:sizeof(uint32_t) options:MTLResourceStorageModeShared]; [command_encoder buildAccelerationStructure:acceleration_structure descriptor:descriptor scratchBuffer:scratch_buffer scratchBufferOffset:0]; [command_encoder writeCompactedAccelerationStructureSize:acceleration_structure toBuffer:compacteds_size_buffer offset:0]; [command_encoder endEncoding]; [command_buffer commit]; [command_buffer waitUntilCompleted]; uint32_t compacted_size = *(uint32_t *)compacteds_size_buffer.contents; id compacted_acceleration_structure = [device newAccelerationStructureWithSize:compacted_size]; command_buffer = [queue commandBuffer]; command_encoder = [command_buffer accelerationStructureCommandEncoder]; [command_encoder copyAndCompactAccelerationStructure:acceleration_structure toAccelerationStructure:compacted_acceleration_structure]; [command_encoder endEncoding]; [command_buffer commit]; return compacted_acceleration_structure; } void gpu_raytrace_acceleration_structure_init(gpu_raytrace_acceleration_structure_t *accel) { vb_count = 0; instances_count = 0; } void gpu_raytrace_acceleration_structure_add(gpu_raytrace_acceleration_structure_t *accel, gpu_buffer_t *_vb, gpu_buffer_t *_ib, iron_matrix4x4_t _transform) { int vb_i = -1; for (int i = 0; i < vb_count; ++i) { if (_vb == vb[i]) { vb_i = i; break; } } if (vb_i == -1) { vb_i = vb_count; vb[vb_count] = _vb; ib[vb_count] = _ib; vb_count++; } inst_t inst = { .i = vb_i, .m = _transform }; instances[instances_count] = inst; instances_count++; } void _gpu_raytrace_acceleration_structure_destroy_bottom(gpu_raytrace_acceleration_structure_t *accel) { // for (int i = 0; i < vb_count_last; ++i) { // } _primitive_accels = nil; } void _gpu_raytrace_acceleration_structure_destroy_top(gpu_raytrace_acceleration_structure_t *accel) { _instance_accel = nil; } void gpu_raytrace_acceleration_structure_build(gpu_raytrace_acceleration_structure_t *accel, gpu_buffer_t *_vb_full, gpu_buffer_t *_ib_full) { bool build_bottom = false; for (int i = 0; i < 16; ++i) { if (vb_last[i] != vb[i]) { build_bottom = true; } vb_last[i] = vb[i]; } if (vb_count_last > 0) { if (build_bottom) { _gpu_raytrace_acceleration_structure_destroy_bottom(accel); } _gpu_raytrace_acceleration_structure_destroy_top(accel); } vb_count_last = vb_count; if (vb_count == 0) { return; } id device = getMetalDevice(); if (!device.supportsRaytracing) { return; } MTLResourceOptions options = MTLResourceStorageModeShared; MTLAccelerationStructureTriangleGeometryDescriptor *descriptor = [MTLAccelerationStructureTriangleGeometryDescriptor descriptor]; descriptor.indexType = MTLIndexTypeUInt32; descriptor.indexBuffer = (__bridge id)ib[0]->impl.metal_buffer; descriptor.vertexBuffer = (__bridge id)vb[0]->impl.metal_buffer; descriptor.vertexStride = vb[0]->stride; descriptor.triangleCount = ib[0]->count / 3; descriptor.vertexFormat = MTLAttributeFormatShort4Normalized; MTLPrimitiveAccelerationStructureDescriptor *accel_descriptor = [MTLPrimitiveAccelerationStructureDescriptor descriptor]; accel_descriptor.geometryDescriptors = @[ descriptor ]; id acceleration_structure = create_acceleration_sctructure(accel_descriptor); _primitive_accels = [[NSMutableArray alloc] init]; [_primitive_accels addObject:acceleration_structure]; id instance_buffer = [device newBufferWithLength:sizeof(MTLAccelerationStructureInstanceDescriptor) * 1 options:options]; MTLAccelerationStructureInstanceDescriptor *instance_descriptors = (MTLAccelerationStructureInstanceDescriptor *)instance_buffer.contents; instance_descriptors[0].accelerationStructureIndex = 0; instance_descriptors[0].options = MTLAccelerationStructureInstanceOptionOpaque; instance_descriptors[0].mask = 1; instance_descriptors[0].transformationMatrix.columns[0] = MTLPackedFloat3Make(instances[0].m.m[0], instances[0].m.m[1], instances[0].m.m[2]); instance_descriptors[0].transformationMatrix.columns[1] = MTLPackedFloat3Make(instances[0].m.m[4], instances[0].m.m[5], instances[0].m.m[6]); instance_descriptors[0].transformationMatrix.columns[2] = MTLPackedFloat3Make(instances[0].m.m[8], instances[0].m.m[9], instances[0].m.m[10]); instance_descriptors[0].transformationMatrix.columns[3] = MTLPackedFloat3Make(instances[0].m.m[12], instances[0].m.m[13], instances[0].m.m[14]); MTLInstanceAccelerationStructureDescriptor *inst_accel_descriptor = [MTLInstanceAccelerationStructureDescriptor descriptor]; inst_accel_descriptor.instancedAccelerationStructures = _primitive_accels; inst_accel_descriptor.instanceCount = 1; inst_accel_descriptor.instanceDescriptorBuffer = instance_buffer; _instance_accel = create_acceleration_sctructure(inst_accel_descriptor); } void gpu_raytrace_acceleration_structure_destroy(gpu_raytrace_acceleration_structure_t *accel) {} void gpu_raytrace_set_textures(gpu_texture_t *texpaint0, gpu_texture_t *texpaint1, gpu_texture_t *texpaint2, gpu_texture_t *texenv, gpu_texture_t *texsobol, gpu_texture_t *texscramble, gpu_texture_t *texrank) { _texpaint0 = texpaint0; _texpaint1 = texpaint1; _texpaint2 = texpaint2; _texenv = texenv; _texsobol = texsobol; _texscramble = texscramble; _texrank = texrank; } void gpu_raytrace_set_acceleration_structure(gpu_raytrace_acceleration_structure_t *_accel) { accel = _accel; } void gpu_raytrace_set_pipeline(gpu_raytrace_pipeline_t *_pipeline) { pipeline = _pipeline; } void gpu_raytrace_set_target(gpu_texture_t *_output) { output = _output; } void gpu_raytrace_dispatch_rays() { id device = getMetalDevice(); if (!device.supportsRaytracing) return; dispatch_semaphore_wait(_semaphore, DISPATCH_TIME_FOREVER); id queue = getMetalQueue(); id command_buffer = [queue commandBuffer]; __block dispatch_semaphore_t sem = _semaphore; [command_buffer addCompletedHandler:^(id buffer) { dispatch_semaphore_signal(sem); }]; NSUInteger width = output->width; NSUInteger height = output->height; MTLSize threads_per_threadgroup = MTLSizeMake(8, 8, 1); MTLSize threadgroups = MTLSizeMake((width + threads_per_threadgroup.width - 1) / threads_per_threadgroup.width, (height + threads_per_threadgroup.height - 1) / threads_per_threadgroup.height, 1); id compute_encoder = [command_buffer computeCommandEncoder]; [compute_encoder setBuffer:(__bridge id)constant_buf->impl.metal_buffer offset:0 atIndex:0]; [compute_encoder setAccelerationStructure:_instance_accel atBufferIndex:1]; [compute_encoder setBuffer: (__bridge id)ib[0]->impl.metal_buffer offset:0 atIndex:2]; [compute_encoder setBuffer: (__bridge id)vb[0]->impl.metal_buffer offset:0 atIndex:3]; [compute_encoder setTexture:(__bridge id)output->impl._tex atIndex:0]; [compute_encoder setTexture:(__bridge id)_texpaint0->impl._tex atIndex:1]; [compute_encoder setTexture:(__bridge id)_texpaint1->impl._tex atIndex:2]; [compute_encoder setTexture:(__bridge id)_texpaint2->impl._tex atIndex:3]; [compute_encoder setTexture:(__bridge id)_texenv->impl._tex atIndex:4]; [compute_encoder setTexture:(__bridge id)_texsobol->impl._tex atIndex:5]; [compute_encoder setTexture:(__bridge id)_texscramble->impl._tex atIndex:6]; [compute_encoder setTexture:(__bridge id)_texrank->impl._tex atIndex:7]; for (id primitive_accel in _primitive_accels) { [compute_encoder useResource:primitive_accel usage:MTLResourceUsageRead]; } [compute_encoder setComputePipelineState:_raytracing_pipeline]; [compute_encoder dispatchThreadgroups:threadgroups threadsPerThreadgroup:threads_per_threadgroup]; [compute_encoder endEncoding]; [command_buffer commit]; }