From f34550e692714000bfcf82764f83cf3f00f68d40 Mon Sep 17 00:00:00 2001 From: Albin Chaboissier Date: Thu, 17 Sep 2026 20:52:09 +0200 Subject: [PATCH] working buffer compactor --- shaders/Makefile | 2 +- shaders/compaction.slang | 34 +++-- shaders/compaction.spv | Bin 0 -> 4264 bytes src/main.rs | 92 ++++++++++++ src/voxel_cache/indirect_buffer.rs | 224 +++++++++++++++++++++++++++-- 5 files changed, 326 insertions(+), 26 deletions(-) create mode 100644 shaders/compaction.spv diff --git a/shaders/Makefile b/shaders/Makefile index 888e342..4d59aed 100644 --- a/shaders/Makefile +++ b/shaders/Makefile @@ -1,4 +1,4 @@ -all: voxel.spv +all: voxel.spv compaction.spv %.spv: %.slang slangc $< -O3 -fvk-use-entrypoint-name -target spirv -o $@ diff --git a/shaders/compaction.slang b/shaders/compaction.slang index e0fc62e..35828f2 100644 --- a/shaders/compaction.slang +++ b/shaders/compaction.slang @@ -1,13 +1,13 @@ [[vk::binding(0, 0)]] RWStructuredBuffer count_buffer; +[[vk::binding(1, 0)]] +RWStructuredBuffer index_buffer; [[vk::binding(0, 1)]] RWStructuredBuffer reduced_buffer; [[vk::binding(1, 1)]] RWStructuredBuffer sum_buffer; -[[vk::binding(2, 1)]] -RWStructuredBuffer compaction_buffer; groupshared uint32_t local_data[256 * 2]; static uint32_t THREAD_WIDTH = 256; @@ -50,7 +50,7 @@ void block_sum( var width : uint32_t = 2; while (width <= DATA_WIDTH) { - let dest_index = width * (thread_index + 1) - 1; + let dest_index = width * (local_thread_index + 1) - 1; let get_index = dest_index - (width / 2); // println!("{}, {}", get_index, dest_index); if (dest_index < DATA_WIDTH) @@ -61,10 +61,14 @@ void block_sum( GroupMemoryBarrierWithGroupSync(); } + // Write to reduced buffer + reduced_buffer[workgroup_id.x] = local_data[DATA_WIDTH - 1]; + // reduced_buffer[workgroup_id.x] = local_data[DATA_WIDTH - 1]; + local_data[DATA_WIDTH - 1] = 0; while (width >= 2) { - let dest_index = width * (thread_index + 1) - 1; + let dest_index = width * (local_thread_index + 1) - 1; let get_index = dest_index - (width / 2); // println!("{}, {}", get_index, dest_index); if (dest_index < DATA_WIDTH) @@ -81,9 +85,6 @@ void block_sum( // Dump back to sum buffer sum_buffer[2 * thread_index] = local_data[2 * local_thread_index]; sum_buffer[2 * thread_index + 1] = local_data[2 * local_thread_index + 1]; - - // Write to reduced buffer - reduced_buffer[workgroup_id.x] = local_data[DATA_WIDTH - 1]; } [numthreads(1, 1, 1)] @@ -107,7 +108,7 @@ void linear_reduced_sum( [numthreads(256, 1, 1)] [shader("compute")] -void uniform_add( +void scatter( uint32_t3 workgroup_id: SV_GroupID, uint32_t3 local_thread_id: SV_GroupThreadID, uint32_t3 global_thread_id: SV_DispatchThreadID) @@ -116,9 +117,20 @@ void uniform_add( let local_thread_index = local_thread_id.x; // Gather - let reduced_value = reduced_buffer[global_thread_id.x]; + let reduced_value = reduced_buffer[workgroup_id.x]; // Apply - sum_buffer[thread_index * 2] += reduced_value; - sum_buffer[thread_index * 2 + 1] += reduced_value; + let sum1 = sum_buffer[thread_index * 2] + reduced_value; + let sum2 = sum_buffer[thread_index * 2 + 1] + reduced_value; + + // Scatter index to compact index buffer if predicate is true + if (count_buffer[thread_index * 2] != 0) + { + index_buffer[sum1] = thread_index * 2; + } + + if (count_buffer[thread_index * 2 + 1] != 0) + { + index_buffer[sum2] = thread_index * 2 + 1; + } } diff --git a/shaders/compaction.spv b/shaders/compaction.spv new file mode 100644 index 0000000000000000000000000000000000000000..7e9b2a0699b9ef5581b6a5f4e8e323a687cc662b GIT binary patch literal 4264 zcmZ{mhjWx=5XO&?P=!#%0x=<|DE5LVf`BL}7K(yB9LWKZkm%)3P*D*Pd+#0lILb%Jj;% z&D;AnZQR;7+8C)<_tg4!HFxi>)lZoHL)Fnyb7od1s*1eM%8W`!iFXZ+^zZ8%ZSKcz zLp;@v7@vepMy4R0m8peqDtKscxK^$A)oTOI{@TD;vN@gfW8O4mw7=SD)apge)SHgq zn%0{Ia)#N6K&lv_V`uF-1CDVzcGme|>qLbP*qVXwz~De*Z&}Y?T6y)3nfU6> z;o-sIJq2O*`hQ&7t1}G_57Z8xxXJt5|HJJDZ29!;8Hl#;rF~{;%eD5x((Xi?<9r*K zT)uqwjZ_!=^^U2cp0S>B_Tmq#_hE=uATel(y8y? zL*H}I;`!8^i|$7JmX-EAa2v7!dp_DT?shl!sc7%lZ~lV9KjvG+DLL|dSILjId8eV} zNSR0V)6w2%zWjUPLTtG=U!Qq$FAw^)7nQbW(r*{oFZIqEGmG(*IinH=ZHTd*f zKU}T_%flMP9<{@zcUjjvQg2IyO9^1Oc!tV{>@as(B_eQX>@Vf!5ua)18 z*!sJXR(@~7HfHbMsV%?hUHvm$xj)L{GJZ4K+0?)AOYogHA)YbEe~)fKekJ$T5~GUR zx1r5d|5Wb8?O^kJ5%+sD+Rt|~rfnVb)cH+q0sGtZ3r}0Y?TB+4&s%K+%jps%+}#1z z@BHRzi}mFl?*x0VJ!BWQgMCNeEN{C5>d;VS3u!RJ|H+1-0f zUpToB+fPo6-H+(AH*w~C0PLLBi?6Trc~55F3DzIG3N|-0?gH!EiO7jDeeOr@RcC(R zJae6FXWR#SRa+xFB){3A{<4N|Jb=^?cW8HsQAO?^w6)YflsmK+Y`#0_4h^FHW@HhmPTVvsEKUlvUSxY;$hry}sZ#LYHfUV(Pr1nE# zYa2`Lhr#--Z7uD;_*c>yoG10_VEeRQ`X2?$hq3Hg1FYY3*++Q}Z#mb*=UHRf$45$E zFIs*Np#9|6*g-^}y@@l|qhRNp%u{D?^k+X`1lzy6 zZ>@0s64;nrs+}RcJA<}o;`9+@*;8`V?zOx14}*Yo#Q*>R literal 0 HcmV?d00001 diff --git a/src/main.rs b/src/main.rs index a0afb0a..73bef09 100644 --- a/src/main.rs +++ b/src/main.rs @@ -2,6 +2,7 @@ #![feature(float_algebraic)] use std::collections::HashMap; +use std::collections::HashSet; use std::ops::Div; use std::sync::Arc; use std::sync::mpsc::sync_channel; @@ -68,6 +69,7 @@ use crate::voxel_cache::data::ColorBytes; use crate::voxel_cache::data::DestinationElement; use crate::voxel_cache::data::LocationPoolElement; use crate::voxel_cache::data::StructurePoolElement; +use crate::voxel_cache::indirect_buffer::BufferCompactor; use crate::voxel_cache::producer_interface::CacheProducerInterface; use crate::voxel_cache::producer_interface::CacheRequest; @@ -173,6 +175,96 @@ impl State .await .unwrap(); + let mut encoder = device.create_command_encoder(&Default::default()); + // -------------------------- + let mut count_buffer = vec![0u32; 65536]; + let mut check_buffer = vec![]; + // Spread random numbers throughout the buffer + let mut picked = HashSet::new(); + for i in 1..=1000 + { + let mut r = rand::random_range(0..(count_buffer.len())); + while picked.contains(&r) + { + r = rand::random_range(0..(count_buffer.len())); + } + + picked.insert(r); + check_buffer.push(r); + count_buffer[r] = i; + } + + check_buffer.sort(); + + count_buffer + .iter() + .enumerate() + .filter(|(_, x)| **x != 0) + .take(100) + .for_each(|(x, y)| println!("{}, {}", x, y)); + + let buffer_compactor = BufferCompactor::new(&device, count_buffer.len()); + let gpu_count_buffer = device.create_buffer_init(&wgpu::util::BufferInitDescriptor { + label: None, + contents: bytemuck::cast_slice(count_buffer.as_slice()), + usage: BufferUsages::STORAGE, + }); + let gpu_index_buffer = device.create_buffer_init(&wgpu::util::BufferInitDescriptor { + label: None, + contents: bytemuck::cast_slice(vec![0u32; count_buffer.len()].as_slice()), + usage: BufferUsages::STORAGE | BufferUsages::COPY_SRC, + }); + + buffer_compactor.compact_buffer( + &device, + &mut encoder, + &gpu_count_buffer, + &gpu_index_buffer, + ); + + queue.submit([encoder.finish()]); + // -------------------------- + + wgpu::util::DownloadBuffer::read_buffer( + &device, + &queue, + &buffer_compactor.reduced_buffer.slice(..), + //&buffer_compactor.total_sum_buffer(), + |result| { + let result = result.unwrap(); + let indices: Vec = result + .to_vec() + .chunks(4) + .map(|x| u32::from_ne_bytes([x[0], x[1], x[2], x[3]])) + .collect(); + + println!("Total sum {:?}", indices); + }, + ); + + wgpu::util::DownloadBuffer::read_buffer( + &device, + &queue, + &gpu_index_buffer.slice(..), + |result| { + let result = result.unwrap(); + let indices: Vec = result + .to_vec() + .chunks(4) + .map(|x| u32::from_ne_bytes([x[0], x[1], x[2], x[3]])) + .collect(); + + println!("{:?}", &indices[..1010]); + }, + ); + for _ in 0..16 + { + device.poll(wgpu::wgt::PollType::Poll); + } + + println!("Check buffer : {:?}", check_buffer); + panic!(); + let size = window.inner_size(); let surface = instance.create_surface(window.clone()).unwrap(); diff --git a/src/voxel_cache/indirect_buffer.rs b/src/voxel_cache/indirect_buffer.rs index ec02f6b..150eb64 100644 --- a/src/voxel_cache/indirect_buffer.rs +++ b/src/voxel_cache/indirect_buffer.rs @@ -1,44 +1,240 @@ -use wgpu::{Buffer, BufferUsages, Device, util::DeviceExt}; +use wgpu::BindGroup; +use wgpu::BindGroupLayout; +use wgpu::Buffer; +use wgpu::BufferSlice; +use wgpu::BufferUsages; +use wgpu::CommandEncoder; +use wgpu::ComputePipeline; +use wgpu::Device; +use wgpu::ShaderStages; +use wgpu::util::DeviceExt; -pub struct BufferCompactor +pub struct NonZeroBufferCompactor { size: usize, block_count: usize, reduced_buffer: Buffer, sum_buffer: Buffer, - compaction_buffer: Buffer, + count_buffer_bg_layout: BindGroupLayout, + compaction_bg: BindGroup, + + reduce_sum_pipeline: ComputePipeline, + partial_sum_pipeline: ComputePipeline, + scatter_pipeline: ComputePipeline, } -impl BufferCompactor +impl NonZeroBufferCompactor { const THREAD_COUNT: usize = 256; pub fn new(device: &Device, size: usize) -> Self { - let block_count = size.div_ceil(size); + let block_count = size.div_ceil(Self::THREAD_COUNT * 2); let reduced_buffer = device.create_buffer_init(&wgpu::util::BufferInitDescriptor { label: Some("indirect_buffer_reduced"), - contents: bytemuck::cast_slice(vec![0; block_count].as_slice()), - usage: BufferUsages::STORAGE, + contents: bytemuck::cast_slice(vec![0; block_count + 1].as_slice()), + usage: BufferUsages::STORAGE | BufferUsages::COPY_SRC, }); let sum_buffer = device.create_buffer_init(&wgpu::util::BufferInitDescriptor { label: Some("indirect_buffer_sum"), - contents: bytemuck::cast_slice(vec![0; size].as_slice()), + contents: bytemuck::cast_slice( + vec![0; block_count * Self::THREAD_COUNT * 2].as_slice(), + ), usage: BufferUsages::STORAGE, }); - let compaction_buffer = device.create_buffer_init(&wgpu::util::BufferInitDescriptor { - label: Some("indirect_buffer_compaction"), - contents: bytemuck::cast_slice(vec![0; size].as_slice()), - usage: BufferUsages::STORAGE, + let count_buffer_bg_layout = + device.create_bind_group_layout(&wgpu::BindGroupLayoutDescriptor { + label: Some("count_buffer_bg_layout"), + entries: &[ + wgpu::BindGroupLayoutEntry { + binding: 0, + visibility: ShaderStages::COMPUTE, + ty: wgpu::BindingType::Buffer { + ty: wgpu::BufferBindingType::Storage { read_only: false }, + has_dynamic_offset: false, + min_binding_size: None, + }, + count: None, + }, + wgpu::BindGroupLayoutEntry { + binding: 1, + visibility: ShaderStages::COMPUTE, + ty: wgpu::BindingType::Buffer { + ty: wgpu::BufferBindingType::Storage { read_only: false }, + has_dynamic_offset: false, + min_binding_size: None, + }, + count: None, + }, + ], + }); + + let compaction_bg_layout = + device.create_bind_group_layout(&wgpu::BindGroupLayoutDescriptor { + label: Some("count_buffer_bg_layout"), + entries: &[ + wgpu::BindGroupLayoutEntry { + binding: 0, + visibility: ShaderStages::COMPUTE, + ty: wgpu::BindingType::Buffer { + ty: wgpu::BufferBindingType::Storage { read_only: false }, + has_dynamic_offset: false, + min_binding_size: None, + }, + count: None, + }, + wgpu::BindGroupLayoutEntry { + binding: 1, + visibility: ShaderStages::COMPUTE, + ty: wgpu::BindingType::Buffer { + ty: wgpu::BufferBindingType::Storage { read_only: false }, + has_dynamic_offset: false, + min_binding_size: None, + }, + count: None, + }, + ], + }); + + let compaction_bg = device.create_bind_group(&wgpu::BindGroupDescriptor { + label: Some("compaction_bg"), + layout: &compaction_bg_layout, + entries: &[ + wgpu::BindGroupEntry { + binding: 0, + resource: reduced_buffer.as_entire_binding(), + }, + wgpu::BindGroupEntry { + binding: 1, + resource: sum_buffer.as_entire_binding(), + }, + ], }); - BufferCompactor { + let compaction_pipeline_layout = + device.create_pipeline_layout(&wgpu::PipelineLayoutDescriptor { + label: Some("compaction_pipeline_layout"), + bind_group_layouts: &[Some(&count_buffer_bg_layout), Some(&compaction_bg_layout)], + immediate_size: 0, + }); + + let compaction_shader_module = unsafe { + device.create_shader_module_trusted( + wgpu::ShaderModuleDescriptor { + label: Some("compaction_shader_layout"), + source: wgpu::ShaderSource::SpirV(wgpu::__macro_helpers::Cow::Borrowed( + wgpu::include_spirv_source!("../../shaders/compaction.spv"), + )), + }, + wgpu::ShaderRuntimeChecks { + bounds_checks: false, + force_loop_bounding: false, + ray_query_initialization_tracking: false, + task_shader_dispatch_tracking: false, + mesh_shader_primitive_indices_clamp: false, + int_div_checks: false, + }, + ) + }; + + let reduce_sum_pipeline = + device.create_compute_pipeline(&wgpu::ComputePipelineDescriptor { + label: Some("reduce_sum_pipeline"), + layout: Some(&compaction_pipeline_layout), + module: &compaction_shader_module, + entry_point: Some("block_sum"), + compilation_options: wgpu::PipelineCompilationOptions::default(), + cache: None, + }); + let partial_sum_pipeline = + device.create_compute_pipeline(&wgpu::ComputePipelineDescriptor { + label: Some("partial_sum_pipeline "), + layout: Some(&compaction_pipeline_layout), + module: &compaction_shader_module, + entry_point: Some("linear_reduced_sum"), + compilation_options: wgpu::PipelineCompilationOptions::default(), + cache: None, + }); + let scatter_pipeline = device.create_compute_pipeline(&wgpu::ComputePipelineDescriptor { + label: Some("uniform_add_pipeline"), + layout: Some(&compaction_pipeline_layout), + module: &compaction_shader_module, + entry_point: Some("scatter"), + compilation_options: wgpu::PipelineCompilationOptions::default(), + cache: None, + }); + + NonZeroBufferCompactor { size, block_count, reduced_buffer, sum_buffer, - compaction_buffer, + count_buffer_bg_layout, + compaction_bg, + reduce_sum_pipeline, + partial_sum_pipeline, + scatter_pipeline, } } + + pub fn total_sum_buffer(&self) -> BufferSlice<'_> + { + self.reduced_buffer + .slice(((size_of::() * self.block_count) as u64)..) + } + + pub fn copy_count_to_buffer(&self, encoder: &mut CommandEncoder, target: &Buffer) + { + encoder.copy_buffer_to_buffer( + &self.reduced_buffer, + (size_of::() * self.block_count) as u64, + target, + 0, + size_of::() as u64, + ); + } + + pub fn compact_buffer( + &self, + device: &Device, + encoder: &mut CommandEncoder, + count_buffer: &Buffer, + index_buffer: &Buffer, + ) + { + let count_buffer_bind_group = device.create_bind_group(&wgpu::BindGroupDescriptor { + label: Some("count_buffer_bind_group"), + layout: &self.count_buffer_bg_layout, + entries: &[ + wgpu::BindGroupEntry { + binding: 0, + resource: count_buffer.as_entire_binding(), + }, + wgpu::BindGroupEntry { + binding: 1, + resource: index_buffer.as_entire_binding(), + }, + ], + }); + let mut compute_pass = encoder.begin_compute_pass(&wgpu::ComputePassDescriptor { + label: Some("buffer_compactor_compute_pass"), + timestamp_writes: None, + }); + + compute_pass.set_bind_group(0, &count_buffer_bind_group, &[]); + compute_pass.set_bind_group(1, &self.compaction_bg, &[]); + + // Perform local running sum, then reduce + compute_pass.set_pipeline(&self.reduce_sum_pipeline); + compute_pass.dispatch_workgroups(self.block_count as u32, 1, 1); + + // Partial sums are now in the reuced sum buffer. Perform linear naive runnig sum on it + compute_pass.set_pipeline(&self.partial_sum_pipeline); + compute_pass.dispatch_workgroups(1, 1, 1); + + // We can now reapply the partial running sum on the subblocks + compute_pass.set_pipeline(&self.scatter_pipeline); + compute_pass.dispatch_workgroups(self.block_count as u32, 1, 1); + } }