This commit is contained in:
2025-05-28 21:12:45 +01:00
parent a97f96b8d2
commit fc71f54f37
6 changed files with 907 additions and 560 deletions
Generated
+645 -382
View File
File diff suppressed because it is too large Load Diff
+8 -8
View File
@@ -6,10 +6,10 @@ edition = "2021"
# See more keys and their definitions at https://doc.rust-lang.org/cargo/reference/manifest.html
[dependencies]
vulkano = "0.34"
vulkano-shaders = { version = "0.34", features = ["shaderc-debug"] }
vulkano = "0.35.1"
vulkano-shaders = { version = "0.35.0", features = ["shaderc-debug"] }
winit = "0.30"
vulkano-util = "0.34"
vulkano-util = "0.35.0"
obj = "0.10"
bytemuck = { version = "1.13", features = [
@@ -18,11 +18,11 @@ bytemuck = { version = "1.13", features = [
"min_const_generics",
] }
glam = "0.29"
glam = "0.30.3"
egui = "0.30"
egui_winit_vulkano = "0.27"
egui_plot = "0.30"
egui = "0.31.1"
egui_winit_vulkano = "0.28.0"
egui_plot = "0.32.1"
serde = { version = "1", features = ["derive"] }
serde_json = "1"
@@ -31,7 +31,7 @@ utf-8 = "0.7"
rayon = "1.7"
rand = "0.8"
rand = "0.9.1"
foldhash = "*"
log = "0.4"
+1 -1
View File
@@ -194,7 +194,7 @@ pub(crate) fn gui_up(gui: &mut Gui, state: &mut GState) {
.map(|(x, y)| [x as f64, *y])
.collect::<Vec<_>>()
.into();
let line = Line::new(fps);
let line = Line::new("fps", fps);
ui.heading("FPS");
Plot::new("fps")
.view_aspect(2.0)
+5 -5
View File
@@ -1,9 +1,9 @@
# Interpreter redesign
Ground up redesign of the interpreter
Maximum mesh shaders at once is 32*32*2, or 2048. For 65,536 vgprs, each mesh shader can only use 32.
Maximum mesh shaders per SM is 128. For 65,536 vgprs, each mesh shader can use 512.
Instead of a stack, use SSA and then limited registers . This is a bit more compile friendly.
Instead of a stack, use SSA and then limited registers. This is a bit more compile friendly.
There are 16 available registers. Register 0 is always 0, and register 15 is the next item in the const tape.
This leaves 14 usable 32 bit float registers.
@@ -21,9 +21,9 @@ array of 8 floats, and at the end they are evaluated once.
Steps:
1. 8x8x1 task shaders per mesh (64)
2. each task shader works out 2x2x8 blocks (32) where 8 is depth
3. each task shader spawns 4x4x2 (8x8x1?) mesh shaders per block.
4. each mesh shader outputs 4 verts and 2 triangles, for a total 128/64 per workgroup.
2. each task shader works out 4x4x2 blocks (32) where 2 is depth
3. each task shader spawns 2x2x32 (128) mesh shaders per block.
4. each subgroup outputs 4 verts and 2 triangles, for a total 16/8 per workgroup.
For a mesh that takes up 1024x1024 pixel on screen, each quad takes up 16x16 pixels.
+50 -20
View File
@@ -23,6 +23,7 @@ use foldhash::{HashMap, HashMapExt, HashSet};
use glam::{self, EulerRot, Mat3, Mat4, Vec3, vec3};
use log::{error, info, trace};
use rayon::prelude::*;
use rspirv::binary::Disassemble;
use simplelog::{CombinedLogger, Config, TermLogger, WriteLogger};
use ssa::{SSAInput, SSAOpcode, SSAOpcodeSized, SSATape};
use vulkano::{
@@ -206,7 +207,9 @@ impl App {
let library = VulkanLibrary::new().expect("Vulkan is not installed???");
let required_extensions = Surface::required_extensions(event_loop).unwrap();
let instance = Instance::new(library, InstanceCreateInfo {
let instance = Instance::new(
library,
InstanceCreateInfo {
enabled_extensions: InstanceExtensions {
ext_surface_maintenance1: true,
..required_extensions
@@ -216,7 +219,8 @@ impl App {
application_name: Some(env!("CARGO_PKG_NAME").to_owned()),
application_version: app_version(),
..Default::default()
})
},
)
.unwrap();
let mut device_extensions = DeviceExtensions {
@@ -290,7 +294,9 @@ impl App {
device_extensions.khr_dynamic_rendering = true;
}
let (device, mut queues) = Device::new(physical_device, DeviceCreateInfo {
let (device, mut queues) = Device::new(
physical_device,
DeviceCreateInfo {
enabled_extensions: device_extensions,
queue_create_infos: vec![
QueueCreateInfo {
@@ -318,7 +324,8 @@ impl App {
..DeviceFeatures::empty()
},
..Default::default()
})
},
)
.expect("Unable to initialize device");
let graphics_queue = queues.next().expect("Unable to retrieve queues");
@@ -334,13 +341,15 @@ impl App {
Default::default(),
));
let uniform_buffer_allocator =
SubbufferAllocator::new(memory_allocator.clone(), SubbufferAllocatorCreateInfo {
let uniform_buffer_allocator = SubbufferAllocator::new(
memory_allocator.clone(),
SubbufferAllocatorCreateInfo {
buffer_usage: BufferUsage::UNIFORM_BUFFER | BufferUsage::STORAGE_BUFFER,
memory_type_filter: MemoryTypeFilter::PREFER_DEVICE
| MemoryTypeFilter::HOST_SEQUENTIAL_WRITE,
..Default::default()
});
},
);
let pipeline_cache = get_pipeline_cache(device.clone());
@@ -546,10 +555,13 @@ impl ApplicationHandler for App {
let surface_capabilities = self
.device
.physical_device()
.surface_capabilities(&surface, SurfaceInfo {
.surface_capabilities(
&surface,
SurfaceInfo {
present_mode: Some(present_mode),
..Default::default()
})
},
)
.unwrap();
let (image_format, _) = self
@@ -558,7 +570,10 @@ impl ApplicationHandler for App {
.surface_formats(&surface, Default::default())
.unwrap()[0];
Swapchain::new(self.device.clone(), surface.clone(), SwapchainCreateInfo {
Swapchain::new(
self.device.clone(),
surface.clone(),
SwapchainCreateInfo {
min_image_count: 3
.max(surface_capabilities.min_image_count)
.min(surface_capabilities.max_image_count.unwrap_or(u32::MAX)),
@@ -577,7 +592,8 @@ impl ApplicationHandler for App {
present_mode,
..Default::default()
})
},
)
.unwrap()
};
@@ -1304,10 +1320,13 @@ impl App {
}
builder
.next_subpass(Default::default(), SubpassBeginInfo {
.next_subpass(
Default::default(),
SubpassBeginInfo {
contents: SubpassContents::SecondaryCommandBuffers,
..Default::default()
})
},
)
.unwrap()
.execute_commands(guicb)
.unwrap()
@@ -1400,7 +1419,9 @@ fn framebuffer_generation(
.map(|image| {
let view = ImageView::new_default(image.clone()).unwrap();
Framebuffer::new(render_pass.clone(), FramebufferCreateInfo {
Framebuffer::new(
render_pass.clone(),
FramebufferCreateInfo {
attachments: if MSAA_ENABLE {
vec![
intermediary.as_ref().unwrap().clone(),
@@ -1411,7 +1432,8 @@ fn framebuffer_generation(
vec![view, depth_buffer.clone()]
},
..Default::default()
})
},
)
.unwrap()
})
.collect::<Vec<_>>();
@@ -1733,6 +1755,7 @@ fn object_size_dependent_setup(
for csg in state {
let tape = csg.parts.compile_to_gpu();
let tape2 = csg.parts.compile_to_spirv();
let mut description = implicit_fs::Description::default();
@@ -1820,11 +1843,15 @@ fn object_size_dependent_setup(
)
.unwrap();
let staging = SubbufferAllocator::new(allocator.clone(), SubbufferAllocatorCreateInfo {
let staging = SubbufferAllocator::new(
allocator.clone(),
SubbufferAllocatorCreateInfo {
buffer_usage: BufferUsage::TRANSFER_SRC,
memory_type_filter: MemoryTypeFilter::PREFER_HOST | MemoryTypeFilter::HOST_SEQUENTIAL_WRITE,
memory_type_filter: MemoryTypeFilter::PREFER_HOST
| MemoryTypeFilter::HOST_SEQUENTIAL_WRITE,
..Default::default()
});
},
);
let csg_scene = gpu_buffer(
&[&scene],
@@ -1884,10 +1911,13 @@ fn get_pipeline_cache(device: Arc<Device>) -> Arc<PipelineCache> {
};
unsafe {
PipelineCache::new(device, PipelineCacheCreateInfo {
PipelineCache::new(
device,
PipelineCacheCreateInfo {
initial_data,
..Default::default()
})
},
)
}
.unwrap()
}
+135 -81
View File
@@ -484,6 +484,8 @@ impl SSATape {
.unwrap();
let pos = b.function_parameter(vec3).unwrap();
b.begin_block(None).unwrap();
let mut mapping = HashMap::<u32, u32>::new();
for (line, instruction) in self.tape.iter().enumerate() {
@@ -667,10 +669,13 @@ impl SSATape {
&mut mapping,
instruction,
|b, val_a, val_b| {
b.ext_inst(float, None, glsl, spirv::GLOp::Atan2 as u32, [
IdRef(val_a),
IdRef(val_b),
])
b.ext_inst(
float,
None,
glsl,
spirv::GLOp::Atan2 as u32,
[IdRef(val_a), IdRef(val_b)],
)
.unwrap()
},
);
@@ -682,10 +687,13 @@ impl SSATape {
&mut mapping,
instruction,
|b, val_a, val_b| {
b.ext_inst(float, None, glsl, spirv::GLOp::FMin as u32, [
IdRef(val_a),
IdRef(val_b),
])
b.ext_inst(
float,
None,
glsl,
spirv::GLOp::FMin as u32,
[IdRef(val_a), IdRef(val_b)],
)
.unwrap()
},
);
@@ -698,10 +706,13 @@ impl SSATape {
&mut mapping,
instruction,
|b, val_a, val_b| {
b.ext_inst(float, None, glsl, spirv::GLOp::FMax as u32, [
IdRef(val_a),
IdRef(val_b),
])
b.ext_inst(
float,
None,
glsl,
spirv::GLOp::FMax as u32,
[IdRef(val_a), IdRef(val_b)],
)
.unwrap()
},
);
@@ -717,10 +728,13 @@ impl SSATape {
let val_a = b.composite_construct(vec3, None, [a_x, a_y, a_z]).unwrap();
let val_b = b.composite_construct(vec3, None, [b_x, b_y, b_z]).unwrap();
let cross = b
.ext_inst(vec3, None, glsl, spirv::GLOp::Cross as u32, [
IdRef(val_a),
IdRef(val_b),
])
.ext_inst(
vec3,
None,
glsl,
spirv::GLOp::Cross as u32,
[IdRef(val_a), IdRef(val_b)],
)
.unwrap();
mapping.insert(
instruction.outputs[0],
@@ -761,9 +775,13 @@ impl SSATape {
let vector = [void, vec1, vec2, vec3, vec4][instruction.opcode.size as usize];
let val_a = b.composite_construct(vector, None, val_a).unwrap();
let length = b
.ext_inst(float, None, glsl, spirv::GLOp::Length as u32, [IdRef(
val_a,
)])
.ext_inst(
float,
None,
glsl,
spirv::GLOp::Length as u32,
[IdRef(val_a)],
)
.unwrap();
mapping.insert(instruction.outputs[0], length);
},
@@ -782,10 +800,13 @@ impl SSATape {
let val_a = b.composite_construct(vector, None, val_a).unwrap();
let val_b = b.composite_construct(vector, None, val_b).unwrap();
let distance = b
.ext_inst(float, None, glsl, spirv::GLOp::Distance as u32, [
IdRef(val_a),
IdRef(val_b),
])
.ext_inst(
float,
None,
glsl,
spirv::GLOp::Distance as u32,
[IdRef(val_a), IdRef(val_b)],
)
.unwrap();
mapping.insert(instruction.outputs[0], distance);
},
@@ -804,10 +825,13 @@ impl SSATape {
let val_a = b.composite_construct(vector, None, val_a).unwrap();
let val_b = b.composite_construct(vector, None, val_b).unwrap();
let normal = b
.ext_inst(vector, None, glsl, spirv::GLOp::Normalize as u32, [
IdRef(val_a),
IdRef(val_b),
])
.ext_inst(
vector,
None,
glsl,
spirv::GLOp::Normalize as u32,
[IdRef(val_a), IdRef(val_b)],
)
.unwrap();
for i in 0..instruction.opcode.size as usize {
mapping.insert(
@@ -933,21 +957,25 @@ impl SSATape {
let div_k = b.f_div(float, None, mul_half, k).unwrap();
let add_half = b.f_add(float, None, div_k, half_const).unwrap();
let h = b
.ext_inst(float, None, glsl, spirv::GLOp::FClamp as u32, [
IdRef(add_half),
IdRef(zero_const),
IdRef(one_const),
])
.ext_inst(
float,
None,
glsl,
spirv::GLOp::FClamp as u32,
[IdRef(add_half), IdRef(zero_const), IdRef(one_const)],
)
.unwrap();
let negh = b.f_sub(float, None, one_const, h).unwrap();
let h_negh = b.f_mul(float, None, h, negh).unwrap();
let kh_negh = b.f_mul(float, None, k, h_negh).unwrap();
let mix = b
.ext_inst(float, None, glsl, spirv::GLOp::FMix as u32, [
IdRef(d2),
IdRef(d1),
IdRef(h),
])
.ext_inst(
float,
None,
glsl,
spirv::GLOp::FMix as u32,
[IdRef(d2), IdRef(d1), IdRef(h)],
)
.unwrap();
b.f_sub(float, None, mix, kh_negh).unwrap()
});
@@ -962,22 +990,26 @@ impl SSATape {
let div_k = b.f_div(float, None, mul_half, k).unwrap();
let add_half = b.f_sub(float, None, half_const, div_k).unwrap();
let h = b
.ext_inst(float, None, glsl, spirv::GLOp::FClamp as u32, [
IdRef(add_half),
IdRef(zero_const),
IdRef(one_const),
])
.ext_inst(
float,
None,
glsl,
spirv::GLOp::FClamp as u32,
[IdRef(add_half), IdRef(zero_const), IdRef(one_const)],
)
.unwrap();
let negh = b.f_sub(float, None, one_const, h).unwrap();
let h_negh = b.f_mul(float, None, h, negh).unwrap();
let kh_negh = b.f_mul(float, None, k, h_negh).unwrap();
let negate = b.f_negate(float, None, d1).unwrap();
let mix = b
.ext_inst(float, None, glsl, spirv::GLOp::FMix as u32, [
IdRef(d2),
IdRef(negate),
IdRef(h),
])
.ext_inst(
float,
None,
glsl,
spirv::GLOp::FMix as u32,
[IdRef(d2), IdRef(negate), IdRef(h)],
)
.unwrap();
b.f_add(float, None, mix, kh_negh).unwrap()
});
@@ -991,11 +1023,13 @@ impl SSATape {
&mut mapping,
instruction,
|b, val_a, val_b, val_c| {
b.ext_inst(float, None, glsl, spirv::GLOp::FClamp as u32, [
IdRef(val_a),
IdRef(val_b),
IdRef(val_c),
])
b.ext_inst(
float,
None,
glsl,
spirv::GLOp::FClamp as u32,
[IdRef(val_a), IdRef(val_b), IdRef(val_c)],
)
.unwrap()
},
);
@@ -1007,11 +1041,13 @@ impl SSATape {
&mut mapping,
instruction,
|b, val_a, val_b, val_c| {
b.ext_inst(float, None, glsl, spirv::GLOp::FMix as u32, [
IdRef(val_a),
IdRef(val_b),
IdRef(val_c),
])
b.ext_inst(
float,
None,
glsl,
spirv::GLOp::FMix as u32,
[IdRef(val_a), IdRef(val_b), IdRef(val_c)],
)
.unwrap()
},
);
@@ -1023,11 +1059,13 @@ impl SSATape {
&mut mapping,
instruction,
|b, val_a, val_b, val_c| {
b.ext_inst(float, None, glsl, spirv::GLOp::Fma as u32, [
IdRef(val_a),
IdRef(val_b),
IdRef(val_c),
])
b.ext_inst(
float,
None,
glsl,
spirv::GLOp::Fma as u32,
[IdRef(val_a), IdRef(val_b), IdRef(val_c)],
)
.unwrap()
},
);
@@ -1078,38 +1116,54 @@ impl SSATape {
.composite_construct(vec3, None, [zero, zero, zero])
.unwrap();
let q_limit = b
.ext_inst(vec3, None, glsl, spirv::GLOp::FMax as u32, [
IdRef(q),
IdRef(zero_vec3),
])
.ext_inst(
vec3,
None,
glsl,
spirv::GLOp::FMax as u32,
[IdRef(q), IdRef(zero_vec3)],
)
.unwrap();
let length = b
.ext_inst(float, None, glsl, spirv::GLOp::Length as u32, [IdRef(
q_limit,
)])
.ext_inst(
float,
None,
glsl,
spirv::GLOp::Length as u32,
[IdRef(q_limit)],
)
.unwrap();
let q_x = b.composite_extract(float, None, q, [0]).unwrap();
let q_y = b.composite_extract(float, None, q, [1]).unwrap();
let q_z = b.composite_extract(float, None, q, [2]).unwrap();
let max1 = b
.ext_inst(float, None, glsl, spirv::GLOp::FMax as u32, [
IdRef(q_x),
IdRef(q_y),
])
.ext_inst(
float,
None,
glsl,
spirv::GLOp::FMax as u32,
[IdRef(q_x), IdRef(q_y)],
)
.unwrap();
let max2 = b
.ext_inst(float, None, glsl, spirv::GLOp::FMax as u32, [
IdRef(max1),
IdRef(q_z),
])
.ext_inst(
float,
None,
glsl,
spirv::GLOp::FMax as u32,
[IdRef(max1), IdRef(q_z)],
)
.unwrap();
let min = b
.ext_inst(float, None, glsl, spirv::GLOp::FMax as u32, [
IdRef(max2),
IdRef(zero),
])
.ext_inst(
float,
None,
glsl,
spirv::GLOp::FMax as u32,
[IdRef(max2), IdRef(zero)],
)
.unwrap();
mapping.insert(