--wip-- [skip ci]

This commit is contained in:
2025-06-24 20:34:54 +01:00
parent 11ee139f43
commit 17cad7761a
2 changed files with 179 additions and 91 deletions
+178 -91
View File
@@ -1,11 +1,16 @@
use glam::{self, FloatExt, Vec3Swizzles}; use std::simd::{
StdFloat,
cmp::{SimdPartialEq, SimdPartialOrd},
num::{SimdFloat, SimdInt},
};
use crate::{ use crate::{
CSG, CSG,
ssa::{SSAInput, SSAInstruction, SSAOpcode}, ssa::{SSAInput, SSAInstruction, SSAOpcode},
}; };
type Value = f32; type Value = core::simd::f32x8;
type IValue = core::simd::i32x8;
#[derive(Clone, Debug)] #[derive(Clone, Debug)]
pub(crate) struct Interpreter<'csg> { pub(crate) struct Interpreter<'csg> {
@@ -17,7 +22,7 @@ impl Interpreter<'_> {
fn load(&self, consider: SSAInput) -> Value { fn load(&self, consider: SSAInput) -> Value {
match consider { match consider {
SSAInput::Register(r) => self.value_map[r as usize], SSAInput::Register(r) => self.value_map[r as usize],
SSAInput::Constant(c) => c, SSAInput::Constant(c) => Value::splat(c),
} }
} }
@@ -28,7 +33,7 @@ impl Interpreter<'_> {
//imitate glsl, not rust //imitate glsl, not rust
fn sign(f: Value) -> Value { fn sign(f: Value) -> Value {
if f == 0. { f } else { f.signum() } f.simd_eq(Value::splat(0.0)).select(f, f.signum())
} }
fn fract(f: Value) -> Value { fn fract(f: Value) -> Value {
@@ -38,13 +43,13 @@ fn fract(f: Value) -> Value {
impl<'csg> Interpreter<'csg> { impl<'csg> Interpreter<'csg> {
pub(crate) fn new(csg: &'csg CSG) -> Self { pub(crate) fn new(csg: &'csg CSG) -> Self {
Interpreter { Interpreter {
value_map: vec![0.; csg.parts.last_output as usize], value_map: vec![Value::splat(0.); csg.parts.last_output as usize],
csg, csg,
} }
} }
fn clear_stacks(&mut self) { fn clear_stacks(&mut self) {
self.value_map = vec![0.; self.csg.parts.last_output as usize]; self.value_map = vec![Value::splat(0.); self.csg.parts.last_output as usize];
} }
fn param_one(&mut self, instruction: &SSAInstruction, func: impl Fn(Value) -> Value) { fn param_one(&mut self, instruction: &SSAInstruction, func: impl Fn(Value) -> Value) {
@@ -62,7 +67,11 @@ impl<'csg> Interpreter<'csg> {
} }
} }
fn param_three(&mut self, instruction: &SSAInstruction, func: impl Fn(Value, Value, Value) -> Value) { fn param_three(
&mut self,
instruction: &SSAInstruction,
func: impl Fn(Value, Value, Value) -> Value,
) {
for i in 0..instruction.opcode.size as usize { for i in 0..instruction.opcode.size as usize {
let val_a = self.load(instruction.inputs[i]); let val_a = self.load(instruction.inputs[i]);
let val_b = self.load(instruction.inputs[i + instruction.opcode.size as usize]); let val_b = self.load(instruction.inputs[i + instruction.opcode.size as usize]);
@@ -85,7 +94,7 @@ impl<'csg> Interpreter<'csg> {
} }
} }
pub(crate) fn scene(&mut self, p: glam::Vec3) -> Value { pub(crate) fn scene(&mut self, px: Value, py: Value, pz: Value) -> Value {
self.clear_stacks(); self.clear_stacks();
for instruction in &self.csg.parts.tape { for instruction in &self.csg.parts.tape {
@@ -95,9 +104,9 @@ impl<'csg> Interpreter<'csg> {
return self.load(instruction.inputs[0]); return self.load(instruction.inputs[0]);
}, },
SSAPosition => { SSAPosition => {
self.store(instruction.outputs[0], p.x); self.store(instruction.outputs[0], px);
self.store(instruction.outputs[1], p.y); self.store(instruction.outputs[1], py);
self.store(instruction.outputs[2], p.z); self.store(instruction.outputs[2], pz);
}, },
SSAAdd => { SSAAdd => {
self.param_two(instruction, |val_a, val_b| val_a + val_b); self.param_two(instruction, |val_a, val_b| val_a + val_b);
@@ -118,11 +127,11 @@ impl<'csg> Interpreter<'csg> {
self.param_two(instruction, |val_a, val_b| val_a.atan2(val_b)); self.param_two(instruction, |val_a, val_b| val_a.atan2(val_b));
}, },
SSAMin => { SSAMin => {
self.param_two(instruction, |val_a, val_b| val_a.min(val_b)); self.param_two(instruction, |val_a, val_b| val_a.simd_min(val_b));
}, },
SSAMinMaterial => todo!(), SSAMinMaterial => todo!(),
SSAMax => { SSAMax => {
self.param_two(instruction, |val_a, val_b| val_a.max(val_b)); self.param_two(instruction, |val_a, val_b| val_a.simd_max(val_b));
}, },
SSAMaxMaterial => todo!(), SSAMaxMaterial => todo!(),
SSACross => { SSACross => {
@@ -132,12 +141,9 @@ impl<'csg> Interpreter<'csg> {
let b_x = self.load(instruction.inputs[3]); let b_x = self.load(instruction.inputs[3]);
let b_y = self.load(instruction.inputs[4]); let b_y = self.load(instruction.inputs[4]);
let b_z = self.load(instruction.inputs[5]); let b_z = self.load(instruction.inputs[5]);
let val_a = glam::vec3(a_x, a_y, a_z); self.store(instruction.outputs[0], (a_y * b_z) + (a_z * b_y));
let val_b = glam::vec3(b_x, b_y, b_z); self.store(instruction.outputs[1], (a_z * b_x) + (a_x * b_z));
let cross = val_a.cross(val_b); self.store(instruction.outputs[2], (a_x * b_y) + (a_y * b_x));
self.store(instruction.outputs[0], cross.x);
self.store(instruction.outputs[1], cross.y);
self.store(instruction.outputs[2], cross.z);
}, },
SSADot => { SSADot => {
let val_a = (0..instruction.opcode.size) let val_a = (0..instruction.opcode.size)
@@ -146,27 +152,47 @@ impl<'csg> Interpreter<'csg> {
let val_b = (instruction.opcode.size..(instruction.opcode.size * 2)) let val_b = (instruction.opcode.size..(instruction.opcode.size * 2))
.map(|i| self.load(instruction.inputs[i as usize])) .map(|i| self.load(instruction.inputs[i as usize]))
.collect::<Vec<_>>(); .collect::<Vec<_>>();
self.store(instruction.outputs[0], match instruction.opcode.size { self.store(
1 => val_a[0] * val_b[0], instruction.outputs[0],
2 => glam::vec2(val_a[0], val_a[1]).dot(glam::vec2(val_b[0], val_b[1])), match instruction.opcode.size {
3 => glam::vec3(val_a[0], val_a[1], val_a[2]) 1 => val_a[0] * val_b[0],
.dot(glam::vec3(val_b[0], val_b[1], val_b[2])), 2 => (val_a[0] * val_b[0]) + (val_a[1] * val_b[1]),
4 => glam::vec4(val_a[0], val_a[1], val_a[2], val_a[3]) 3 => {
.dot(glam::vec4(val_b[0], val_b[1], val_b[2], val_b[3])), (val_a[0] * val_b[0])
_ => unreachable!(), + (val_a[1] * val_b[1])
}); + (val_a[2] * val_b[2])
},
4 => {
(val_a[0] * val_b[0])
+ (val_a[1] * val_b[1])
+ (val_a[2] * val_b[2])
+ (val_a[3] * val_b[3])
},
_ => unreachable!(),
},
);
}, },
SSALength => { SSALength => {
let val_a = (0..instruction.opcode.size) let val_a = (0..instruction.opcode.size)
.map(|i| self.load(instruction.inputs[i as usize])) .map(|i| self.load(instruction.inputs[i as usize]))
.collect::<Vec<_>>(); .collect::<Vec<_>>();
self.store(instruction.outputs[0], match instruction.opcode.size { self.store(
1 => val_a[0], instruction.outputs[0],
2 => glam::vec2(val_a[0], val_a[1]).length(), match instruction.opcode.size {
3 => glam::vec3(val_a[0], val_a[1], val_a[2]).length(), 1 => val_a[0],
4 => glam::vec4(val_a[0], val_a[1], val_a[2], val_a[3]).length(), 2 => ((val_a[0] * val_a[0]) + (val_a[1] * val_a[1])).sqrt(),
_ => unreachable!(), 3 => ((val_a[0] * val_a[0])
}); + (val_a[1] * val_a[1])
+ (val_a[2] * val_a[2]))
.sqrt(),
4 => ((val_a[0] * val_a[0])
+ (val_a[1] * val_a[1])
+ (val_a[2] * val_a[2])
+ (val_a[3] * val_a[3]))
.sqrt(),
_ => unreachable!(),
},
);
}, },
SSADistance => { SSADistance => {
let val_a = (0..instruction.opcode.size) let val_a = (0..instruction.opcode.size)
@@ -175,17 +201,27 @@ impl<'csg> Interpreter<'csg> {
let val_b = (instruction.opcode.size..(instruction.opcode.size * 2)) let val_b = (instruction.opcode.size..(instruction.opcode.size * 2))
.map(|i| self.load(instruction.inputs[i as usize])) .map(|i| self.load(instruction.inputs[i as usize]))
.collect::<Vec<_>>(); .collect::<Vec<_>>();
self.store(instruction.outputs[0], match instruction.opcode.size { self.store(
1 => val_a[0] - val_b[0], instruction.outputs[0],
2 => { match instruction.opcode.size {
glam::vec2(val_a[0], val_a[1]).distance(glam::vec2(val_b[0], val_b[1])) 1 => val_b[0] - val_a[0],
2 => (((val_a[0] - val_b[0]) * (val_a[0] - val_b[0]))
+ ((val_a[1] - val_b[1]) * (val_a[1] - val_b[1])))
.sqrt(),
3 => (((val_a[0] - val_b[0]) * (val_a[0] - val_b[0]))
+ ((val_a[1] - val_b[1]) * (val_a[1] - val_b[1]))
+ ((val_a[2] - val_b[2]) * (val_a[2] - val_b[2]))
+ ((val_a[3] - val_b[3]) * (val_a[3] - val_b[3])))
.sqrt(),
3 => (((val_a[0] - val_b[0]) * (val_a[0] - val_b[0]))
+ ((val_a[1] - val_b[1]) * (val_a[1] - val_b[1]))
+ ((val_a[2] - val_b[2]) * (val_a[2] - val_b[2]))
+ ((val_a[3] - val_b[3]) * (val_a[3] - val_b[3]))
+ ((val_a[4] - val_b[4]) * (val_a[4] - val_b[4])))
.sqrt(),
_ => unreachable!(),
}, },
3 => glam::vec3(val_a[0], val_a[1], val_a[2]) );
.distance(glam::vec3(val_b[0], val_b[1], val_b[2])),
4 => glam::vec4(val_a[0], val_a[1], val_a[2], val_a[3])
.distance(glam::vec4(val_b[0], val_b[1], val_b[2], val_b[3])),
_ => unreachable!(),
});
}, },
SSANormalize => { SSANormalize => {
let val_a = (0..instruction.opcode.size) let val_a = (0..instruction.opcode.size)
@@ -194,26 +230,28 @@ impl<'csg> Interpreter<'csg> {
match instruction.opcode.size { match instruction.opcode.size {
1 => { 1 => {
self.store(instruction.outputs[0], 1.0); self.store(instruction.outputs[0], Value::splat(1.0));
}, },
2 => { 2 => {
let norm = glam::vec2(val_a[0], val_a[1]).normalize(); let max = val_a[0].simd_max(val_a[1]);
self.store(instruction.outputs[0], norm.x); self.store(instruction.outputs[0], val_a[0] / max);
self.store(instruction.outputs[1], norm.y); self.store(instruction.outputs[1], val_a[1] / max);
}, },
3 => { 3 => {
let norm = glam::vec3(val_a[0], val_a[1], val_a[2]).normalize(); let max = val_a[0].simd_max(val_a[1]).simd_max(val_a[2]);
self.store(instruction.outputs[0], norm.x); self.store(instruction.outputs[0], val_a[0] / max);
self.store(instruction.outputs[1], norm.y); self.store(instruction.outputs[1], val_a[1] / max);
self.store(instruction.outputs[2], norm.z); self.store(instruction.outputs[2], val_a[2] / max);
}, },
4 => { 4 => {
let norm = let max = val_a[0]
glam::vec4(val_a[0], val_a[1], val_a[2], val_a[3]).normalize(); .simd_max(val_a[1])
self.store(instruction.outputs[0], norm.x); .simd_max(val_a[2])
self.store(instruction.outputs[1], norm.y); .simd_max(val_a[3]);
self.store(instruction.outputs[2], norm.z); self.store(instruction.outputs[0], val_a[0] / max);
self.store(instruction.outputs[3], norm.w); self.store(instruction.outputs[1], val_a[1] / max);
self.store(instruction.outputs[2], val_a[2] / max);
self.store(instruction.outputs[3], val_a[3] / max);
}, },
_ => unreachable!(), _ => unreachable!(),
}; };
@@ -243,7 +281,37 @@ impl<'csg> Interpreter<'csg> {
self.param_one(instruction, |val_a| val_a.cos()); self.param_one(instruction, |val_a| val_a.cos());
}, },
SSATan => { SSATan => {
self.param_one(instruction, |val_a| val_a.tan()); self.param_one(instruction, |val_a| {
let sign_bit = val_a.is_sign_positive();
let x = val_a.abs();
let y = x * Value::splat(core::f32::consts::FRAC_2_PI * 2.);
let emm2 = (y.cast::<i32>() + IValue::splat(1)) & IValue::splat(!1);
let y = emm2.cast::<f32>();
let polymask = (emm2 & IValue::splat(2)).simd_eq(IValue::splat(0));
const MINUS_CEPHES_DP1: Value = Value::splat(-0.78515625);
const MINUS_CEPHES_DP2: Value = Value::splat(-2.4187564849853515625e-4);
const MINUS_CEPHES_DP3: Value = Value::splat(-3.77489497744594108e-8);
let z = ((x + (y * MINUS_CEPHES_DP1)) + (y * MINUS_CEPHES_DP2))
+ (y * MINUS_CEPHES_DP3);
let zz = z * z;
const TANCOF_P0: Value = Value::splat(9.38540185543E-3);
const TANCOF_P1: Value = Value::splat(3.11992232697E-3);
const TANCOF_P2: Value = Value::splat(2.44301354525E-2);
const TANCOF_P3: Value = Value::splat(5.34112807005E-2);
const TANCOF_P4: Value = Value::splat(1.33387994085E-1);
const TANCOF_P5: Value = Value::splat(3.33331568548E-1);
let y = ((((((((TANCOF_P0 * zz) + TANCOF_P1) * zz) + TANCOF_P2 * zz)
+ TANCOF_P3 * zz)
+ TANCOF_P4 * zz)
+ TANCOF_P5)
* zz
* z)
+ z;
let y2 = (Value::splat(1.0) / y)
* sign_bit.select(Value::splat(-1.0), Value::splat(0.0));
let y = polymask.select(y, y2);
y * sign_bit.select(Value::splat(-1.0), Value::splat(0.0))
});
}, },
SSAAsin => { SSAAsin => {
self.param_one(instruction, |val_a| val_a.asin()); self.param_one(instruction, |val_a| val_a.asin());
@@ -271,23 +339,31 @@ impl<'csg> Interpreter<'csg> {
}, },
SSASmoothMin => { SSASmoothMin => {
self.param_three(instruction, |d1, d2, k| { self.param_three(instruction, |d1, d2, k| {
let h = (0.5 + 0.5 * (d2 - d1) / k).clamp(0.0, 1.0); let h = (Value::splat(0.5) + (Value::splat(0.5) * (d2 - d1) / k))
return d2.lerp(d1, h) - k * h * (1.0 - h); .simd_clamp(Value::splat(0.0), Value::splat(1.0));
return ((d2 * (Value::splat(1.0) - h)) + (d1 * h))
- k * h * (Value::splat(1.0) - h);
}); });
}, },
SSASmoothMax => { SSASmoothMax => {
self.param_three(instruction, |d1, d2, k| { self.param_three(instruction, |d1, d2, k| {
let h = (0.5 - 0.5 * (d2 + d1) / k).clamp(0.0, 1.0); let h = (Value::splat(0.5) - (Value::splat(0.5) * (d2 + d1) / k))
return d2.lerp(-d1, h) + k * h * (1.0 - h); .simd_clamp(Value::splat(0.0), Value::splat(1.0));
return ((d2 * (Value::splat(1.0) - h)) + (-d1 * h))
+ k * h * (Value::splat(1.0) - h);
}); });
}, },
SSASmoothMinMaterial => todo!(), SSASmoothMinMaterial => todo!(),
SSASmoothMaxMaterial => todo!(), SSASmoothMaxMaterial => todo!(),
SSAClamp => { SSAClamp => {
self.param_three(instruction, |val_a, val_b, val_c| val_a.clamp(val_b, val_c)); self.param_three(instruction, |val_a, val_b, val_c| {
val_a.simd_clamp(val_b, val_c)
});
}, },
SSAMix => { SSAMix => {
self.param_three(instruction, |val_a, val_b, val_c| val_a.lerp(val_b, val_c)); self.param_three(instruction, |val_a, val_b, val_c| {
(val_a * (Value::splat(1.0) - val_c)) + (val_b * val_c)
});
}, },
SSAFMA => { SSAFMA => {
self.param_three(instruction, |val_a, val_b, val_c| { self.param_three(instruction, |val_a, val_b, val_c| {
@@ -299,8 +375,10 @@ impl<'csg> Interpreter<'csg> {
let pos_y = self.load(instruction.inputs[1]); let pos_y = self.load(instruction.inputs[1]);
let pos_z = self.load(instruction.inputs[2]); let pos_z = self.load(instruction.inputs[2]);
let radius = self.load(instruction.inputs[3]); let radius = self.load(instruction.inputs[3]);
let pos = glam::vec3(pos_x, pos_y, pos_z); self.store(
self.store(instruction.outputs[0], pos.length() - radius); instruction.outputs[0],
((pos_x * pos_x) + (pos_y * pos_y) + (pos_z * pos_z)).sqrt() - radius,
);
}, },
SSASDFBox => { SSASDFBox => {
let pos_x = self.load(instruction.inputs[0]); let pos_x = self.load(instruction.inputs[0]);
@@ -309,12 +387,16 @@ impl<'csg> Interpreter<'csg> {
let rad_x = self.load(instruction.inputs[3]); let rad_x = self.load(instruction.inputs[3]);
let rad_y = self.load(instruction.inputs[4]); let rad_y = self.load(instruction.inputs[4]);
let rad_z = self.load(instruction.inputs[5]); let rad_z = self.load(instruction.inputs[5]);
let pos = glam::vec3(pos_x, pos_y, pos_z); let qx = pos_x.abs() - rad_x;
let rad = glam::vec3(rad_x, rad_y, rad_z); let qy = pos_y.abs() - rad_y;
let q = pos.abs() - rad; let qz = pos_z.abs() - rad_z;
let qxmax = qx.simd_max(Value::splat(0.0));
let qymax = qy.simd_max(Value::splat(0.0));
let qzmax = qz.simd_max(Value::splat(0.0));
self.store( self.store(
instruction.outputs[0], instruction.outputs[0],
q.max(glam::Vec3::ZERO).length() + q.x.max(q.y.max(q.z)).min(0.0), ((qxmax * qxmax) + (qymax * qymax) + (qzmax * qzmax)).sqrt()
+ qx.simd_max(qy.simd_max(qz)).simd_min(Value::splat(0.0)),
); );
}, },
SSASDFTorus => { SSASDFTorus => {
@@ -323,40 +405,45 @@ impl<'csg> Interpreter<'csg> {
let pos_z = self.load(instruction.inputs[2]); let pos_z = self.load(instruction.inputs[2]);
let radius1 = self.load(instruction.inputs[3]); let radius1 = self.load(instruction.inputs[3]);
let radius2 = self.load(instruction.inputs[4]); let radius2 = self.load(instruction.inputs[4]);
let p = glam::vec3(pos_x, pos_y, pos_z); let q = ((pos_x * pos_x) + (pos_z * pos_z)).sqrt() - radius1;
let q = glam::vec2(p.xz().length() - radius1, p.y); self.store(
self.store(instruction.outputs[0], q.length() - radius2); instruction.outputs[0],
((q * q) + (pos_y * pos_y)).sqrt() - radius2,
);
}, },
SSACompare => { SSACompare => {
self.param_two(instruction, |val_a, val_b| { self.param_two(instruction, |val_a, val_b| {
match val_a.total_cmp(&val_b) { val_a.simd_gt(val_b).select(
std::cmp::Ordering::Less => -1., Value::splat(1.0),
std::cmp::Ordering::Equal => 0., val_a
std::cmp::Ordering::Greater => 1., .simd_lt(val_b)
} .select(Value::splat(-1.0), Value::splat(0.0)),
)
}); });
}, },
SSAAnd => { SSAAnd => {
self.param_two( self.param_two(instruction, |val_a, val_b| {
instruction, val_a.simd_eq(Value::splat(0.0)).select(val_a, val_b)
|val_a, val_b| if val_a == 0. { val_a } else { val_b }, });
);
}, },
SSAOr => { SSAOr => {
self.param_two( self.param_two(instruction, |val_a, val_b| {
instruction, val_a.simd_eq(Value::splat(0.0)).select(val_b, val_a)
|val_a, val_b| if val_a == 0. { val_b } else { val_a }, });
);
}, },
SSARecip => { SSARecip => {
self.param_one(instruction, |val_a| val_a.recip()); self.param_one(instruction, |val_a| val_a.recip());
}, },
SSANot => { SSANot => {
self.param_one(instruction, |val_a| if val_a == 0. { 1. } else { 0. }); self.param_one(instruction, |val_a| {
val_a
.simd_eq(Value::splat(0.0))
.select(Value::splat(1.0), Value::splat(0.0))
});
}, },
SSAStop => return 0., SSAStop => return Value::splat(0.0),
} }
} }
return Value::NAN; return Value::splat(f32::NAN);
} }
} }
+1
View File
@@ -1,5 +1,6 @@
#![feature(variant_count)] #![feature(variant_count)]
#![feature(array_chunks)] #![feature(array_chunks)]
#![feature(portable_simd)]
use std::{ use std::{
error::Error, error::Error,