--wip-- [skip ci]
This commit is contained in:
+178
-91
@@ -1,11 +1,16 @@
|
||||
use glam::{self, FloatExt, Vec3Swizzles};
|
||||
use std::simd::{
|
||||
StdFloat,
|
||||
cmp::{SimdPartialEq, SimdPartialOrd},
|
||||
num::{SimdFloat, SimdInt},
|
||||
};
|
||||
|
||||
use crate::{
|
||||
CSG,
|
||||
ssa::{SSAInput, SSAInstruction, SSAOpcode},
|
||||
};
|
||||
|
||||
type Value = f32;
|
||||
type Value = core::simd::f32x8;
|
||||
type IValue = core::simd::i32x8;
|
||||
|
||||
#[derive(Clone, Debug)]
|
||||
pub(crate) struct Interpreter<'csg> {
|
||||
@@ -17,7 +22,7 @@ impl Interpreter<'_> {
|
||||
fn load(&self, consider: SSAInput) -> Value {
|
||||
match consider {
|
||||
SSAInput::Register(r) => self.value_map[r as usize],
|
||||
SSAInput::Constant(c) => c,
|
||||
SSAInput::Constant(c) => Value::splat(c),
|
||||
}
|
||||
}
|
||||
|
||||
@@ -28,7 +33,7 @@ impl Interpreter<'_> {
|
||||
|
||||
//imitate glsl, not rust
|
||||
fn sign(f: Value) -> Value {
|
||||
if f == 0. { f } else { f.signum() }
|
||||
f.simd_eq(Value::splat(0.0)).select(f, f.signum())
|
||||
}
|
||||
|
||||
fn fract(f: Value) -> Value {
|
||||
@@ -38,13 +43,13 @@ fn fract(f: Value) -> Value {
|
||||
impl<'csg> Interpreter<'csg> {
|
||||
pub(crate) fn new(csg: &'csg CSG) -> Self {
|
||||
Interpreter {
|
||||
value_map: vec![0.; csg.parts.last_output as usize],
|
||||
value_map: vec![Value::splat(0.); csg.parts.last_output as usize],
|
||||
csg,
|
||||
}
|
||||
}
|
||||
|
||||
fn clear_stacks(&mut self) {
|
||||
self.value_map = vec![0.; self.csg.parts.last_output as usize];
|
||||
self.value_map = vec![Value::splat(0.); self.csg.parts.last_output as usize];
|
||||
}
|
||||
|
||||
fn param_one(&mut self, instruction: &SSAInstruction, func: impl Fn(Value) -> Value) {
|
||||
@@ -62,7 +67,11 @@ impl<'csg> Interpreter<'csg> {
|
||||
}
|
||||
}
|
||||
|
||||
fn param_three(&mut self, instruction: &SSAInstruction, func: impl Fn(Value, Value, Value) -> Value) {
|
||||
fn param_three(
|
||||
&mut self,
|
||||
instruction: &SSAInstruction,
|
||||
func: impl Fn(Value, Value, Value) -> Value,
|
||||
) {
|
||||
for i in 0..instruction.opcode.size as usize {
|
||||
let val_a = self.load(instruction.inputs[i]);
|
||||
let val_b = self.load(instruction.inputs[i + instruction.opcode.size as usize]);
|
||||
@@ -85,7 +94,7 @@ impl<'csg> Interpreter<'csg> {
|
||||
}
|
||||
}
|
||||
|
||||
pub(crate) fn scene(&mut self, p: glam::Vec3) -> Value {
|
||||
pub(crate) fn scene(&mut self, px: Value, py: Value, pz: Value) -> Value {
|
||||
self.clear_stacks();
|
||||
|
||||
for instruction in &self.csg.parts.tape {
|
||||
@@ -95,9 +104,9 @@ impl<'csg> Interpreter<'csg> {
|
||||
return self.load(instruction.inputs[0]);
|
||||
},
|
||||
SSAPosition => {
|
||||
self.store(instruction.outputs[0], p.x);
|
||||
self.store(instruction.outputs[1], p.y);
|
||||
self.store(instruction.outputs[2], p.z);
|
||||
self.store(instruction.outputs[0], px);
|
||||
self.store(instruction.outputs[1], py);
|
||||
self.store(instruction.outputs[2], pz);
|
||||
},
|
||||
SSAAdd => {
|
||||
self.param_two(instruction, |val_a, val_b| val_a + val_b);
|
||||
@@ -118,11 +127,11 @@ impl<'csg> Interpreter<'csg> {
|
||||
self.param_two(instruction, |val_a, val_b| val_a.atan2(val_b));
|
||||
},
|
||||
SSAMin => {
|
||||
self.param_two(instruction, |val_a, val_b| val_a.min(val_b));
|
||||
self.param_two(instruction, |val_a, val_b| val_a.simd_min(val_b));
|
||||
},
|
||||
SSAMinMaterial => todo!(),
|
||||
SSAMax => {
|
||||
self.param_two(instruction, |val_a, val_b| val_a.max(val_b));
|
||||
self.param_two(instruction, |val_a, val_b| val_a.simd_max(val_b));
|
||||
},
|
||||
SSAMaxMaterial => todo!(),
|
||||
SSACross => {
|
||||
@@ -132,12 +141,9 @@ impl<'csg> Interpreter<'csg> {
|
||||
let b_x = self.load(instruction.inputs[3]);
|
||||
let b_y = self.load(instruction.inputs[4]);
|
||||
let b_z = self.load(instruction.inputs[5]);
|
||||
let val_a = glam::vec3(a_x, a_y, a_z);
|
||||
let val_b = glam::vec3(b_x, b_y, b_z);
|
||||
let cross = val_a.cross(val_b);
|
||||
self.store(instruction.outputs[0], cross.x);
|
||||
self.store(instruction.outputs[1], cross.y);
|
||||
self.store(instruction.outputs[2], cross.z);
|
||||
self.store(instruction.outputs[0], (a_y * b_z) + (a_z * b_y));
|
||||
self.store(instruction.outputs[1], (a_z * b_x) + (a_x * b_z));
|
||||
self.store(instruction.outputs[2], (a_x * b_y) + (a_y * b_x));
|
||||
},
|
||||
SSADot => {
|
||||
let val_a = (0..instruction.opcode.size)
|
||||
@@ -146,27 +152,47 @@ impl<'csg> Interpreter<'csg> {
|
||||
let val_b = (instruction.opcode.size..(instruction.opcode.size * 2))
|
||||
.map(|i| self.load(instruction.inputs[i as usize]))
|
||||
.collect::<Vec<_>>();
|
||||
self.store(instruction.outputs[0], match instruction.opcode.size {
|
||||
1 => val_a[0] * val_b[0],
|
||||
2 => glam::vec2(val_a[0], val_a[1]).dot(glam::vec2(val_b[0], val_b[1])),
|
||||
3 => glam::vec3(val_a[0], val_a[1], val_a[2])
|
||||
.dot(glam::vec3(val_b[0], val_b[1], val_b[2])),
|
||||
4 => glam::vec4(val_a[0], val_a[1], val_a[2], val_a[3])
|
||||
.dot(glam::vec4(val_b[0], val_b[1], val_b[2], val_b[3])),
|
||||
_ => unreachable!(),
|
||||
});
|
||||
self.store(
|
||||
instruction.outputs[0],
|
||||
match instruction.opcode.size {
|
||||
1 => val_a[0] * val_b[0],
|
||||
2 => (val_a[0] * val_b[0]) + (val_a[1] * val_b[1]),
|
||||
3 => {
|
||||
(val_a[0] * val_b[0])
|
||||
+ (val_a[1] * val_b[1])
|
||||
+ (val_a[2] * val_b[2])
|
||||
},
|
||||
4 => {
|
||||
(val_a[0] * val_b[0])
|
||||
+ (val_a[1] * val_b[1])
|
||||
+ (val_a[2] * val_b[2])
|
||||
+ (val_a[3] * val_b[3])
|
||||
},
|
||||
_ => unreachable!(),
|
||||
},
|
||||
);
|
||||
},
|
||||
SSALength => {
|
||||
let val_a = (0..instruction.opcode.size)
|
||||
.map(|i| self.load(instruction.inputs[i as usize]))
|
||||
.collect::<Vec<_>>();
|
||||
self.store(instruction.outputs[0], match instruction.opcode.size {
|
||||
1 => val_a[0],
|
||||
2 => glam::vec2(val_a[0], val_a[1]).length(),
|
||||
3 => glam::vec3(val_a[0], val_a[1], val_a[2]).length(),
|
||||
4 => glam::vec4(val_a[0], val_a[1], val_a[2], val_a[3]).length(),
|
||||
_ => unreachable!(),
|
||||
});
|
||||
self.store(
|
||||
instruction.outputs[0],
|
||||
match instruction.opcode.size {
|
||||
1 => val_a[0],
|
||||
2 => ((val_a[0] * val_a[0]) + (val_a[1] * val_a[1])).sqrt(),
|
||||
3 => ((val_a[0] * val_a[0])
|
||||
+ (val_a[1] * val_a[1])
|
||||
+ (val_a[2] * val_a[2]))
|
||||
.sqrt(),
|
||||
4 => ((val_a[0] * val_a[0])
|
||||
+ (val_a[1] * val_a[1])
|
||||
+ (val_a[2] * val_a[2])
|
||||
+ (val_a[3] * val_a[3]))
|
||||
.sqrt(),
|
||||
_ => unreachable!(),
|
||||
},
|
||||
);
|
||||
},
|
||||
SSADistance => {
|
||||
let val_a = (0..instruction.opcode.size)
|
||||
@@ -175,17 +201,27 @@ impl<'csg> Interpreter<'csg> {
|
||||
let val_b = (instruction.opcode.size..(instruction.opcode.size * 2))
|
||||
.map(|i| self.load(instruction.inputs[i as usize]))
|
||||
.collect::<Vec<_>>();
|
||||
self.store(instruction.outputs[0], match instruction.opcode.size {
|
||||
1 => val_a[0] - val_b[0],
|
||||
2 => {
|
||||
glam::vec2(val_a[0], val_a[1]).distance(glam::vec2(val_b[0], val_b[1]))
|
||||
self.store(
|
||||
instruction.outputs[0],
|
||||
match instruction.opcode.size {
|
||||
1 => val_b[0] - val_a[0],
|
||||
2 => (((val_a[0] - val_b[0]) * (val_a[0] - val_b[0]))
|
||||
+ ((val_a[1] - val_b[1]) * (val_a[1] - val_b[1])))
|
||||
.sqrt(),
|
||||
3 => (((val_a[0] - val_b[0]) * (val_a[0] - val_b[0]))
|
||||
+ ((val_a[1] - val_b[1]) * (val_a[1] - val_b[1]))
|
||||
+ ((val_a[2] - val_b[2]) * (val_a[2] - val_b[2]))
|
||||
+ ((val_a[3] - val_b[3]) * (val_a[3] - val_b[3])))
|
||||
.sqrt(),
|
||||
3 => (((val_a[0] - val_b[0]) * (val_a[0] - val_b[0]))
|
||||
+ ((val_a[1] - val_b[1]) * (val_a[1] - val_b[1]))
|
||||
+ ((val_a[2] - val_b[2]) * (val_a[2] - val_b[2]))
|
||||
+ ((val_a[3] - val_b[3]) * (val_a[3] - val_b[3]))
|
||||
+ ((val_a[4] - val_b[4]) * (val_a[4] - val_b[4])))
|
||||
.sqrt(),
|
||||
_ => unreachable!(),
|
||||
},
|
||||
3 => glam::vec3(val_a[0], val_a[1], val_a[2])
|
||||
.distance(glam::vec3(val_b[0], val_b[1], val_b[2])),
|
||||
4 => glam::vec4(val_a[0], val_a[1], val_a[2], val_a[3])
|
||||
.distance(glam::vec4(val_b[0], val_b[1], val_b[2], val_b[3])),
|
||||
_ => unreachable!(),
|
||||
});
|
||||
);
|
||||
},
|
||||
SSANormalize => {
|
||||
let val_a = (0..instruction.opcode.size)
|
||||
@@ -194,26 +230,28 @@ impl<'csg> Interpreter<'csg> {
|
||||
|
||||
match instruction.opcode.size {
|
||||
1 => {
|
||||
self.store(instruction.outputs[0], 1.0);
|
||||
self.store(instruction.outputs[0], Value::splat(1.0));
|
||||
},
|
||||
2 => {
|
||||
let norm = glam::vec2(val_a[0], val_a[1]).normalize();
|
||||
self.store(instruction.outputs[0], norm.x);
|
||||
self.store(instruction.outputs[1], norm.y);
|
||||
let max = val_a[0].simd_max(val_a[1]);
|
||||
self.store(instruction.outputs[0], val_a[0] / max);
|
||||
self.store(instruction.outputs[1], val_a[1] / max);
|
||||
},
|
||||
3 => {
|
||||
let norm = glam::vec3(val_a[0], val_a[1], val_a[2]).normalize();
|
||||
self.store(instruction.outputs[0], norm.x);
|
||||
self.store(instruction.outputs[1], norm.y);
|
||||
self.store(instruction.outputs[2], norm.z);
|
||||
let max = val_a[0].simd_max(val_a[1]).simd_max(val_a[2]);
|
||||
self.store(instruction.outputs[0], val_a[0] / max);
|
||||
self.store(instruction.outputs[1], val_a[1] / max);
|
||||
self.store(instruction.outputs[2], val_a[2] / max);
|
||||
},
|
||||
4 => {
|
||||
let norm =
|
||||
glam::vec4(val_a[0], val_a[1], val_a[2], val_a[3]).normalize();
|
||||
self.store(instruction.outputs[0], norm.x);
|
||||
self.store(instruction.outputs[1], norm.y);
|
||||
self.store(instruction.outputs[2], norm.z);
|
||||
self.store(instruction.outputs[3], norm.w);
|
||||
let max = val_a[0]
|
||||
.simd_max(val_a[1])
|
||||
.simd_max(val_a[2])
|
||||
.simd_max(val_a[3]);
|
||||
self.store(instruction.outputs[0], val_a[0] / max);
|
||||
self.store(instruction.outputs[1], val_a[1] / max);
|
||||
self.store(instruction.outputs[2], val_a[2] / max);
|
||||
self.store(instruction.outputs[3], val_a[3] / max);
|
||||
},
|
||||
_ => unreachable!(),
|
||||
};
|
||||
@@ -243,7 +281,37 @@ impl<'csg> Interpreter<'csg> {
|
||||
self.param_one(instruction, |val_a| val_a.cos());
|
||||
},
|
||||
SSATan => {
|
||||
self.param_one(instruction, |val_a| val_a.tan());
|
||||
self.param_one(instruction, |val_a| {
|
||||
let sign_bit = val_a.is_sign_positive();
|
||||
let x = val_a.abs();
|
||||
let y = x * Value::splat(core::f32::consts::FRAC_2_PI * 2.);
|
||||
let emm2 = (y.cast::<i32>() + IValue::splat(1)) & IValue::splat(!1);
|
||||
let y = emm2.cast::<f32>();
|
||||
let polymask = (emm2 & IValue::splat(2)).simd_eq(IValue::splat(0));
|
||||
const MINUS_CEPHES_DP1: Value = Value::splat(-0.78515625);
|
||||
const MINUS_CEPHES_DP2: Value = Value::splat(-2.4187564849853515625e-4);
|
||||
const MINUS_CEPHES_DP3: Value = Value::splat(-3.77489497744594108e-8);
|
||||
let z = ((x + (y * MINUS_CEPHES_DP1)) + (y * MINUS_CEPHES_DP2))
|
||||
+ (y * MINUS_CEPHES_DP3);
|
||||
let zz = z * z;
|
||||
const TANCOF_P0: Value = Value::splat(9.38540185543E-3);
|
||||
const TANCOF_P1: Value = Value::splat(3.11992232697E-3);
|
||||
const TANCOF_P2: Value = Value::splat(2.44301354525E-2);
|
||||
const TANCOF_P3: Value = Value::splat(5.34112807005E-2);
|
||||
const TANCOF_P4: Value = Value::splat(1.33387994085E-1);
|
||||
const TANCOF_P5: Value = Value::splat(3.33331568548E-1);
|
||||
let y = ((((((((TANCOF_P0 * zz) + TANCOF_P1) * zz) + TANCOF_P2 * zz)
|
||||
+ TANCOF_P3 * zz)
|
||||
+ TANCOF_P4 * zz)
|
||||
+ TANCOF_P5)
|
||||
* zz
|
||||
* z)
|
||||
+ z;
|
||||
let y2 = (Value::splat(1.0) / y)
|
||||
* sign_bit.select(Value::splat(-1.0), Value::splat(0.0));
|
||||
let y = polymask.select(y, y2);
|
||||
y * sign_bit.select(Value::splat(-1.0), Value::splat(0.0))
|
||||
});
|
||||
},
|
||||
SSAAsin => {
|
||||
self.param_one(instruction, |val_a| val_a.asin());
|
||||
@@ -271,23 +339,31 @@ impl<'csg> Interpreter<'csg> {
|
||||
},
|
||||
SSASmoothMin => {
|
||||
self.param_three(instruction, |d1, d2, k| {
|
||||
let h = (0.5 + 0.5 * (d2 - d1) / k).clamp(0.0, 1.0);
|
||||
return d2.lerp(d1, h) - k * h * (1.0 - h);
|
||||
let h = (Value::splat(0.5) + (Value::splat(0.5) * (d2 - d1) / k))
|
||||
.simd_clamp(Value::splat(0.0), Value::splat(1.0));
|
||||
return ((d2 * (Value::splat(1.0) - h)) + (d1 * h))
|
||||
- k * h * (Value::splat(1.0) - h);
|
||||
});
|
||||
},
|
||||
SSASmoothMax => {
|
||||
self.param_three(instruction, |d1, d2, k| {
|
||||
let h = (0.5 - 0.5 * (d2 + d1) / k).clamp(0.0, 1.0);
|
||||
return d2.lerp(-d1, h) + k * h * (1.0 - h);
|
||||
let h = (Value::splat(0.5) - (Value::splat(0.5) * (d2 + d1) / k))
|
||||
.simd_clamp(Value::splat(0.0), Value::splat(1.0));
|
||||
return ((d2 * (Value::splat(1.0) - h)) + (-d1 * h))
|
||||
+ k * h * (Value::splat(1.0) - h);
|
||||
});
|
||||
},
|
||||
SSASmoothMinMaterial => todo!(),
|
||||
SSASmoothMaxMaterial => todo!(),
|
||||
SSAClamp => {
|
||||
self.param_three(instruction, |val_a, val_b, val_c| val_a.clamp(val_b, val_c));
|
||||
self.param_three(instruction, |val_a, val_b, val_c| {
|
||||
val_a.simd_clamp(val_b, val_c)
|
||||
});
|
||||
},
|
||||
SSAMix => {
|
||||
self.param_three(instruction, |val_a, val_b, val_c| val_a.lerp(val_b, val_c));
|
||||
self.param_three(instruction, |val_a, val_b, val_c| {
|
||||
(val_a * (Value::splat(1.0) - val_c)) + (val_b * val_c)
|
||||
});
|
||||
},
|
||||
SSAFMA => {
|
||||
self.param_three(instruction, |val_a, val_b, val_c| {
|
||||
@@ -299,8 +375,10 @@ impl<'csg> Interpreter<'csg> {
|
||||
let pos_y = self.load(instruction.inputs[1]);
|
||||
let pos_z = self.load(instruction.inputs[2]);
|
||||
let radius = self.load(instruction.inputs[3]);
|
||||
let pos = glam::vec3(pos_x, pos_y, pos_z);
|
||||
self.store(instruction.outputs[0], pos.length() - radius);
|
||||
self.store(
|
||||
instruction.outputs[0],
|
||||
((pos_x * pos_x) + (pos_y * pos_y) + (pos_z * pos_z)).sqrt() - radius,
|
||||
);
|
||||
},
|
||||
SSASDFBox => {
|
||||
let pos_x = self.load(instruction.inputs[0]);
|
||||
@@ -309,12 +387,16 @@ impl<'csg> Interpreter<'csg> {
|
||||
let rad_x = self.load(instruction.inputs[3]);
|
||||
let rad_y = self.load(instruction.inputs[4]);
|
||||
let rad_z = self.load(instruction.inputs[5]);
|
||||
let pos = glam::vec3(pos_x, pos_y, pos_z);
|
||||
let rad = glam::vec3(rad_x, rad_y, rad_z);
|
||||
let q = pos.abs() - rad;
|
||||
let qx = pos_x.abs() - rad_x;
|
||||
let qy = pos_y.abs() - rad_y;
|
||||
let qz = pos_z.abs() - rad_z;
|
||||
let qxmax = qx.simd_max(Value::splat(0.0));
|
||||
let qymax = qy.simd_max(Value::splat(0.0));
|
||||
let qzmax = qz.simd_max(Value::splat(0.0));
|
||||
self.store(
|
||||
instruction.outputs[0],
|
||||
q.max(glam::Vec3::ZERO).length() + q.x.max(q.y.max(q.z)).min(0.0),
|
||||
((qxmax * qxmax) + (qymax * qymax) + (qzmax * qzmax)).sqrt()
|
||||
+ qx.simd_max(qy.simd_max(qz)).simd_min(Value::splat(0.0)),
|
||||
);
|
||||
},
|
||||
SSASDFTorus => {
|
||||
@@ -323,40 +405,45 @@ impl<'csg> Interpreter<'csg> {
|
||||
let pos_z = self.load(instruction.inputs[2]);
|
||||
let radius1 = self.load(instruction.inputs[3]);
|
||||
let radius2 = self.load(instruction.inputs[4]);
|
||||
let p = glam::vec3(pos_x, pos_y, pos_z);
|
||||
let q = glam::vec2(p.xz().length() - radius1, p.y);
|
||||
self.store(instruction.outputs[0], q.length() - radius2);
|
||||
let q = ((pos_x * pos_x) + (pos_z * pos_z)).sqrt() - radius1;
|
||||
self.store(
|
||||
instruction.outputs[0],
|
||||
((q * q) + (pos_y * pos_y)).sqrt() - radius2,
|
||||
);
|
||||
},
|
||||
SSACompare => {
|
||||
self.param_two(instruction, |val_a, val_b| {
|
||||
match val_a.total_cmp(&val_b) {
|
||||
std::cmp::Ordering::Less => -1.,
|
||||
std::cmp::Ordering::Equal => 0.,
|
||||
std::cmp::Ordering::Greater => 1.,
|
||||
}
|
||||
val_a.simd_gt(val_b).select(
|
||||
Value::splat(1.0),
|
||||
val_a
|
||||
.simd_lt(val_b)
|
||||
.select(Value::splat(-1.0), Value::splat(0.0)),
|
||||
)
|
||||
});
|
||||
},
|
||||
SSAAnd => {
|
||||
self.param_two(
|
||||
instruction,
|
||||
|val_a, val_b| if val_a == 0. { val_a } else { val_b },
|
||||
);
|
||||
self.param_two(instruction, |val_a, val_b| {
|
||||
val_a.simd_eq(Value::splat(0.0)).select(val_a, val_b)
|
||||
});
|
||||
},
|
||||
SSAOr => {
|
||||
self.param_two(
|
||||
instruction,
|
||||
|val_a, val_b| if val_a == 0. { val_b } else { val_a },
|
||||
);
|
||||
self.param_two(instruction, |val_a, val_b| {
|
||||
val_a.simd_eq(Value::splat(0.0)).select(val_b, val_a)
|
||||
});
|
||||
},
|
||||
SSARecip => {
|
||||
self.param_one(instruction, |val_a| val_a.recip());
|
||||
},
|
||||
SSANot => {
|
||||
self.param_one(instruction, |val_a| if val_a == 0. { 1. } else { 0. });
|
||||
self.param_one(instruction, |val_a| {
|
||||
val_a
|
||||
.simd_eq(Value::splat(0.0))
|
||||
.select(Value::splat(1.0), Value::splat(0.0))
|
||||
});
|
||||
},
|
||||
SSAStop => return 0.,
|
||||
SSAStop => return Value::splat(0.0),
|
||||
}
|
||||
}
|
||||
return Value::NAN;
|
||||
return Value::splat(f32::NAN);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -1,5 +1,6 @@
|
||||
#![feature(variant_count)]
|
||||
#![feature(array_chunks)]
|
||||
#![feature(portable_simd)]
|
||||
|
||||
use std::{
|
||||
error::Error,
|
||||
|
||||
Reference in New Issue
Block a user