New interval

This commit is contained in:
2025-02-01 23:00:52 +00:00
parent c71b187fae
commit a97f96b8d2
14 changed files with 1872 additions and 5024 deletions
+5 -2
View File
@@ -32,8 +32,11 @@ fn main() -> io::Result<()> {
continue;
}
let before_equals = entries[0].split("=").map(str::trim).collect::<Vec<&str>>();
let name = &before_equals[0][11..];
let before_equals = entries[0]
.split("=uint8_t")
.map(str::trim)
.collect::<Vec<&str>>();
let name = &before_equals[0][14..];
let value = &before_equals[1][..before_equals[1].len() - 1];
let comment = entries[1];
-4
View File
@@ -60,10 +60,6 @@ float FARPLANE;
#define interval_frags
#include "interpreter.glsl"
layout(set = 0, binding = 20, std430) restrict readonly buffer fragmentMasks {
uint8_t masks[][MASK_ARRAY_LENGTH];
} fragmentpassmasks;
#ifdef debug
vec3 getNormal(vec3 p, float dens) {
vec3 n;
-1186
View File
File diff suppressed because it is too large Load Diff
+5 -10
View File
@@ -10,26 +10,21 @@ struct Results {
uint stat;
};
layout(set = 0, binding = 30, std430) buffer ResultsArray {
layout(set = 1, binding = 30, std430) buffer ResultsArray {
Results r[];
} results;
layout(local_size_x = 32, local_size_y = 8, local_size_z = 1) in;
layout(local_size_x = 500, local_size_y = 1, local_size_z = 1) in;
void main()
{
DescriptionIndex = 0;
default_mask();
uint major_position = gl_LocalInvocationID.x;
uint minor_position = gl_LocalInvocationID.y;
uint minor_integer_cache[8];
program_counter = gl_LocalInvocationID.x;
desc = scene_description.desc[(DescriptionIndex) + 1];
get_caches;
results.r[OPPos].code = minor_integer_cache[minor_position];
results.r[OPPos].stat = STATIC_OPCODE_ARRAY[OPPos];
results.r[program_counter].code = next_opcode();
//results.r[program_counter].stat = STATIC_OPCODE_ARRAY[program_counter];
}
-3
View File
@@ -16,9 +16,6 @@ layout(location=0)out VertexOutput
vec4 position;
}vertexOutput[];
layout(set=0,binding=20, std430)restrict writeonly buffer fragmentMasks{
uint8_t masks[][MASK_ARRAY_LENGTH];
}fragmentpassmasks;
void main()
{
-10
View File
@@ -10,16 +10,6 @@
layout(local_size_x=32,local_size_y=1,local_size_z=1)in;
struct MeshMasks
{
uint8_t masks[32][MASK_ARRAY_LENGTH]; //928
uint8_t enabled[32]; //32
vec3 bottomleft; //12
vec3 topright; //12
uint globalindex; //4
}; //total = 988 bytes
taskPayloadSharedEXT MeshMasks meshmasks;
shared uint index;
void main()
+66 -60
View File
@@ -1,69 +1,75 @@
#ifndef instruction_set
#define instruction_set
const uint8_t OPCopy =uint8_t(0); // Returns the input. Useful for copying registers.
const uint8_t OPAdd =uint8_t(1); // Adds a vector to a vector component-wise.
const uint8_t OPSub =uint8_t(2); // Subtracts a vector from a vector component-wise.
const uint8_t OPMul =uint8_t(3); // Multiplies a vector and a vector component-wise.
const uint8_t OPDiv =uint8_t(4); // Divides a vector by a vector component-wise.
const uint8_t OPMod =uint8_t(5); // Calculates a vector modulo a vector component-wise.
const uint8_t OPRem =uint8_t(6); // Calculates a vector remainder a vector component-wise.
const uint8_t OPPow =uint8_t(7); // Calculates a vector to the power of a vector component-wise.
const uint8_t OPAtan2 =uint8_t(8); // Calculates a vector Atan2 a vector component-wise.
const uint8_t OPMin =uint8_t(9); // Calculates the minimum of a vector and a vector component-wise.
const uint8_t OPMax =uint8_t(10); // Calculates the maximum of a vector and a vector component-wise.
const uint8_t OPNegate =uint8_t(11); // Returns the negation of all components of a vector.
const uint8_t OPRound =uint8_t(12); // Returns all components of a vector rounded to the nearest integer, 0.5 away from zero.
const uint8_t OPRoundEven =uint8_t(13); // Returns all components of a vector rounded to the nearest integer, 0.5 to even.
const uint8_t OPTrunc =uint8_t(14); // Returns all components of a vector rounded to the nearest integer, 0.5 towards zero.
const uint8_t OPAbs =uint8_t(15); // Returns the absolute value of all components of a vector.
const uint8_t OPSign =uint8_t(16); // Returns the sign of all components of a vector.
const uint8_t OPFloor =uint8_t(17); // Returns the floor of all components of a vector.
const uint8_t OPCeil =uint8_t(18); // Returns the ceiling of all components of a vector.
const uint8_t OPFract =uint8_t(19); // Returns the fractional part of all components of a vector.
const uint8_t OPSin =uint8_t(20); // Returns the sine of all components of a vector.
const uint8_t OPCos =uint8_t(21); // Returns the cosine of all components of a vector.
const uint8_t OPTan =uint8_t(22); // Returns the tangent of all components of a vector.
const uint8_t OPAsin =uint8_t(23); // Returns the arc sine of all components of a vector.
const uint8_t OPAcos =uint8_t(24); // Returns the arc cosine of all components of a vector.
const uint8_t OPAtan =uint8_t(25); // Returns the arc tangent of all components of a vector.
const uint8_t OPSinh =uint8_t(26); // Returns the hyperbolic sine of all components of a vector.
const uint8_t OPCosh =uint8_t(27); // Returns the hyperbolic cosine of all components of a vector.
const uint8_t OPTanh =uint8_t(28); // Returns the hyperbolic tangent of all components of a vector.
const uint8_t OPAsinh =uint8_t(29); // Returns the hyperbolic arc sine of all components of a vector.
const uint8_t OPAcosh =uint8_t(30); // Returns the hyperbolic arc cosine of all components of a vector.
const uint8_t OPAtanh =uint8_t(31); // Returns the hyperbolic arc tangent of all components of a vector.
const uint8_t OPExp =uint8_t(32); // Returns e raised to all components of a vector.
const uint8_t OPLog =uint8_t(33); // Returns the natural logarithm of all components of a vector.
const uint8_t OPExp2 =uint8_t(34); // Returns 2 raised to all components of a vector.
const uint8_t OPLog2 =uint8_t(35); // Returns the base 2 logarithm of all components of a vector.
const uint8_t OPSqrt =uint8_t(36); // Returns the square root of all components of a vector.
const uint8_t OPInverseSqrt =uint8_t(37); // Returns one over the square root of all components of a vector.
const uint8_t OPSquare =uint8_t(38); // Returns the square of all components of a vector.
const uint8_t OPCube =uint8_t(39); // Returns the cube of all components of a vector.
const uint8_t OPSmoothMin =uint8_t(40); // Returns the smooth minimum between a vector and a vector, varied by a vector.
const uint8_t OPSmoothMax =uint8_t(41); // Returns the smooth maximum between a vector and a vector, varied by a vector.
const uint8_t OPClamp =uint8_t(42); // Clamps a vector between a vector and a vector.
const uint8_t OPMix =uint8_t(43); // Mixes between a vector and a vector, varied by a vector.
const uint8_t OPStep =uint8_t(44); // Steps between a vector and a vector, varied by a vector.
const uint8_t OPSmoothStep =uint8_t(45); // Smooth Steps between a vector and a vector, varied by a vector.
const uint8_t OPFMA =uint8_t(46); // Calculates a vector multiplied by a vector, then adds a vector.
const uint8_t OPDot =uint8_t(47); // Returns the dot product of two vectors.
const uint8_t OPLength =uint8_t(48); // Returns the length (magnitude) of a vector.
const uint8_t OPDistance =uint8_t(49); // Returns the length (magnitude) of the vector between two vectors.
const uint8_t OPNormalize =uint8_t(50); // Returns the normalised version of a vector.
// Element wise
const uint8_t OPCopy =uint8_t(1); // Returns the input. Useful for copying registers.
const uint8_t OPNop =uint8_t((0*64)+63); // No operation.
const uint8_t OPStop =uint8_t((1*64)+63); // Stops execution of the tape and returns 0.
// Fidget VM compat
// Two parameter
const uint8_t OPAdd =uint8_t(2); // Adds a vector to a vector component-wise.
const uint8_t OPSub =uint8_t(3); // Subtracts a vector from a vector component-wise.
const uint8_t OPMul =uint8_t(4); // Multiplies a vector and a vector component-wise.
const uint8_t OPDiv =uint8_t(5); // Divides a vector by a vector component-wise.
const uint8_t OPAtan2 =uint8_t(6); // Calculates a vector Atan2 a vector component-wise.
const uint8_t OPMin =uint8_t(7); // Calculates the minimum of a vector and a vector component-wise.
const uint8_t OPMax =uint8_t(8); // Calculates the maximum of a vector and a vector component-wise.
const uint8_t OPCompare =uint8_t(9); // Threeway comparison operator.
const uint8_t OPMod =uint8_t(10); // Calculates a vector modulo a vector component-wise.
const uint8_t OPAnd =uint8_t(11); // If both arguments are non-zero, returns the right-hand argument. Otherwise, returns zero.
const uint8_t OPOr =uint8_t(12); // If the left-hand argument is non-zero, it is returned. Otherwise, the right-hand argument is returned.
// One parameter
const uint8_t OPNegate =uint8_t(13); // Returns the negation of all components of a vector.
const uint8_t OPAbs =uint8_t(14); // Returns the absolute value of all components of a vector.
const uint8_t OPRecip =uint8_t(15); // Returns 1 over all components of a vector.
const uint8_t OPSqrt =uint8_t(16); // Returns the square root of all components of a vector.
const uint8_t OPSquare =uint8_t(17); // Returns the square of all components of a vector.
const uint8_t OPFloor =uint8_t(18); // Returns the floor of all components of a vector.
const uint8_t OPCeil =uint8_t(19); // Returns the ceiling of all components of a vector.
const uint8_t OPRound =uint8_t(20); // Returns all components of a vector rounded to the nearest integer, 0.5 away from zero.
const uint8_t OPSin =uint8_t(21); // Returns the sine of all components of a vector.
const uint8_t OPCos =uint8_t(22); // Returns the cosine of all components of a vector.
const uint8_t OPTan =uint8_t(23); // Returns the tangent of all components of a vector.
const uint8_t OPAsin =uint8_t(24); // Returns the arc sine of all components of a vector.
const uint8_t OPAcos =uint8_t(25); // Returns the arc cosine of all components of a vector.
const uint8_t OPAtan =uint8_t(26); // Returns the arc tangent of all components of a vector.
const uint8_t OPExp =uint8_t(27); // Returns e raised to all components of a vector.
const uint8_t OPLog =uint8_t(28); // Returns the natural logarithm of all components of a vector.
const uint8_t OPNot =uint8_t(29); // The output is 1 if the argument is 0, and 0 otherwise.
// Additional
// One Parameter
const uint8_t OPFract =uint8_t(30); // Returns the fractional part of all components of a vector.
const uint8_t OPCube =uint8_t(31); // Returns the cube of all components of a vector.
// Three parameter
const uint8_t OPSmoothMin =uint8_t(32); // Returns the smooth minimum between a vector and a vector, varied by a vector.
const uint8_t OPSmoothMax =uint8_t(33); // Returns the smooth maximum between a vector and a vector, varied by a vector.
const uint8_t OPClamp =uint8_t(34); // Clamps a vector between a vector and a vector.
const uint8_t OPMix =uint8_t(35); // Mixes between a vector and a vector, varied by a vector.
const uint8_t OPFMA =uint8_t(36); // Calculates a vector multiplied by a vector, then adds a vector.
// Non-element wise
// One parameter
const uint8_t OPLength =uint8_t(37); // Returns the length (magnitude) of a vector.
const uint8_t OPNormalize =uint8_t(38); // Returns the normalised version of a vector.
// Two parameter
const uint8_t OPDot =uint8_t(39); // Returns the dot product of two vectors.
const uint8_t OPDistance =uint8_t(40); // Returns the length (magnitude) of the vector between two vectors.
// Bookkeeping
const uint8_t OPNop =uint8_t((3*64)+63); // No operation.
const uint8_t OPReturn =uint8_t((2*64)+63); // Stops execution of the tape and returns a single value.
const uint8_t OPPosition =uint8_t((3*64)+63); // Returns the current position being sampled.
const uint8_t OPMinMaterial =uint8_t((0*64)+62); // Calculates the minimum of two Vec1s, and also carries over the relevant material metadata.
const uint8_t OPMaxMaterial =uint8_t((1*64)+62); // Calculates the maximum of two Vec1s, and also carries over the relevant material metadata.
const uint8_t OPPosition =uint8_t((1*64)+63); // Returns the current position being sampled.
// Special
const uint8_t OPMinMaterial =uint8_t((0*64)+63); // Calculates the minimum of two Vec1s, and also carries over the relevant material metadata.
const uint8_t OPMaxMaterial =uint8_t((3*64)+62); // Calculates the maximum of two Vec1s, and also carries over the relevant material metadata.
const uint8_t OPSmoothMinMaterial =uint8_t((2*64)+62); // Returns the smooth minimum between a Vec1 and a Vec1, varied by a Vec1, and also carries over the relevant material metadata.
const uint8_t OPSmoothMaxMaterial =uint8_t((3*64)+62); // Returns the smooth maximum between a Vec1 and a Vec1, varied by a Vec1, and also carries over the relevant material metadata.
const uint8_t OPCross =uint8_t((0*64)+61); // Returns the cross product of two Vec3s.
const uint8_t OPSDFSphere =uint8_t((1*64)+61); // Returns the distance to a sphere.
const uint8_t OPSmoothMaxMaterial =uint8_t((1*64)+62); // Returns the smooth maximum between a Vec1 and a Vec1, varied by a Vec1, and also carries over the relevant material metadata.
const uint8_t OPCross =uint8_t((0*64)+62); // Returns the cross product of two Vec3s.
// SDFs
const uint8_t OPSDFSphere =uint8_t((3*64)+61); // Returns the distance to a sphere.
const uint8_t OPSDFBox =uint8_t((2*64)+61); // Returns the distance to a box.
const uint8_t OPSDFTorus =uint8_t((3*64)+61); // Returns the distance to a torus.
const uint8_t OPSDFTorus =uint8_t((1*64)+61); // Returns the distance to a torus.
#endif
+183 -150
View File
@@ -18,7 +18,7 @@ float load(uint8_t reg) {
if (reg == 0) {
return 0.;
} else if (reg == 15) {
return load_const();
return next_const();
} else {
return registers[reg - 1];
}
@@ -31,136 +31,137 @@ void store(uint8_t reg, float value) {
}
float input_float() {
return load(load_register());
return load(next_register());
}
vec2 input_vec2() {
return vec2(load(load_register()), load(load_register()));
return vec2(load(next_register()), load(next_register()));
}
vec3 input_vec3() {
return vec3(load(load_register()), load(load_register()), load(load_register()));
return vec3(load(next_register()), load(next_register()), load(next_register()));
}
vec4 input_vec4() {
return vec4(load(load_register()), load(load_register()), load(load_register()), load(load_register()));
return vec4(load(next_register()), load(next_register()), load(next_register()), load(next_register()));
}
void output_float(float v) {
store(load_register(), v);
store(next_register(), v);
}
void output_vec2(vec2 v) {
store(load_register(), v.x);
store(load_register(), v.y);
store(next_register(), v.x);
store(next_register(), v.y);
}
void output_vec3(vec3 v) {
store(load_register(), v.x);
store(load_register(), v.y);
store(load_register(), v.z);
store(next_register(), v.x);
store(next_register(), v.y);
store(next_register(), v.z);
}
void output_vec4(vec4 v) {
store(load_register(), v.x);
store(load_register(), v.y);
store(load_register(), v.z);
store(load_register(), v.w);
store(next_register(), v.x);
store(next_register(), v.y);
store(next_register(), v.z);
store(next_register(), v.w);
}
#define ewise_vec4_one(func) \
vec4 input1 = input_vec4(); \
output_vec4(vec4(func(input1.x), func(input1.y), func(input1.z), func(input1.w)));
#define ewise_vec3_one(func) \
vec3 input1 = input_vec3(); \
output_vec3(vec3(func(input1.x), func(input1.y), func(input1.z)));
#define ewise_one(len, func) for (int i = 0; i < len; i++) { \
switch ((mask[program_counter] >> (i * 4)) & 3) { \
case MASK_EXECUTE: \
output_float(func(input_float())); \
break; \
case MASK_COPY_LEFT: \
output_float(input_float()); \
break; \
case MASK_COPY_RIGHT : \
case MASK_DO_NOTHING : \
jump_registers(2); \
jump_const(uint(mask[program_counter]>>((i*4)+2))&3); \
break ; \
} \
}
#define ewise_vec2_one(func) \
vec2 input1 = input_vec2(); \
output_vec2(vec2(func(input1.x), func(input1.y)));
#define ewise_two(len, func) for (int i = 0; i < len; i++) { \
switch ((mask[program_counter] >> (i * 4)) & 3) { \
case MASK_EXECUTE: \
output_float(func(input_float(), input_float())); \
break; \
case MASK_COPY_LEFT: \
float inp = input_float(); \
jump_registers(1); \
jump_const(uint(mask[program_counter] >> ((i * 4) + 2)) & 3); \
output_float(inp); \
break; \
case MASK_COPY_RIGHT: \
jump_registers(1); \
jump_const(uint(mask[program_counter] >> ((i * 4) + 2)) & 3); \
output_float(input_float()); \
break; \
case MASK_DO_NOTHING: \
jump_registers(3); \
jump_const(uint(mask[program_counter] >> ((i * 4) + 2)) & 3); \
break; \
} \
}
#define ewise_vec1_one(func) \
float input1 = input_float(); \
output_float(func(input1));
#define ewise_three(len, func) for (int i = 0; i < len; i++) { \
switch ((mask[program_counter] >> (i * 4)) & 3) { \
case MASK_EXECUTE: \
output_float(func(input_float(), input_float(), input_float())); \
break; \
case MASK_COPY_LEFT: {\
float inp = input_float(); \
jump_registers(2); \
jump_const(uint(mask[program_counter] >> ((i * 4) + 2)) & 3); \
output_float(inp); }\
break; \
case MASK_COPY_RIGHT: {\
input_float(); \
float inp = input_float(); \
input_float(); \
output_float(inp); }\
break; \
case MASK_DO_NOTHING: \
jump_registers(4); \
jump_const(uint(mask[program_counter] >> ((i * 4) + 2)) & 3); \
break; \
} \
}
#define ewise_vec4_two(func) \
vec4 input1 = input_vec4(); \
vec4 input2 = input_vec4(); \
output_vec4(vec4(func(input1.x, input2.x), func(input1.y, input2.y), func(input1.z, input2.z), func(input1.w, input2.w)));
#define ewise_vec3_two(func) \
vec3 input1 = input_vec3(); \
vec3 input2 = input_vec3(); \
output_vec3(vec3(func(input1.x, input2.x), func(input1.y, input2.y), func(input1.z, input2.z)));
#define ewise_vec2_two(func) \
vec2 input1 = input_vec2(); \
vec2 input2 = input_vec2(); \
output_vec2(vec2(func(input1.x, input2.x), func(input1.y, input2.y)));
#define ewise_vec1_two(func) \
float input1 = input_float(); \
float input2 = input_float(); \
output_float(func(input1, input2));
#define ewise_vec4_three(func) \
vec4 input1 = input_vec4(); \
vec4 input2 = input_vec4(); \
vec4 input3 = input_vec4(); \
output_vec4(vec4(func(input1.x, input2.x, input3.x), func(input1.y, input2.y, input3.y), func(input1.z, input2.z, input3.z), func(input1.w, input2.w, input3.w)));
#define ewise_vec3_three(func) \
vec3 input1 = input_vec3(); \
vec3 input2 = input_vec3(); \
vec3 input3 = input_vec3(); \
output_vec3(vec3(func(input1.x, input2.x, input3.x), func(input1.y, input2.y, input3.y), func(input1.z, input2.z, input3.z)));
#define ewise_vec2_three(func) \
vec2 input1 = input_vec2(); \
vec2 input2 = input_vec2(); \
vec2 input3 = input_vec2(); \
output_vec2(vec2(func(input1.x, input2.x, input3.x), func(input1.y, input2.y, input3.y)));
#define ewise_vec1_three(func) \
float input1 = input_float(); \
float input2 = input_float(); \
float input3 = input_float(); \
output_float(func(input1, input2, input3));
#define ewise_vec4_four(func) \
vec4 input1 = input_vec4(); \
vec4 input2 = input_vec4(); \
vec4 input3 = input_vec4(); \
vec4 input4 = input_vec4(); \
output_vec4(vec4(func(input1.x, input2.x, input3.x, input4.x), func(input1.y, input2.y, input3.y, input4.y), func(input1.z, input2.z, input3.z, input4.z), func(input1.w, input2.w, input3.z, input4.w)));
#define ewise_vec3_four(func) \
vec3 input1 = input_vec3(); \
vec3 input2 = input_vec3(); \
vec3 input3 = input_vec3(); \
vec3 input4 = input_vec3(); \
output_vec3(vec3(func(input1.x, input2.x, input3.x, input4.x), func(input1.y, input2.y, input3.y, input4.y), func(input1.z, input2.z, input3.z, input4.z)));
#define ewise_vec2_four(func) \
vec2 input1 = input_vec2(); \
vec2 input2 = input_vec2(); \
vec2 input3 = input_vec2(); \
vec2 input4 = input_vec2(); \
output_vec2(vec2(func(input1.x, input2.x, input3.x, input4.x), func(input1.y, input2.y, input3.y, input4.y)));
#define ewise_vec1_four(func) \
float input1 = input_float(); \
float input2 = input_float(); \
float input3 = input_float(); \
float input4 = input_float(); \
output_float(func(input1, input2, input3, input4));
#define ewise_four(len, func) for (int i = 0; i < len; i++) { \
switch ((mask[program_counter] >> (i * 4)) & 3) { \
case MASK_EXECUTE: \
output_float(func(input_float(), input_float(), input_float(), input_float())); \
break; \
case MASK_COPY_LEFT: {\
float inp = input_float(); \
jump_registers(3); \
jump_const(uint(mask[program_counter] >> ((i * 4) + 2)) & 3); \
output_float(inp); }\
break; \
case MASK_COPY_RIGHT: {\
input_float(); \
float inp = input_float(); \
input_float(); \
input_float(); \
output_float(inp); }\
break; \
case MASK_DO_NOTHING: \
jump_registers(5); \
jump_const(uint(mask[program_counter] >> ((i * 4) + 2)) & 3); \
break; \
} \
}
#define ewise_all(opcode, count, func) \
case opcode ## Vec1: {ewise_vec1_ ## count(func);} break; \
case opcode ## Vec2: {ewise_vec2_ ## count(func);} break; \
case opcode ## Vec3: {ewise_vec3_ ## count(func);} break; \
case opcode ## Vec4: {ewise_vec4_ ## count(func);} break; \
case opcode ## Vec1: {ewise_ ## count(1, func);} break; \
case opcode ## Vec2: {ewise_ ## count(2, func);} break; \
case opcode ## Vec3: {ewise_ ## count(3, func);} break; \
case opcode ## Vec4: {ewise_ ## count(4, func);} break; \
//monotonic
float copyof(float in1)
@@ -197,43 +198,38 @@ float modof(float in1, float in2)
return mod(in1, in2);
}
float remof(float in1, float in2)
{
return mod(in1, in2);
}
//always monotonic for x>0
float powof(float in1, float in2)
{
return pow(in1, in2);
}
float opSmoothUnion( float d1, float d2, float k )
float opSmoothUnion(float d1, float d2, float k)
{
float h = clamp( 0.5 + 0.5*(d2-d1)/k, 0.0, 1.0 );
return mix( d2, d1, h ) - k*h*(1.0-h);
float h = clamp(0.5 + 0.5 * (d2 - d1) / k, 0.0, 1.0);
return mix(d2, d1, h) - k * h * (1.0 - h);
}
float opSmoothSubtraction( float d1, float d2, float k )
float opSmoothSubtraction(float d1, float d2, float k)
{
float h = clamp( 0.5 - 0.5*(d2+d1)/k, 0.0, 1.0 );
return mix( d2, -d1, h ) + k*h*(1.0-h);
float h = clamp(0.5 - 0.5 * (d2 + d1) / k, 0.0, 1.0);
return mix(d2, -d1, h) + k * h * (1.0 - h);
}
float opSmoothIntersection( float d1, float d2, float k )
float opSmoothIntersection(float d1, float d2, float k)
{
float h = clamp( 0.5 - 0.5*(d2-d1)/k, 0.0, 1.0 );
return mix( d2, d1, h ) + k*h*(1.0-h);
float h = clamp(0.5 - 0.5 * (d2 - d1) / k, 0.0, 1.0);
return mix(d2, d1, h) + k * h * (1.0 - h);
}
//monotonic
float stepof( float d1, float d2 )
float stepof(float d1, float d2)
{
return step(d1, d2);
}
//monotonic
float smoothstepof( float d1, float d2, float k )
float smoothstepof(float d1, float d2, float k)
{
return smoothstep(d1, d2, k);
}
@@ -450,10 +446,52 @@ float truncof(float in1)
return trunc(in1);
}
//handled
float recipof(float in1)
{
return 1.0 / in1;
}
//handled
float compareof(float in1, float in2)
{
if (isnan(in1)) {
return in1;
}
if (isnan(in2)) {
return in2;
}
if (in1 < in2) {
return -1.;
}
if (in1 > in2) {
return 1.;
}
return 0.;
}
//handled
float orof(float in1, float in2)
{
return mix(in2, in1, in1 == 0.);
}
//handled
float andof(float in1, float in2)
{
return mix(in1, in2, in1 == 0.);
}
//handled
float notof(float in1)
{
return mix(1., 0., in1 == 0.);
}
vec3 scene(vec3 p, bool materials)
{
uint program_counter = 0;
uint nibble_counter = 0;
uint io_counter = 0;
uint const_counter = 0;
desc = scene_description.desc[(DescriptionIndex) + 1];
@@ -461,8 +499,16 @@ vec3 scene(vec3 p, bool materials)
clear_registers();
while (program_counter < EXECUTION_LIMIT) {
uint8_t code;
code = load_opcode();
uint8_t code = next_opcode();
uint[2] elements = reg_elementwise(code);
if (elements[0] == 0 && (mask[program_counter] & 3) != MASK_EXECUTE) {
uint io = uint(reg_input(code) + reg_output(code));
jump_registers(io);
jump_const(uint(mask[program_counter] >> 2));
continue;
}
switch (uint32_t(code))
{
ewise_all(OPCopy, one, copyof);
@@ -471,49 +517,38 @@ vec3 scene(vec3 p, bool materials)
ewise_all(OPSub, two, subof);
ewise_all(OPMul, two, mulof);
ewise_all(OPDiv, two, divof);
ewise_all(OPMod, two, modof);
ewise_all(OPRem, two, remof);
ewise_all(OPPow, two, powof);
ewise_all(OPAtan2, two, atan2of);
ewise_all(OPMin, two, minof);
ewise_all(OPMax, two, maxof);
ewise_all(OPStep, two, stepof);
ewise_all(OPCompare, two, compareof);
ewise_all(OPMod, two, modof);
ewise_all(OPAnd, two, andof);
ewise_all(OPOr, two, orof);
ewise_all(OPNegate, one, negateof);
ewise_all(OPRound, one, roundof);
ewise_all(OPRoundEven, one, roundevenof);
ewise_all(OPTrunc, one, truncof);
ewise_all(OPAbs, one, absof);
ewise_all(OPSign, one, signof);
ewise_all(OPRecip, one, recipof);
ewise_all(OPSqrt, one, sqrtof);
ewise_all(OPSquare, one, squareof);
ewise_all(OPFloor, one, floorof);
ewise_all(OPCeil, one, ceilof);
ewise_all(OPFract, one, fractof);
ewise_all(OPRound, one, roundof);
ewise_all(OPSin, one, sinof);
ewise_all(OPCos, one, cosof);
ewise_all(OPTan, one, tanof);
ewise_all(OPAsin, one, asinof);
ewise_all(OPAcos, one, acosof);
ewise_all(OPAtan, one, atanof);
ewise_all(OPSinh, one, sinhof);
ewise_all(OPCosh, one, coshof);
ewise_all(OPTanh, one, tanhof);
ewise_all(OPAsinh, one, asinhof);
ewise_all(OPAcosh, one, acoshof);
ewise_all(OPAtanh, one, atanhof);
ewise_all(OPExp, one, expof);
ewise_all(OPLog, one, logof);
ewise_all(OPExp2, one, exp2of);
ewise_all(OPLog2, one, log2of);
ewise_all(OPSqrt, one, sqrtof);
ewise_all(OPInverseSqrt, one, inversesqrtof);
ewise_all(OPSquare, one, squareof);
ewise_all(OPCube, one, cubeof);
ewise_all(OPNot, one, notof);
ewise_all(OPFract, one, fractof);
ewise_all(OPCube, one, cubeof);
ewise_all(OPSmoothMin, three, opSmoothUnion);
ewise_all(OPSmoothMax, three, opSmoothIntersection);
ewise_all(OPClamp, three, clampof);
ewise_all(OPMix, three, mixof);
ewise_all(OPSmoothStep, three, smoothstepof);
ewise_all(OPFMA, three, fmaof);
case OPDotVec1:
@@ -634,22 +669,22 @@ vec3 scene(vec3 p, bool materials)
case OPSmoothMinMaterial:
{
ewise_vec1_three(opSmoothUnion);
ewise_three(1, opSmoothUnion);
}
break;
case OPSmoothMaxMaterial:
{
ewise_vec1_three(opSmoothIntersection);
ewise_three(1, opSmoothIntersection);
}
break;
case OPMinMaterial:
{
ewise_vec1_two(minof);
ewise_two(1, minof);
}
break;
case OPMaxMaterial:
{
ewise_vec1_two(maxof);
ewise_two(1, maxof);
}
break;
@@ -685,8 +720,6 @@ vec3 scene(vec3 p, bool materials)
case OPNop:
break;
case OPStop:
return vec3(-1.);
case OPReturn:
return vec3(input_float());
@@ -704,7 +737,7 @@ vec3 scene(vec3 p, bool materials)
#endif
}
}
return vec3(-1.);
return vec3(input_float());
}
#endif//ifndef interpreter
+38 -75
View File
@@ -1,8 +1,8 @@
use glam::{self, FloatExt, Vec3Swizzles};
use crate::{
ssa::{SSAInput, SSAInstruction, SSAOpcode},
CSG,
ssa::{SSAInput, SSAInstruction, SSAOpcode},
};
#[derive(Clone, Debug)]
@@ -89,9 +89,6 @@ impl<'csg> Interpreter<'csg> {
for instruction in &self.csg.parts.tape {
use SSAOpcode::*;
match instruction.opcode.opcode {
SSAStop => {
return f32::NAN;
},
SSAReturn => {
return self.load(instruction.inputs[0]);
},
@@ -115,12 +112,6 @@ impl<'csg> Interpreter<'csg> {
SSAMod => {
self.param_two(instruction, |val_a, val_b| val_a % val_b);
},
SSARem => {
self.param_two(instruction, |val_a: f32, val_b| val_a % val_b);
},
SSAPow => {
self.param_two(instruction, |val_a: f32, val_b| val_a.powf(val_b));
},
SSAAtan2 => {
self.param_two(instruction, |val_a: f32, val_b| val_a.atan2(val_b));
},
@@ -153,9 +144,7 @@ impl<'csg> Interpreter<'csg> {
let val_b = (instruction.opcode.size..(instruction.opcode.size * 2))
.map(|i| self.load(instruction.inputs[i as usize]))
.collect::<Vec<_>>();
self.store(
instruction.outputs[0],
match instruction.opcode.size {
self.store(instruction.outputs[0], match instruction.opcode.size {
1 => val_a[0] * val_b[0],
2 => glam::vec2(val_a[0], val_a[1]).dot(glam::vec2(val_b[0], val_b[1])),
3 => glam::vec3(val_a[0], val_a[1], val_a[2])
@@ -163,23 +152,19 @@ impl<'csg> Interpreter<'csg> {
4 => glam::vec4(val_a[0], val_a[1], val_a[2], val_a[3])
.dot(glam::vec4(val_b[0], val_b[1], val_b[2], val_b[3])),
_ => unreachable!(),
},
);
});
},
SSALength => {
let val_a = (0..instruction.opcode.size)
.map(|i| self.load(instruction.inputs[i as usize]))
.collect::<Vec<_>>();
self.store(
instruction.outputs[0],
match instruction.opcode.size {
self.store(instruction.outputs[0], match instruction.opcode.size {
1 => val_a[0],
2 => glam::vec2(val_a[0], val_a[1]).length(),
3 => glam::vec3(val_a[0], val_a[1], val_a[2]).length(),
4 => glam::vec4(val_a[0], val_a[1], val_a[2], val_a[3]).length(),
_ => unreachable!(),
},
);
});
},
SSADistance => {
let val_a = (0..instruction.opcode.size)
@@ -188,19 +173,17 @@ impl<'csg> Interpreter<'csg> {
let val_b = (instruction.opcode.size..(instruction.opcode.size * 2))
.map(|i| self.load(instruction.inputs[i as usize]))
.collect::<Vec<_>>();
self.store(
instruction.outputs[0],
match instruction.opcode.size {
self.store(instruction.outputs[0], match instruction.opcode.size {
1 => val_a[0] - val_b[0],
2 => glam::vec2(val_a[0], val_a[1])
.distance(glam::vec2(val_b[0], val_b[1])),
2 => {
glam::vec2(val_a[0], val_a[1]).distance(glam::vec2(val_b[0], val_b[1]))
},
3 => glam::vec3(val_a[0], val_a[1], val_a[2])
.distance(glam::vec3(val_b[0], val_b[1], val_b[2])),
4 => glam::vec4(val_a[0], val_a[1], val_a[2], val_a[3])
.distance(glam::vec4(val_b[0], val_b[1], val_b[2], val_b[3])),
_ => unreachable!(),
},
);
});
},
SSANormalize => {
let val_a = (0..instruction.opcode.size)
@@ -239,18 +222,9 @@ impl<'csg> Interpreter<'csg> {
SSARound => {
self.param_one(instruction, |val_a: f32| val_a.round());
},
SSARoundEven => {
self.param_one(instruction, |val_a: f32| val_a.round_ties_even());
},
SSATrunc => {
self.param_one(instruction, |val_a: f32| val_a.trunc());
},
SSAAbs => {
self.param_one(instruction, |val_a: f32| val_a.abs());
},
SSASign => {
self.param_one(instruction, |val_a: f32| sign(val_a));
},
SSAFloor => {
self.param_one(instruction, |val_a: f32| val_a.floor());
},
@@ -278,42 +252,15 @@ impl<'csg> Interpreter<'csg> {
SSAAtan => {
self.param_one(instruction, |val_a: f32| val_a.atan());
},
SSASinh => {
self.param_one(instruction, |val_a: f32| val_a.sinh());
},
SSACosh => {
self.param_one(instruction, |val_a: f32| val_a.cosh());
},
SSATanh => {
self.param_one(instruction, |val_a: f32| val_a.tanh());
},
SSAAsinh => {
self.param_one(instruction, |val_a: f32| val_a.asinh());
},
SSAAcosh => {
self.param_one(instruction, |val_a: f32| val_a.acosh());
},
SSAAtanh => {
self.param_one(instruction, |val_a: f32| val_a.atanh());
},
SSAExp => {
self.param_one(instruction, |val_a: f32| val_a.exp());
},
SSALog => {
self.param_one(instruction, |val_a: f32| val_a.ln());
},
SSAExp2 => {
self.param_one(instruction, |val_a: f32| val_a.exp2());
},
SSALog2 => {
self.param_one(instruction, |val_a: f32| val_a.log2());
},
SSASqrt => {
self.param_one(instruction, |val_a: f32| val_a.sqrt());
},
SSAInverseSqrt => {
self.param_one(instruction, |val_a: f32| 1.0 / val_a.sqrt());
},
SSASquare => {
self.param_one(instruction, |val_a: f32| val_a * val_a);
},
@@ -340,18 +287,6 @@ impl<'csg> Interpreter<'csg> {
SSAMix => {
self.param_three(instruction, |val_a, val_b, val_c| val_a.lerp(val_b, val_c));
},
SSAStep => {
self.param_two(
instruction,
|val_a, val_b| if val_a < val_b { 0. } else { 1. },
);
},
SSASmoothStep => {
self.param_three(instruction, |x, edge0, edge1| {
let t = ((x - edge0) / (edge1 - edge0)).clamp(0., 1.);
t * t * (3. - 2. * t)
});
},
SSAFMA => {
self.param_three(instruction, |val_a, val_b, val_c| {
val_a.mul_add(val_b, val_c)
@@ -390,6 +325,34 @@ impl<'csg> Interpreter<'csg> {
let q = glam::vec2(p.xz().length() - radius1, p.y);
self.store(instruction.outputs[0], q.length() - radius2);
},
SSACompare => {
self.param_two(instruction, |val_a: f32, val_b: f32| {
match val_a.total_cmp(&val_b) {
std::cmp::Ordering::Less => -1.,
std::cmp::Ordering::Equal => 0.,
std::cmp::Ordering::Greater => 1.,
}
});
},
SSAAnd => {
self.param_two(
instruction,
|val_a: f32, val_b: f32| if val_a == 0. { val_a } else { val_b },
);
},
SSAOr => {
self.param_two(
instruction,
|val_a: f32, val_b: f32| if val_a == 0. { val_b } else { val_a },
);
},
SSARecip => {
self.param_one(instruction, |val_a: f32| val_a.recip());
},
SSANot => {
self.param_one(instruction, |val_a: f32| if val_a == 0. { 1. } else { 0. });
},
SSAStop => return 0.,
}
}
return f32::NAN;
+726 -2717
View File
File diff suppressed because it is too large Load Diff
+140 -238
View File
@@ -1,7 +1,9 @@
#![feature(variant_count)]
#![feature(array_chunks)]
use std::{
error::Error,
fs::{remove_file, rename, File},
fs::{File, remove_file, rename},
io::{Cursor, Read, Write},
path::{Path, PathBuf},
sync::Arc,
@@ -16,41 +18,44 @@ const SAMPLE_RATE_SHADING: f32 = 1.0;
const MSAA_SAMPLES_ACTUAL: u32 = if MSAA_ENABLE { MSAA_SAMPLES } else { 1 };
use bytemuck::{Pod, Zeroable};
use egui_winit_vulkano::{Gui, GuiConfig};
use foldhash::{HashMap, HashMapExt, HashSet};
use glam::{self, vec3, EulerRot, Mat3, Mat4, Vec3};
use glam::{self, EulerRot, Mat3, Mat4, Vec3, vec3};
use log::{error, info, trace};
use rayon::prelude::*;
use simplelog::{CombinedLogger, Config, TermLogger, WriteLogger};
use ssa::{SSAInput, SSAOpcode, SSAOpcodeSized, SSATape};
use vulkano::{
Validated, Version, VulkanError, VulkanLibrary,
buffer::{
Buffer, BufferContents, BufferCreateInfo, BufferUsage, Subbuffer,
allocator::{SubbufferAllocator, SubbufferAllocatorCreateInfo},
Buffer, BufferCreateInfo, BufferUsage, Subbuffer,
},
command_buffer::{
allocator::StandardCommandBufferAllocator, AutoCommandBufferBuilder, CommandBufferUsage,
CopyBufferInfo, PrimaryCommandBufferAbstract, RenderPassBeginInfo, SubpassBeginInfo,
SubpassContents,
AutoCommandBufferBuilder, CommandBufferUsage, CopyBufferInfo, PrimaryCommandBufferAbstract,
RenderPassBeginInfo, SubpassBeginInfo, SubpassContents,
allocator::StandardCommandBufferAllocator,
},
descriptor_set::{
allocator::StandardDescriptorSetAllocator, DescriptorSet, WriteDescriptorSet,
DescriptorSet, WriteDescriptorSet, allocator::StandardDescriptorSetAllocator,
},
device::{
physical::PhysicalDeviceType, Device, DeviceCreateInfo, DeviceExtensions, DeviceFeatures,
DeviceOwned, Queue, QueueCreateInfo, QueueFlags,
Device, DeviceCreateInfo, DeviceExtensions, DeviceFeatures, DeviceOwned, Queue,
QueueCreateInfo, QueueFlags, physical::PhysicalDeviceType,
},
format::Format,
image::{view::ImageView, Image, ImageCreateInfo, ImageType, ImageUsage},
image::{Image, ImageCreateInfo, ImageType, ImageUsage, view::ImageView},
instance::{Instance, InstanceCreateInfo, InstanceExtensions},
memory::allocator::{
AllocationCreateInfo, MemoryAllocatePreference, MemoryTypeFilter, StandardMemoryAllocator,
},
pipeline::{
ComputePipeline, DynamicState, GraphicsPipeline, Pipeline, PipelineBindPoint,
PipelineLayout, PipelineShaderStageCreateInfo,
cache::{PipelineCache, PipelineCacheCreateInfo},
compute::ComputePipelineCreateInfo,
graphics::{
GraphicsPipelineCreateInfo,
color_blend::{ColorBlendAttachmentState, ColorBlendState},
depth_stencil::{DepthState, DepthStencilState},
input_assembly::InputAssemblyState,
@@ -58,20 +63,16 @@ use vulkano::{
rasterization::{CullMode, FrontFace, RasterizationState},
vertex_input::{Vertex, VertexDefinition},
viewport::{Viewport, ViewportState},
GraphicsPipelineCreateInfo,
},
layout::PipelineDescriptorSetLayoutCreateInfo,
ComputePipeline, DynamicState, GraphicsPipeline, Pipeline, PipelineBindPoint,
PipelineLayout, PipelineShaderStageCreateInfo,
},
render_pass::{Framebuffer, FramebufferCreateInfo, RenderPass, Subpass},
shader::{ShaderModule, SpecializationConstant},
swapchain::{
acquire_next_image, PresentMode, Surface, SurfaceInfo, Swapchain, SwapchainCreateInfo,
SwapchainPresentInfo,
PresentMode, Surface, SurfaceInfo, Swapchain, SwapchainCreateInfo, SwapchainPresentInfo,
acquire_next_image,
},
sync::{self, GpuFuture},
Validated, Version, VulkanError, VulkanLibrary,
};
use winit::{
application::ApplicationHandler,
@@ -205,9 +206,7 @@ impl App {
let library = VulkanLibrary::new().expect("Vulkan is not installed???");
let required_extensions = Surface::required_extensions(event_loop).unwrap();
let instance = Instance::new(
library,
InstanceCreateInfo {
let instance = Instance::new(library, InstanceCreateInfo {
enabled_extensions: InstanceExtensions {
ext_surface_maintenance1: true,
..required_extensions
@@ -217,8 +216,7 @@ impl App {
application_name: Some(env!("CARGO_PKG_NAME").to_owned()),
application_version: app_version(),
..Default::default()
},
)
})
.unwrap();
let mut device_extensions = DeviceExtensions {
@@ -292,9 +290,7 @@ impl App {
device_extensions.khr_dynamic_rendering = true;
}
let (device, mut queues) = Device::new(
physical_device,
DeviceCreateInfo {
let (device, mut queues) = Device::new(physical_device, DeviceCreateInfo {
enabled_extensions: device_extensions,
queue_create_infos: vec![
QueueCreateInfo {
@@ -314,6 +310,7 @@ impl App {
shader_int16: true,
shader_int8: true,
storage_buffer8_bit_access: true,
storage_buffer16_bit_access: true,
geometry_shader: true,
primitive_fragment_shading_rate: true,
maintenance4: true,
@@ -321,8 +318,7 @@ impl App {
..DeviceFeatures::empty()
},
..Default::default()
},
)
})
.expect("Unable to initialize device");
let graphics_queue = queues.next().expect("Unable to retrieve queues");
@@ -338,15 +334,13 @@ impl App {
Default::default(),
));
let uniform_buffer_allocator = SubbufferAllocator::new(
memory_allocator.clone(),
SubbufferAllocatorCreateInfo {
let uniform_buffer_allocator =
SubbufferAllocator::new(memory_allocator.clone(), SubbufferAllocatorCreateInfo {
buffer_usage: BufferUsage::UNIFORM_BUFFER | BufferUsage::STORAGE_BUFFER,
memory_type_filter: MemoryTypeFilter::PREFER_DEVICE
| MemoryTypeFilter::HOST_SEQUENTIAL_WRITE,
..Default::default()
},
);
});
let pipeline_cache = get_pipeline_cache(device.clone());
@@ -502,7 +496,7 @@ mod implicit_fs {
vulkan_version: "1.3",
spirv_version: "1.6",
define: [("implicit","1")],
custom_derives: [Debug, Clone, Copy],
custom_derives: [Debug, Clone, Copy, Default],
}
}
@@ -552,13 +546,10 @@ impl ApplicationHandler for App {
let surface_capabilities = self
.device
.physical_device()
.surface_capabilities(
&surface,
SurfaceInfo {
.surface_capabilities(&surface, SurfaceInfo {
present_mode: Some(present_mode),
..Default::default()
},
)
})
.unwrap();
let (image_format, _) = self
@@ -567,10 +558,7 @@ impl ApplicationHandler for App {
.surface_formats(&surface, Default::default())
.unwrap()[0];
Swapchain::new(
self.device.clone(),
surface.clone(),
SwapchainCreateInfo {
Swapchain::new(self.device.clone(), surface.clone(), SwapchainCreateInfo {
min_image_count: 3
.max(surface_capabilities.min_image_count)
.min(surface_capabilities.max_image_count.unwrap_or(u32::MAX)),
@@ -589,8 +577,7 @@ impl ApplicationHandler for App {
present_mode,
..Default::default()
},
)
})
.unwrap()
};
@@ -759,8 +746,6 @@ impl ApplicationHandler for App {
&self.gstate.csg,
self.command_buffer_allocator.clone(),
self.transfer_queue.clone(),
None,
false,
);
let viewport = Viewport {
@@ -1081,38 +1066,20 @@ impl App {
self.descriptor_set_allocator.clone(),
implicit_layout.clone(),
[
WriteDescriptorSet::buffer(0, uniform_buffer_subbuffer),
WriteDescriptorSet::buffer(1, cam_set),
WriteDescriptorSet::buffer(2, rcx.subbuffers.desc.clone()),
WriteDescriptorSet::buffer(3, rcx.subbuffers.scene.clone()),
WriteDescriptorSet::buffer(4, rcx.subbuffers.floats.clone()),
WriteDescriptorSet::buffer(5, rcx.subbuffers.vec2s.clone()),
//WriteDescriptorSet::buffer(6, rcx.subbuffers.vec3s.clone()),
WriteDescriptorSet::buffer(7, rcx.subbuffers.vec4s.clone()),
//WriteDescriptorSet::buffer(8, rcx.subbuffers.mat2s.clone()),
//WriteDescriptorSet::buffer(9, rcx.subbuffers.mat3s.clone()),
//WriteDescriptorSet::buffer(10, rcx.subbuffers.mat4s.clone()),
//WriteDescriptorSet::buffer(11, rcx.subbuffers.mats.clone()),
WriteDescriptorSet::buffer(12, rcx.subbuffers.deps.clone()),
WriteDescriptorSet::buffer(20, rcx.subbuffers.masks.clone()),
WriteDescriptorSet::buffer(0, rcx.subbuffers.desc.clone()),
WriteDescriptorSet::buffer(1, rcx.subbuffers.scene.clone()),
WriteDescriptorSet::buffer(2, rcx.subbuffers.masks.clone()),
],
[],
)
.unwrap();
if COMPUTE_FUZZING {
let mut fake_csg = vec![];
for i in 0..1 {
fake_csg.push()
}
let (compute_subbuffers, scene) = object_size_dependent_setup(
self.memory_allocator.clone(),
&fake_csg,
&self.gstate.csg,
self.command_buffer_allocator.clone(),
self.transfer_queue.clone(),
Some([1., 1., 1., 1., 5., 1.]),
true,
);
let compute_result_buffer: Subbuffer<[cs::Results]> = self
@@ -1177,22 +1144,8 @@ impl App {
self.descriptor_set_allocator.clone(),
compute_layout.clone(),
[
//WriteDescriptorSet::buffer(0, uniform_buffer_subbuffer.clone()),
//WriteDescriptorSet::buffer(1, cam_set.clone()),
WriteDescriptorSet::buffer(2, compute_subbuffers.desc.clone()),
WriteDescriptorSet::buffer(3, compute_subbuffers.scene.clone()),
//WriteDescriptorSet::buffer(4, compute_subbuffers.floats.clone()),
//WriteDescriptorSet::buffer(5, compute_subbuffers.vec2s.clone()),
//WriteDescriptorSet::buffer(6, compute_subbuffers.vec3s.clone()),
//WriteDescriptorSet::buffer(7, compute_subbuffers.vec4s.clone()),
//WriteDescriptorSet::buffer(8, compute_subbuffers.mat2s.clone()),
//WriteDescriptorSet::buffer(9, compute_subbuffers.mat3s.clone()),
//WriteDescriptorSet::buffer(10,
// compute_subbuffers.mat4s.clone()),
// WriteDescriptorSet::buffer(11, compute_subbuffers.mats.clone()),
//WriteDescriptorSet::buffer(12, compute_subbuffers.deps),
//WriteDescriptorSet::buffer(20,
// compute_subbuffers.masks.clone()),
WriteDescriptorSet::buffer(0, compute_subbuffers.desc.clone()),
WriteDescriptorSet::buffer(1, compute_subbuffers.scene.clone()),
WriteDescriptorSet::buffer(30, compute_result_buffer.clone()),
],
[],
@@ -1212,7 +1165,7 @@ impl App {
.bind_descriptor_sets(
PipelineBindPoint::Compute,
compute_pipeline.layout().clone(),
0, // 0 is the index of our set
1,
compute_set,
)
.unwrap();
@@ -1298,7 +1251,7 @@ impl App {
PipelineBindPoint::Graphics,
rcx.mesh_pipeline.layout().clone(),
0,
mesh_set,
mesh_set.clone(),
)
.unwrap();
@@ -1329,7 +1282,7 @@ impl App {
PipelineBindPoint::Graphics,
rcx.implicit_pipeline.layout().clone(),
0,
implicit_set,
(mesh_set, implicit_set),
)
.unwrap();
@@ -1351,13 +1304,10 @@ impl App {
}
builder
.next_subpass(
Default::default(),
SubpassBeginInfo {
.next_subpass(Default::default(), SubpassBeginInfo {
contents: SubpassContents::SecondaryCommandBuffers,
..Default::default()
},
)
})
.unwrap()
.execute_commands(guicb)
.unwrap()
@@ -1450,9 +1400,7 @@ fn framebuffer_generation(
.map(|image| {
let view = ImageView::new_default(image.clone()).unwrap();
Framebuffer::new(
render_pass.clone(),
FramebufferCreateInfo {
Framebuffer::new(render_pass.clone(), FramebufferCreateInfo {
attachments: if MSAA_ENABLE {
vec![
intermediary.as_ref().unwrap().clone(),
@@ -1463,8 +1411,7 @@ fn framebuffer_generation(
vec![view, depth_buffer.clone()]
},
..Default::default()
},
)
})
.unwrap()
})
.collect::<Vec<_>>();
@@ -1710,53 +1657,24 @@ fn pipeline_recompile(
(mesh_pipeline, implicit_pipeline)
}
#[repr(C)]
#[derive(Clone, Copy, Pod, Zeroable, Default, Debug)]
struct Description {
pointers: [u32; 9],
bounds: [f32; 6],
}
struct Subbuffers {
masks: Subbuffer<[[u8; 29]]>,
floats: Subbuffer<[f32]>,
vec2s: Subbuffer<[[f32; 2]]>,
//vec3s: Subbuffer<[[f32; 4]]>,
vec4s: Subbuffer<[[f32; 4]]>,
mat2s: Subbuffer<[[[f32; 2]; 2]]>,
mat3s: Subbuffer<[[[f32; 3]; 3]]>,
mat4s: Subbuffer<[[[f32; 4]; 4]]>,
mats: Subbuffer<[[[f32; 4]; 4]]>,
masks: Subbuffer<[[u8; 500]]>,
scene: Subbuffer<[[u32; 4]]>,
deps: Subbuffer<[[u8; 2]]>,
desc: Subbuffer<[Description]>,
}
impl PartialEq<InputTypes> for Inputs {
fn eq(&self, other: &InputTypes) -> bool {
match *self {
Inputs::Variable => true,
Inputs::Float(_) => *other == InputTypes::Float,
Inputs::Vec2(_) => *other == InputTypes::Vec2,
Inputs::Vec3(_) => *other == InputTypes::Vec3,
Inputs::Vec4(_) => *other == InputTypes::Vec4,
Inputs::Mat2(_) => *other == InputTypes::Mat2,
Inputs::Mat3(_) => *other == InputTypes::Mat3,
Inputs::Mat4(_) => *other == InputTypes::Mat4,
}
}
desc: Subbuffer<[implicit_fs::Description]>,
}
fn gpu_buffer<T>(
input: Vec<T>,
input: &[&[T]],
allocator: Arc<StandardMemoryAllocator>,
sub_allocator: &SubbufferAllocator,
command_allocator: Arc<StandardCommandBufferAllocator>,
transfer_queue: Arc<Queue>,
) -> Subbuffer<[T]>
where
T: bytemuck::Pod + Send + Sync,
T: BufferContents + Copy,
{
let total_len = input.iter().map(|i| i.len() as u64).sum::<u64>();
let buffer = Buffer::new_slice(
allocator,
BufferCreateInfo {
@@ -1768,12 +1686,19 @@ where
allocate_preference: MemoryAllocatePreference::AlwaysAllocate,
..Default::default()
},
(input.len()) as u64,
total_len,
)
.unwrap();
let staging = sub_allocator.allocate_slice(input.len() as u64).unwrap();
staging.write().unwrap().copy_from_slice(&input[..]);
let staging = sub_allocator.allocate_slice(total_len).unwrap();
let mut writer = staging.write().unwrap();
let mut pointer = 0;
for input in input {
writer[pointer..(pointer + input.len())].copy_from_slice(&input[..]);
pointer += input.len();
}
drop(writer);
let mut builder = AutoCommandBufferBuilder::primary(
command_allocator,
@@ -1802,31 +1727,82 @@ fn object_size_dependent_setup(
state: &Vec<CSG>,
command_allocator: Arc<StandardCommandBufferAllocator>,
queue: Arc<Queue>,
set_bound: Option<[f32; 6]>,
actual: bool,
) -> (Subbuffers, Vec<[u32; 4]>) {
let mut floats: Vec<f32> = vec![Default::default()];
let mut vec2s: Vec<[f32; 2]> = vec![Default::default()];
let mut vec4s: Vec<[f32; 4]> = vec![Default::default()];
let mut mat2s: Vec<[[f32; 2]; 2]> = vec![Default::default()];
let mut mat3s: Vec<[[f32; 3]; 3]> = vec![Default::default()];
let mut mat4s: Vec<[[f32; 4]; 4]> = vec![Default::default()];
let mats: Vec<[[f32; 4]; 4]> = vec![Default::default()];
let mut scene: Vec<[u32; 4]> = vec![Default::default()];
let mut deps: Vec<[u8; 2]> = vec![Default::default()];
let mut desc: Vec<Description> = vec![Default::default()];
let mut scene: Vec<[u32; 4]> = vec![];
let mut desc: Vec<implicit_fs::Description> = vec![Default::default()];
'nextcsg: for csg in state {}
for csg in state {
let tape = csg.parts.compile_to_gpu();
let mut description = implicit_fs::Description::default();
description.scene = (scene.len() / 4) as u32;
let chunks = tape.instructions.array_chunks::<16>();
for opcode in chunks.clone() {
scene.push(
opcode
.array_chunks::<4>()
.map(|smol| u32::from_le_bytes(*smol))
.collect::<Vec<_>>()
.try_into()
.unwrap(),
);
}
if chunks.remainder().len() > 0 {
let mut remainder = [0; 4];
for (i, item) in chunks.remainder().iter().enumerate() {
remainder[i / 4] |= (*item as u32) << ((i % 4) * 8);
}
scene.push(remainder);
}
description.io = (scene.len() / 4) as u32;
let chunks = tape.io.array_chunks::<16>();
for reg in chunks.clone() {
scene.push(
reg.array_chunks::<4>()
.map(|smol| u32::from_le_bytes(*smol))
.collect::<Vec<_>>()
.try_into()
.unwrap(),
);
}
if chunks.remainder().len() > 0 {
let mut remainder = [0; 4];
for (i, item) in chunks.remainder().iter().enumerate() {
remainder[i / 4] |= (*item as u32) << ((i % 4) * 8);
}
scene.push(remainder);
}
description.constants = (scene.len() / 4) as u32;
let chunks = tape.constants.array_chunks::<4>();
for reg in chunks.clone() {
scene.push(reg.map(|f| f.to_bits()));
}
if chunks.remainder().len() > 0 {
let mut remainder = [0; 4];
for (i, item) in chunks.remainder().iter().enumerate() {
remainder[i] = item.to_bits();
}
scene.push(remainder);
}
let mut interpret = interpreter::Interpreter::new(csg);
description.bounds[0] = interpret.scene(vec3(-10000., 0., 0.)) - 10000.0;
description.bounds[1] = interpret.scene(vec3(0., -10000., 0.)) - 10000.0;
description.bounds[2] = interpret.scene(vec3(0., 0., -10000.)) - 10000.0;
description.bounds[3] = 10000.0 - interpret.scene(vec3(10000., 0., 0.));
description.bounds[4] = 10000.0 - interpret.scene(vec3(0., 10000., 0.));
description.bounds[5] = 10000.0 - interpret.scene(vec3(0., 0., 10000.));
desc.push(description);
}
trace!("floats: {floats:?}");
trace!("vec2s: {vec2s:?}");
trace!("vec3/4s: {vec4s:?}");
trace!("mat2s: {mat2s:?}");
trace!("mat3s: {mat3s:?}");
trace!("mat4s: {mat4s:?}");
trace!("mats: {mats:?}");
trace!("scene: {scene:?}");
trace!("deps: {deps:?}");
trace!("desc: {desc:?}");
let fragment_masks_buffer = Buffer::new_slice(
@@ -1840,106 +1816,35 @@ fn object_size_dependent_setup(
allocate_preference: MemoryAllocatePreference::AlwaysAllocate,
..Default::default()
},
((desc.len() - 1) * (4 * 4 * 4) * (4 * 4 * 2) * 29) as u64,
((desc.len() - 1) * (4 * 4 * 4) * (4 * 4 * 2) * 500) as u64,
)
.unwrap();
let staging = SubbufferAllocator::new(
allocator.clone(),
SubbufferAllocatorCreateInfo {
let staging = SubbufferAllocator::new(allocator.clone(), SubbufferAllocatorCreateInfo {
buffer_usage: BufferUsage::TRANSFER_SRC,
memory_type_filter: MemoryTypeFilter::PREFER_HOST
| MemoryTypeFilter::HOST_SEQUENTIAL_WRITE,
memory_type_filter: MemoryTypeFilter::PREFER_HOST | MemoryTypeFilter::HOST_SEQUENTIAL_WRITE,
..Default::default()
},
);
});
let csg_scene = gpu_buffer(
scene.clone(),
&[&scene],
allocator.clone(),
&staging,
command_allocator.clone(),
queue.clone(),
);
let csg_desc = gpu_buffer(
desc,
&[&desc],
allocator.clone(),
&staging,
command_allocator.clone(),
queue.clone(),
);
let csg_floats = gpu_buffer(
floats,
allocator.clone(),
&staging,
command_allocator.clone(),
queue.clone(),
);
let csg_vec2s = gpu_buffer(
vec2s,
allocator.clone(),
&staging,
command_allocator.clone(),
queue.clone(),
);
//let csg_vec3s = gpu_buffer(vec3s, &allocator, &staging, command_allocator,
// queue.clone());
let csg_vec4s = gpu_buffer(
vec4s,
allocator.clone(),
&staging,
command_allocator.clone(),
queue.clone(),
);
let csg_mat2s = gpu_buffer(
mat2s,
allocator.clone(),
&staging,
command_allocator.clone(),
queue.clone(),
);
let csg_mat3s = gpu_buffer(
mat3s,
allocator.clone(),
&staging,
command_allocator.clone(),
queue.clone(),
);
let csg_mat4s = gpu_buffer(
mat4s,
allocator.clone(),
&staging,
command_allocator.clone(),
queue.clone(),
);
let csg_mats = gpu_buffer(
mats,
allocator.clone(),
&staging,
command_allocator.clone(),
queue.clone(),
);
let csg_deps = gpu_buffer(
deps,
allocator.clone(),
&staging,
command_allocator.clone(),
queue,
);
(
Subbuffers {
masks: fragment_masks_buffer,
floats: csg_floats,
vec2s: csg_vec2s,
//vec3s: csg_vec3s,
vec4s: csg_vec4s,
mat2s: csg_mat2s,
mat3s: csg_mat3s,
mat4s: csg_mat4s,
mats: csg_mats,
scene: csg_scene,
deps: csg_deps,
desc: csg_desc,
},
scene,
@@ -1979,13 +1884,10 @@ fn get_pipeline_cache(device: Arc<Device>) -> Arc<PipelineCache> {
};
unsafe {
PipelineCache::new(
device,
PipelineCacheCreateInfo {
PipelineCache::new(device, PipelineCacheCreateInfo {
initial_data,
..Default::default()
},
)
})
}
.unwrap()
}
-13
View File
@@ -47,19 +47,6 @@ pub(crate) struct CSG {
pub(crate) type Float = f32;
#[derive(Clone, Copy, Debug, Default, PartialEq)]
pub(crate) enum Inputs {
#[default]
Variable,
Float(Float),
Vec2(Vec2),
Vec3(Vec3),
Vec4(Vec4),
Mat2(Mat2),
Mat3(Mat3),
Mat4(Mat4),
}
pub(crate) fn load_obj(
memory_allocator: &Arc<StandardMemoryAllocator>,
input: &mut dyn Read,
+317 -82
View File
@@ -3,120 +3,235 @@
#include "spec_constants.glsl"
layout(set = 1, binding = 0, std430) uniform SceneDescription {
struct Description {
uint scene;
uint constants;
uint io;
float[6] bounds;
} desc;
layout(set = 1, binding = 1, std430) uniform SceneBuf {
u32vec4 opcodes[MASK_ARRAY_LENGTH];
}
layout(set = 1, binding = 2, std430) uniform FloatConst {
u32vec4 floats[MASK_ARRAY_LENGTH];
}
layout(set = 1, binding = 3, std430) uniform Inputs {
u32vec4 inputs[MASK_ARRAY_LENGTH];
}
layout(set = 1, binding = 0, std430) restrict readonly buffer SceneDescription {
Description desc[];
} scene_description;
const uint8_t MASK_EXECUTE = 0;
const uint8_t MASK_PASS_P1 = 1;
const uint8_t MASK_PASS_P2 = 2;
const uint8_t MASK_PASS_P3 = 3;
const uint8_t MASK_PASS_P4 = 4;
layout(set = 1, binding = 1, std430) restrict readonly buffer SceneBuf {
u32vec4 data[];
} scenes;
uint8_t mask[MASK_ARRAY_LENGTH];
#ifdef interval_frags
#define fragmentmasks_layout readonly
#else
#define fragmentmasks_layout // writeonly
#endif
layout(set = 1, binding = 2, std430) restrict fragmentmasks_layout buffer fragmentMasks {
uint16_t masks[][MASK_ARRAY_LENGTH];
} fragmentpassmasks;
// each mask:
// CCPP CCPP CCPP CCPP
// CC = how many constants to skip
// PP = which input to return: execute, do nothing, copy left, copy right
const uint16_t MASK_EXECUTE = uint16_t(0);
const uint16_t MASK_COPY_LEFT = uint16_t(1);
const uint16_t MASK_COPY_RIGHT = uint16_t(2);
const uint16_t MASK_DO_NOTHING = uint16_t(3);
uint16_t mask[MASK_ARRAY_LENGTH];
void default_mask()
{
for (int i = 0; i < MASK_ARRAY_LENGTH; i++) {
mask[i] = MASK_EXECUTE;
mask[i] = uint16_t(0);
}
}
/// increments once for each opcode
uint program_counter = 0;
/// increments once for each nibble
uint nibble_counter = 0;
uint io_counter = 0;
/// increments for each constant
uint const_counter = 0;
u32vec4 major_integer_unpack;
u32vec4 major_float_unpack;
u32vec4 major_opcode_unpack;
u32vec4 major_io_unpack;
u32vec4 major_const_unpack;
float load_const() {
if ((const_counter % 4) == 0) {
major_float_unpack = floats.floats[desc.floats + (const_counter / 4)];
float load_const(bool reverse) {
if ((reverse && ((const_counter % 4) == 3)) || (!reverse && ((const_counter % 4) == 0))) {
major_const_unpack = scenes.data[desc.constants + (const_counter / 4)];
}
switch (const_counter % 4) {
case 0:
return uintBitsToFloat(major_integer_unpack.x);
return uintBitsToFloat(major_const_unpack.x);
case 1:
return uintBitsToFloat(major_integer_unpack.y);
return uintBitsToFloat(major_const_unpack.y);
case 2:
return uintBitsToFloat(major_integer_unpack.z);
return uintBitsToFloat(major_const_unpack.z);
case 3:
return uintBitsToFloat(major_integer_unpack.w);
return uintBitsToFloat(major_const_unpack.w);
}
}
uint8_t load_byte() {
if ((nibble_counter % 32) == 0) {
major_integer_unpack = scenes.opcodes[desc.scene + (nibble_counter / 32)];
uint8_t load_opcode(bool reverse) {
if ((reverse && ((program_counter % 16) == 15)) || (!reverse && ((program_counter % 16) == 0))) {
major_opcode_unpack = scenes.data[desc.scene + (program_counter / 16)];
}
switch ((nibble_counter / 2) % 16) {
switch (program_counter % 16) {
case 0:
return uint8_t((major_integer_unpack.x >> 0) & 255);
return uint8_t((major_opcode_unpack.x >> 0) & 255);
case 1:
return uint8_t((major_integer_unpack.x >> 8) & 255);
return uint8_t((major_opcode_unpack.x >> 8) & 255);
case 2:
return uint8_t((major_integer_unpack.x >> 16) & 255);
return uint8_t((major_opcode_unpack.x >> 16) & 255);
case 3:
return uint8_t((major_integer_unpack.x >> 24) & 255);
return uint8_t((major_opcode_unpack.x >> 24) & 255);
case 4:
return uint8_t((major_integer_unpack.y >> 0) & 255);
return uint8_t((major_opcode_unpack.y >> 0) & 255);
case 5:
return uint8_t((major_integer_unpack.y >> 8) & 255);
return uint8_t((major_opcode_unpack.y >> 8) & 255);
case 6:
return uint8_t((major_integer_unpack.y >> 16) & 255);
return uint8_t((major_opcode_unpack.y >> 16) & 255);
case 7:
return uint8_t((major_integer_unpack.y >> 24) & 255);
return uint8_t((major_opcode_unpack.y >> 24) & 255);
case 8:
return uint8_t((major_integer_unpack.z >> 0) & 255);
return uint8_t((major_opcode_unpack.z >> 0) & 255);
case 9:
return uint8_t((major_integer_unpack.z >> 8) & 255);
return uint8_t((major_opcode_unpack.z >> 8) & 255);
case 10:
return uint8_t((major_integer_unpack.z >> 16) & 255);
return uint8_t((major_opcode_unpack.z >> 16) & 255);
case 11:
return uint8_t((major_integer_unpack.z >> 24) & 255);
return uint8_t((major_opcode_unpack.z >> 24) & 255);
case 12:
return uint8_t((major_integer_unpack.w >> 0) & 255);
return uint8_t((major_opcode_unpack.w >> 0) & 255);
case 13:
return uint8_t((major_integer_unpack.w >> 8) & 255);
return uint8_t((major_opcode_unpack.w >> 8) & 255);
case 14:
return uint8_t((major_integer_unpack.w >> 16) & 255);
return uint8_t((major_opcode_unpack.w >> 16) & 255);
case 15:
return uint8_t((major_integer_unpack.w >> 24) & 255);
return uint8_t((major_opcode_unpack.w >> 24) & 255);
}
}
uint8_t load_opcode() {
nibble_counter += 1;
nibble_counter &= (~1);
uint8_t load_input(bool reverse) {
if ((reverse && ((io_counter % 32) == 31)) || (!reverse && ((io_counter % 32) == 0))) {
major_io_unpack = scenes.data[desc.io + (io_counter / 32)];
}
switch (io_counter % 32) {
case 0:
return uint8_t((major_io_unpack.x >> 0) & 15);
case 1:
return uint8_t((major_io_unpack.x >> 4) & 15);
case 2:
return uint8_t((major_io_unpack.x >> 8) & 15);
case 3:
return uint8_t((major_io_unpack.x >> 12) & 15);
case 4:
return uint8_t((major_io_unpack.x >> 16) & 15);
case 5:
return uint8_t((major_io_unpack.x >> 20) & 15);
case 6:
return uint8_t((major_io_unpack.x >> 24) & 15);
case 7:
return uint8_t((major_io_unpack.x >> 28) & 15);
case 8:
return uint8_t((major_io_unpack.y >> 0) & 15);
case 9:
return uint8_t((major_io_unpack.y >> 4) & 15);
case 10:
return uint8_t((major_io_unpack.y >> 8) & 15);
case 11:
return uint8_t((major_io_unpack.y >> 12) & 15);
case 12:
return uint8_t((major_io_unpack.y >> 16) & 15);
case 13:
return uint8_t((major_io_unpack.y >> 20) & 15);
case 14:
return uint8_t((major_io_unpack.y >> 24) & 15);
case 15:
return uint8_t((major_io_unpack.y >> 28) & 15);
case 16:
return uint8_t((major_io_unpack.z >> 0) & 15);
case 17:
return uint8_t((major_io_unpack.z >> 4) & 15);
case 18:
return uint8_t((major_io_unpack.z >> 8) & 15);
case 19:
return uint8_t((major_io_unpack.z >> 12) & 15);
case 20:
return uint8_t((major_io_unpack.z >> 16) & 15);
case 21:
return uint8_t((major_io_unpack.z >> 20) & 15);
case 22:
return uint8_t((major_io_unpack.z >> 24) & 15);
case 23:
return uint8_t((major_io_unpack.z >> 28) & 15);
case 24:
return uint8_t((major_io_unpack.w >> 0) & 15);
case 25:
return uint8_t((major_io_unpack.w >> 4) & 15);
case 26:
return uint8_t((major_io_unpack.w >> 8) & 15);
case 27:
return uint8_t((major_io_unpack.w >> 12) & 15);
case 28:
return uint8_t((major_io_unpack.w >> 16) & 15);
case 29:
return uint8_t((major_io_unpack.w >> 20) & 15);
case 30:
return uint8_t((major_io_unpack.w >> 24) & 15);
case 31:
return uint8_t((major_io_unpack.w >> 28) & 15);
}
}
uint8_t prev_opcode() {
program_counter -= 1;
return load_opcode(false);
}
uint8_t next_opcode() {
uint8_t t = load_opcode(true);
program_counter += 1;
return load_byte();
return t;
}
uint8_t load_register() {
if ((nibble_counter % 2) == 0) {
return uint8_t(load_byte() & 15);
}
else if ((nibble_counter % 2) == 1) {
return uint8_t(load_byte() >> 4);
uint8_t prev_register() {
io_counter -= 1;
return load_input(false);
}
uint8_t next_register() {
uint8_t t = load_input(true);
io_counter += 1;
return t;
}
void jump_registers(uint dist) {
bool reload_cache = (io_counter / 32) != ((io_counter + dist) / 32);
io_counter += dist;
if (reload_cache) {
major_io_unpack = scenes.data[desc.io + (io_counter / 32)];
}
}
uint8_t load_mask() {
return mask[program_counter];
float prev_const() {
const_counter -= 1;
return load_const(false);
}
float next_const() {
float t = load_const(true);
const_counter += 1;
return t;
}
void jump_const(uint dist) {
bool reload_cache = (const_counter / 4) != ((const_counter + dist) / 4);
const_counter += dist;
if (reload_cache) {
major_const_unpack = scenes.data[desc.constants + (const_counter / 4)];
}
}
#define unroll_instruction_set(index, name) \
@@ -125,56 +240,176 @@ const uint8_t OPAdd##name = uint8_t(OPAdd+(index<<6));\
const uint8_t OPSub##name = uint8_t(OPSub+(index<<6));\
const uint8_t OPMul##name = uint8_t(OPMul+(index<<6));\
const uint8_t OPDiv##name = uint8_t(OPDiv+(index<<6));\
const uint8_t OPMod##name = uint8_t(OPMod+(index<<6));\
const uint8_t OPRem##name = uint8_t(OPRem+(index<<6));\
const uint8_t OPPow##name = uint8_t(OPPow+(index<<6));\
const uint8_t OPAtan2##name = uint8_t(OPAtan2+(index<<6));\
const uint8_t OPMin##name = uint8_t(OPMin+(index<<6));\
const uint8_t OPMax##name = uint8_t(OPMax+(index<<6));\
const uint8_t OPCompare##name = uint8_t(OPCompare+(index<<6));\
const uint8_t OPMod##name = uint8_t(OPMod+(index<<6));\
const uint8_t OPAnd##name = uint8_t(OPAnd+(index<<6));\
const uint8_t OPOr##name = uint8_t(OPOr+(index<<6));\
const uint8_t OPNegate##name = uint8_t(OPNegate+(index<<6));\
const uint8_t OPRound##name = uint8_t(OPRound+(index<<6));\
const uint8_t OPRoundEven##name = uint8_t(OPRoundEven+(index<<6));\
const uint8_t OPTrunc##name = uint8_t(OPTrunc+(index<<6));\
const uint8_t OPAbs##name = uint8_t(OPAbs+(index<<6));\
const uint8_t OPSign##name = uint8_t(OPSign+(index<<6));\
const uint8_t OPRecip##name = uint8_t(OPRecip+(index<<6));\
const uint8_t OPSqrt##name = uint8_t(OPSqrt+(index<<6));\
const uint8_t OPSquare##name = uint8_t(OPSquare+(index<<6));\
const uint8_t OPFloor##name = uint8_t(OPFloor+(index<<6));\
const uint8_t OPCeil##name = uint8_t(OPCeil+(index<<6));\
const uint8_t OPFract##name = uint8_t(OPFract+(index<<6));\
const uint8_t OPRound##name = uint8_t(OPRound+(index<<6));\
const uint8_t OPSin##name = uint8_t(OPSin+(index<<6));\
const uint8_t OPCos##name = uint8_t(OPCos+(index<<6));\
const uint8_t OPTan##name = uint8_t(OPTan+(index<<6));\
const uint8_t OPAsin##name = uint8_t(OPAsin+(index<<6));\
const uint8_t OPAcos##name = uint8_t(OPAcos+(index<<6));\
const uint8_t OPAtan##name = uint8_t(OPAtan+(index<<6));\
const uint8_t OPSinh##name = uint8_t(OPSinh+(index<<6));\
const uint8_t OPCosh##name = uint8_t(OPCosh+(index<<6));\
const uint8_t OPTanh##name = uint8_t(OPTanh+(index<<6));\
const uint8_t OPAsinh##name = uint8_t(OPAsinh+(index<<6));\
const uint8_t OPAcosh##name = uint8_t(OPAcosh+(index<<6));\
const uint8_t OPAtanh##name = uint8_t(OPAtanh+(index<<6));\
const uint8_t OPExp##name = uint8_t(OPExp+(index<<6));\
const uint8_t OPLog##name = uint8_t(OPLog+(index<<6));\
const uint8_t OPExp2##name = uint8_t(OPExp2+(index<<6));\
const uint8_t OPLog2##name = uint8_t(OPLog2+(index<<6));\
const uint8_t OPSqrt##name = uint8_t(OPSqrt+(index<<6));\
const uint8_t OPInverseSqrt##name = uint8_t(OPInverseSqrt+(index<<6));\
const uint8_t OPSquare##name = uint8_t(OPSquare+(index<<6));\
const uint8_t OPNot##name = uint8_t(OPNot+(index<<6));\
const uint8_t OPFract##name = uint8_t(OPFract+(index<<6));\
const uint8_t OPCube##name = uint8_t(OPCube+(index<<6));\
const uint8_t OPSmoothMin##name = uint8_t(OPSmoothMin+(index<<6));\
const uint8_t OPSmoothMax##name = uint8_t(OPSmoothMax+(index<<6));\
const uint8_t OPClamp##name = uint8_t(OPClamp+(index<<6));\
const uint8_t OPMix##name = uint8_t(OPMix+(index<<6));\
const uint8_t OPStep##name = uint8_t(OPStep+(index<<6));\
const uint8_t OPSmoothStep##name = uint8_t(OPSmoothStep+(index<<6));\
const uint8_t OPFMA##name = uint8_t(OPFMA+(index<<6));\
const uint8_t OPDot##name = uint8_t(OPDot+(index<<6));\
const uint8_t OPLength##name = uint8_t(OPLength+(index<<6));\
const uint8_t OPNormalize##name = uint8_t(OPNormalize+(index<<6)); \
const uint8_t OPDot##name = uint8_t(OPDot+(index<<6));\
const uint8_t OPDistance##name = uint8_t(OPDistance+(index<<6));\
const uint8_t OPNormalize##name = uint8_t(OPNormalize+(index<<6));
unroll_instruction_set(0, Vec1)
unroll_instruction_set(1, Vec2)
unroll_instruction_set(2, Vec3)
unroll_instruction_set(3, Vec4)
uint reg_output(uint8_t opcode) {
switch (uint(opcode)) {
case OPNop:
return 0;
case OPReturn:
return 0;
case OPPosition:
return 3;
case OPMinMaterial:
return 1;
case OPMaxMaterial:
return 1;
case OPSmoothMinMaterial:
return 1;
case OPSmoothMaxMaterial:
return 1;
case OPCross:
return 3;
case OPDistance:
return 1;
case OPLength:
return 1;
case OPDot:
return 1;
case OPSDFSphere:
return 1;
case OPSDFBox:
return 1;
case OPSDFTorus:
return 1;
default:
return (opcode >> 6) + 1;
}
}
uint reg_input(uint8_t opcode) {
switch (uint(opcode)) {
case OPNop:
return 0;
case OPReturn:
return 1;
case OPPosition:
return 0;
case OPMinMaterial:
return 2;
case OPMaxMaterial:
return 2;
case OPSmoothMinMaterial:
return 3;
case OPSmoothMaxMaterial:
return 3;
case OPCross:
return 3 + 3;
case OPSDFSphere:
return 3 + 1;
case OPSDFBox:
return 3 + 3;
case OPSDFTorus:
return 3 + 2;
case OPAdd:
case OPSub:
case OPMul:
case OPDiv:
case OPAtan2:
case OPMin:
case OPMax:
case OPCompare:
case OPMod:
case OPAnd:
case OPOr:
case OPDot:
case OPDistance:
return ((opcode >> 6) + 1) * 2;
case OPSmoothMin:
case OPSmoothMax:
case OPClamp:
case OPMix:
case OPFMA:
return ((opcode >> 6) + 1) * 3;
default:
return ((opcode >> 6) + 1);
}
}
uint[2] reg_elementwise(uint8_t opcode) {
switch (uint(opcode)) {
case OPNop:
case OPReturn:
case OPPosition:
case OPMinMaterial:
case OPMaxMaterial:
case OPSmoothMinMaterial:
case OPSmoothMaxMaterial:
case OPCross:
case OPSDFSphere:
case OPSDFBox:
case OPSDFTorus:
case OPDot:
case OPDistance:
case OPLength:
case OPNormalize:
return uint[2](0, 0);
case OPAdd:
case OPSub:
case OPMul:
case OPDiv:
case OPAtan2:
case OPMin:
case OPCompare:
case OPMod:
case OPAnd:
case OPOr:
return uint[2](2, 1);
case OPSmoothMin:
case OPSmoothMax:
case OPClamp:
case OPMix:
case OPFMA:
return uint[2](3, 1);
default:
return uint[2](1, 1);
}
}
#endif
+271 -353
View File
@@ -9,65 +9,58 @@ const JIT_VERSION: u32 = 1;
#[derive(Debug, Default, PartialEq, Eq, Clone, Copy)]
pub(crate) enum SSAOpcode {
#[default]
SSAStop,
SSAReturn,
SSAPosition,
SSAAdd,
SSASub,
SSAMul,
SSADiv,
SSAMod,
SSARem,
SSAPow,
SSAAtan2,
SSAMin,
SSAMinMaterial,
SSAMax,
SSAMaxMaterial,
SSACross,
SSADot,
SSALength,
SSADistance,
SSANormalize,
SSACompare,
SSAMod,
SSAAnd,
SSAOr,
SSANegate,
SSARound,
SSARoundEven,
SSATrunc,
SSAAbs,
SSASign,
SSARecip,
SSASqrt,
SSASquare,
SSAFloor,
SSACeil,
SSAFract,
SSARound,
SSASin,
SSACos,
SSATan,
SSAAsin,
SSAAcos,
SSAAtan,
SSASinh,
SSACosh,
SSATanh,
SSAAsinh,
SSAAcosh,
SSAAtanh,
SSAExp,
SSALog,
SSAExp2,
SSALog2,
SSASqrt,
SSAInverseSqrt,
SSASquare,
SSANot,
SSAFract,
SSACube,
SSASmoothMin,
SSASmoothMax,
SSASmoothMinMaterial,
SSASmoothMaxMaterial,
SSAClamp,
SSAMix,
SSAStep,
SSASmoothStep,
SSAFMA,
SSADot,
SSALength,
SSADistance,
SSANormalize,
#[default]
SSAStop,
SSAReturn,
SSAPosition,
SSAMinMaterial,
SSAMaxMaterial,
SSASmoothMinMaterial,
SSASmoothMaxMaterial,
SSACross,
SSASDFSphere,
SSASDFBox,
SSASDFTorus,
@@ -102,9 +95,10 @@ pub(crate) struct SSATape {
#[derive(Debug, Default, PartialEq, Eq, Clone, Copy)]
struct GPUOpcode(u8);
struct GPUTape {
instructions: Vec<u8>,
constants: Vec<f32>,
pub(crate) struct GPUTape {
pub instructions: Vec<u8>,
pub io: Vec<u8>,
pub constants: Vec<f32>,
}
impl SSAOpcodeSized {
@@ -143,11 +137,9 @@ impl SSAOpcodeSized {
SSASDFSphere => 3 + 1,
SSASDFBox => 3 + 3,
SSASDFTorus => 3 + 2,
SSAAdd | SSASub | SSAMul | SSADiv | SSAMod | SSARem | SSAPow | SSAAtan2 | SSAMin
| SSAMax | SSADot | SSADistance | SSAStep => self.size * 2,
SSASmoothMin | SSASmoothMax | SSAClamp | SSAMix | SSASmoothStep | SSAFMA => {
self.size * 3
},
SSAAdd | SSASub | SSAMul | SSADiv | SSAAtan2 | SSAMin | SSAMax | SSACompare
| SSAMod | SSAAnd | SSAOr | SSADot | SSADistance => self.size * 2,
SSASmoothMin | SSASmoothMax | SSAClamp | SSAMix | SSAFMA => self.size * 3,
_ => self.size,
}
}
@@ -157,10 +149,10 @@ impl SSAOpcodeSized {
match self.opcode {
SSAStop | SSAReturn | SSAPosition | SSAMinMaterial | SSAMaxMaterial
| SSASmoothMinMaterial | SSASmoothMaxMaterial | SSACross | SSASDFSphere | SSASDFBox
| SSASDFTorus => (0, 0),
SSAAdd | SSASub | SSAMul | SSADiv | SSAMod | SSARem | SSAPow | SSAAtan2 | SSAMin
| SSAMax | SSADot | SSADistance | SSAStep => (2, 1),
SSASmoothMin | SSASmoothMax | SSAClamp | SSAMix | SSASmoothStep | SSAFMA => (3, 1),
| SSASDFTorus | SSADot | SSADistance | SSALength | SSANormalize => (0, 0),
SSAAdd | SSASub | SSAMul | SSADiv | SSAAtan2 | SSAMin | SSACompare | SSAMod
| SSAAnd | SSAOr => (2, 1),
SSASmoothMin | SSASmoothMax | SSAClamp | SSAMix | SSAFMA => (3, 1),
_ => (1, 1),
}
}
@@ -174,7 +166,7 @@ impl SSAOpcodeSized {
GPUOpcode(inst as u8 + ((width - 1) << 6))
}
match self.opcode {
SSAStop => opcode_drop(OPStop, 1),
SSAStop => opcode_drop(OPReturn, 1),
SSAReturn => opcode_drop(OPReturn, 1),
SSAPosition => opcode_drop(OPPosition, 1),
SSAMinMaterial => opcode_drop(OPMinMaterial, 1),
@@ -190,8 +182,6 @@ impl SSAOpcodeSized {
SSAMul => opcode_drop(OPMul, self.size),
SSADiv => opcode_drop(OPDiv, self.size),
SSAMod => opcode_drop(OPMod, self.size),
SSARem => opcode_drop(OPRem, self.size),
SSAPow => opcode_drop(OPPow, self.size),
SSAAtan2 => opcode_drop(OPAtan2, self.size),
SSAMin => opcode_drop(OPMin, self.size),
SSAMax => opcode_drop(OPMax, self.size),
@@ -201,10 +191,7 @@ impl SSAOpcodeSized {
SSANormalize => opcode_drop(OPNormalize, self.size),
SSANegate => opcode_drop(OPNegate, self.size),
SSARound => opcode_drop(OPRound, self.size),
SSARoundEven => opcode_drop(OPRoundEven, self.size),
SSATrunc => opcode_drop(OPTrunc, self.size),
SSAAbs => opcode_drop(OPAbs, self.size),
SSASign => opcode_drop(OPSign, self.size),
SSAFloor => opcode_drop(OPFloor, self.size),
SSACeil => opcode_drop(OPCeil, self.size),
SSAFract => opcode_drop(OPFract, self.size),
@@ -214,33 +201,31 @@ impl SSAOpcodeSized {
SSAAsin => opcode_drop(OPAsin, self.size),
SSAAcos => opcode_drop(OPAcos, self.size),
SSAAtan => opcode_drop(OPAtan, self.size),
SSASinh => opcode_drop(OPSinh, self.size),
SSACosh => opcode_drop(OPCosh, self.size),
SSATanh => opcode_drop(OPTanh, self.size),
SSAAsinh => opcode_drop(OPAsinh, self.size),
SSAAcosh => opcode_drop(OPAcosh, self.size),
SSAAtanh => opcode_drop(OPAtanh, self.size),
SSAExp => opcode_drop(OPExp, self.size),
SSALog => opcode_drop(OPLog, self.size),
SSAExp2 => opcode_drop(OPExp2, self.size),
SSALog2 => opcode_drop(OPLog2, self.size),
SSASqrt => opcode_drop(OPSqrt, self.size),
SSAInverseSqrt => opcode_drop(OPInverseSqrt, self.size),
SSASquare => opcode_drop(OPSquare, self.size),
SSACube => opcode_drop(OPCube, self.size),
SSASmoothMin => opcode_drop(OPSmoothMin, self.size),
SSASmoothMax => opcode_drop(OPSmoothMax, self.size),
SSAClamp => opcode_drop(OPClamp, self.size),
SSAMix => opcode_drop(OPMix, self.size),
SSAStep => opcode_drop(OPStep, self.size),
SSASmoothStep => opcode_drop(OPSmoothStep, self.size),
SSAFMA => opcode_drop(OPFMA, self.size),
SSACompare => opcode_drop(OPCompare, self.size),
SSAAnd => opcode_drop(OPAnd, self.size),
SSAOr => opcode_drop(OPOr, self.size),
SSARecip => opcode_drop(OPRecip, self.size),
SSANot => opcode_drop(OPNot, self.size),
}
}
}
impl SSATape {
pub fn push_instruction(&mut self, opcode: SSAOpcodeSized, inputs: Vec<SSAInput>) -> Vec<SSAInput> {
pub fn push_instruction(
&mut self,
opcode: SSAOpcodeSized,
inputs: Vec<SSAInput>,
) -> Vec<SSAInput> {
assert!(
inputs
.iter()
@@ -341,7 +326,9 @@ impl SSATape {
for ((life_start, life_end), allocation) in
lifetimes.iter().zip(register_allocation.iter_mut())
{
if let Some(register) = register_hold.iter().position(|reg| reg <= life_start) {
if life_start == life_end {
*allocation = 0;
} else if let Some(register) = register_hold.iter().position(|reg| reg <= life_start) {
register_hold[register] = *life_end;
*allocation = (register + 1) as u8;
} else {
@@ -351,9 +338,13 @@ impl SSATape {
let mut gpu_tape = GPUTape {
instructions: vec![],
io: vec![],
constants: self.constants.clone(),
};
let mut low_nibble = true;
let mut staging_byte = 0u8;
for SSAInstruction {
opcode,
inputs,
@@ -363,9 +354,9 @@ impl SSATape {
let code = opcode.to_raw_opcode();
gpu_tape.instructions.push(code.0);
let mut low_nibble = true;
let mut staging_byte = 0u8;
let per_element = opcode.lifetime_elementwise();
if per_element == (0, 0) {
for input in inputs {
let register = match input {
SSAInput::Constant(0.0) => 0,
@@ -377,7 +368,7 @@ impl SSATape {
staging_byte |= register;
} else {
staging_byte |= register << 4;
gpu_tape.instructions.push(staging_byte);
gpu_tape.io.push(staging_byte);
}
low_nibble = !low_nibble;
@@ -390,16 +381,77 @@ impl SSATape {
staging_byte |= register;
} else {
staging_byte |= register << 4;
gpu_tape.instructions.push(staging_byte);
gpu_tape.io.push(staging_byte);
}
low_nibble = !low_nibble;
}
// Stop is implemented as SSAReturn(0);
if opcode.opcode == SSAOpcode::SSAStop {
let register = 0;
if low_nibble {
staging_byte |= register;
} else {
staging_byte |= register << 4;
gpu_tape.io.push(staging_byte);
}
low_nibble = !low_nibble;
}
} else {
let mut input_iterators = (0..per_element.0)
.map(|i| inputs.iter().skip(i.into()).step_by(per_element.0.into()))
.collect::<Vec<_>>();
let mut output_iterators = (0..per_element.1)
.map(|i| outputs.iter().skip(i.into()).step_by(per_element.1.into()))
.collect::<Vec<_>>();
assert_eq!(
opcode.input() / per_element.0,
opcode.output() / per_element.1
);
for _ in 0..(opcode.input() / per_element.0) {
for iterator in input_iterators.iter_mut() {
let &value = iterator.next().unwrap();
let register = match value {
SSAInput::Constant(0.0) => 0,
SSAInput::Constant(_) => 15,
SSAInput::Register(u) => register_allocation[u as usize],
};
if low_nibble {
staging_byte |= register;
} else {
staging_byte |= register << 4;
gpu_tape.io.push(staging_byte);
}
low_nibble = !low_nibble;
}
for iterator in output_iterators.iter_mut() {
let &value = iterator.next().unwrap();
let register = register_allocation[value as usize];
if low_nibble {
staging_byte |= register;
} else {
staging_byte |= register << 4;
gpu_tape.io.push(staging_byte);
}
low_nibble = !low_nibble;
}
}
}
}
if !low_nibble {
gpu_tape.instructions.push(staging_byte);
}
gpu_tape.io.push(staging_byte);
}
gpu_tape
}
@@ -412,6 +464,7 @@ impl SSATape {
let void = b.type_void();
let float = b.type_float(32);
let bool = b.type_bool();
let vec1 = b.type_vector(float, 1);
let vec2 = b.type_vector(float, 2);
let vec3 = b.type_vector(float, 3);
@@ -434,8 +487,8 @@ impl SSATape {
let mut mapping = HashMap::<u32, u32>::new();
for (line, instruction) in self.tape.iter().enumerate() {
use rspirv::dr::Operand::IdRef;
use SSAOpcode::*;
use rspirv::dr::Operand::IdRef;
b.line(jit_string, line as u32, 0);
@@ -541,7 +594,8 @@ impl SSATape {
match instruction.opcode.opcode {
SSAStop => {
b.ret().unwrap();
let zero = b.constant_bit32(float, (0.0f32).to_bits());
b.ret_value(zero).unwrap();
},
SSAReturn => {
let value = input_resolve(float, &mut b, &mapping, instruction.inputs[0]);
@@ -606,33 +660,6 @@ impl SSATape {
|b, val_a, val_b| b.f_mod(float, None, val_a, val_b).unwrap(),
);
},
SSARem => {
param_two(
float,
&mut b,
&mut mapping,
instruction,
|b, val_a, val_b| b.f_rem(float, None, val_a, val_b).unwrap(),
);
},
SSAPow => {
param_two(
float,
&mut b,
&mut mapping,
instruction,
|b, val_a, val_b| {
b.ext_inst(
float,
None,
glsl,
spirv::GLOp::Pow as u32,
[IdRef(val_a), IdRef(val_b)],
)
.unwrap()
},
);
},
SSAAtan2 => {
param_two(
float,
@@ -640,13 +667,10 @@ impl SSATape {
&mut mapping,
instruction,
|b, val_a, val_b| {
b.ext_inst(
float,
None,
glsl,
spirv::GLOp::Atan2 as u32,
[IdRef(val_a), IdRef(val_b)],
)
b.ext_inst(float, None, glsl, spirv::GLOp::Atan2 as u32, [
IdRef(val_a),
IdRef(val_b),
])
.unwrap()
},
);
@@ -658,13 +682,10 @@ impl SSATape {
&mut mapping,
instruction,
|b, val_a, val_b| {
b.ext_inst(
float,
None,
glsl,
spirv::GLOp::FMin as u32,
[IdRef(val_a), IdRef(val_b)],
)
b.ext_inst(float, None, glsl, spirv::GLOp::FMin as u32, [
IdRef(val_a),
IdRef(val_b),
])
.unwrap()
},
);
@@ -677,13 +698,10 @@ impl SSATape {
&mut mapping,
instruction,
|b, val_a, val_b| {
b.ext_inst(
float,
None,
glsl,
spirv::GLOp::FMax as u32,
[IdRef(val_a), IdRef(val_b)],
)
b.ext_inst(float, None, glsl, spirv::GLOp::FMax as u32, [
IdRef(val_a),
IdRef(val_b),
])
.unwrap()
},
);
@@ -699,13 +717,10 @@ impl SSATape {
let val_a = b.composite_construct(vec3, None, [a_x, a_y, a_z]).unwrap();
let val_b = b.composite_construct(vec3, None, [b_x, b_y, b_z]).unwrap();
let cross = b
.ext_inst(
vec3,
None,
glsl,
spirv::GLOp::Cross as u32,
[IdRef(val_a), IdRef(val_b)],
)
.ext_inst(vec3, None, glsl, spirv::GLOp::Cross as u32, [
IdRef(val_a),
IdRef(val_b),
])
.unwrap();
mapping.insert(
instruction.outputs[0],
@@ -746,13 +761,9 @@ impl SSATape {
let vector = [void, vec1, vec2, vec3, vec4][instruction.opcode.size as usize];
let val_a = b.composite_construct(vector, None, val_a).unwrap();
let length = b
.ext_inst(
float,
None,
glsl,
spirv::GLOp::Length as u32,
[IdRef(val_a)],
)
.ext_inst(float, None, glsl, spirv::GLOp::Length as u32, [IdRef(
val_a,
)])
.unwrap();
mapping.insert(instruction.outputs[0], length);
},
@@ -771,13 +782,10 @@ impl SSATape {
let val_a = b.composite_construct(vector, None, val_a).unwrap();
let val_b = b.composite_construct(vector, None, val_b).unwrap();
let distance = b
.ext_inst(
float,
None,
glsl,
spirv::GLOp::Distance as u32,
[IdRef(val_a), IdRef(val_b)],
)
.ext_inst(float, None, glsl, spirv::GLOp::Distance as u32, [
IdRef(val_a),
IdRef(val_b),
])
.unwrap();
mapping.insert(instruction.outputs[0], distance);
},
@@ -796,13 +804,10 @@ impl SSATape {
let val_a = b.composite_construct(vector, None, val_a).unwrap();
let val_b = b.composite_construct(vector, None, val_b).unwrap();
let normal = b
.ext_inst(
vector,
None,
glsl,
spirv::GLOp::Normalize as u32,
[IdRef(val_a), IdRef(val_b)],
)
.ext_inst(vector, None, glsl, spirv::GLOp::Normalize as u32, [
IdRef(val_a),
IdRef(val_b),
])
.unwrap();
for i in 0..instruction.opcode.size as usize {
mapping.insert(
@@ -812,6 +817,12 @@ impl SSATape {
);
}
},
SSARecip => {
param_one(float, &mut b, &mut mapping, instruction, |b, val_a| {
let one = b.constant_bit32(float, (1.0f32).to_bits());
b.f_div(float, None, one, val_a).unwrap()
});
},
SSANegate => {
param_one(float, &mut b, &mut mapping, instruction, |b, val_a| {
b.f_negate(float, None, val_a).unwrap()
@@ -823,36 +834,12 @@ impl SSATape {
.unwrap()
});
},
SSARoundEven => {
param_one(float, &mut b, &mut mapping, instruction, |b, val_a| {
b.ext_inst(
float,
None,
glsl,
spirv::GLOp::RoundEven as u32,
[IdRef(val_a)],
)
.unwrap()
});
},
SSATrunc => {
param_one(float, &mut b, &mut mapping, instruction, |b, val_a| {
b.ext_inst(float, None, glsl, spirv::GLOp::Trunc as u32, [IdRef(val_a)])
.unwrap()
});
},
SSAAbs => {
param_one(float, &mut b, &mut mapping, instruction, |b, val_a| {
b.ext_inst(float, None, glsl, spirv::GLOp::FAbs as u32, [IdRef(val_a)])
.unwrap()
});
},
SSASign => {
param_one(float, &mut b, &mut mapping, instruction, |b, val_a| {
b.ext_inst(float, None, glsl, spirv::GLOp::FSign as u32, [IdRef(val_a)])
.unwrap()
});
},
SSAFloor => {
param_one(float, &mut b, &mut mapping, instruction, |b, val_a| {
b.ext_inst(float, None, glsl, spirv::GLOp::Floor as u32, [IdRef(val_a)])
@@ -907,42 +894,6 @@ impl SSATape {
.unwrap()
});
},
SSASinh => {
param_one(float, &mut b, &mut mapping, instruction, |b, val_a| {
b.ext_inst(float, None, glsl, spirv::GLOp::Sinh as u32, [IdRef(val_a)])
.unwrap()
});
},
SSACosh => {
param_one(float, &mut b, &mut mapping, instruction, |b, val_a| {
b.ext_inst(float, None, glsl, spirv::GLOp::Cosh as u32, [IdRef(val_a)])
.unwrap()
});
},
SSATanh => {
param_one(float, &mut b, &mut mapping, instruction, |b, val_a| {
b.ext_inst(float, None, glsl, spirv::GLOp::Tanh as u32, [IdRef(val_a)])
.unwrap()
});
},
SSAAsinh => {
param_one(float, &mut b, &mut mapping, instruction, |b, val_a| {
b.ext_inst(float, None, glsl, spirv::GLOp::Asinh as u32, [IdRef(val_a)])
.unwrap()
});
},
SSAAcosh => {
param_one(float, &mut b, &mut mapping, instruction, |b, val_a| {
b.ext_inst(float, None, glsl, spirv::GLOp::Acosh as u32, [IdRef(val_a)])
.unwrap()
});
},
SSAAtanh => {
param_one(float, &mut b, &mut mapping, instruction, |b, val_a| {
b.ext_inst(float, None, glsl, spirv::GLOp::Atanh as u32, [IdRef(val_a)])
.unwrap()
});
},
SSAExp => {
param_one(float, &mut b, &mut mapping, instruction, |b, val_a| {
b.ext_inst(float, None, glsl, spirv::GLOp::Exp as u32, [IdRef(val_a)])
@@ -955,36 +906,12 @@ impl SSATape {
.unwrap()
});
},
SSAExp2 => {
param_one(float, &mut b, &mut mapping, instruction, |b, val_a| {
b.ext_inst(float, None, glsl, spirv::GLOp::Exp2 as u32, [IdRef(val_a)])
.unwrap()
});
},
SSALog2 => {
param_one(float, &mut b, &mut mapping, instruction, |b, val_a| {
b.ext_inst(float, None, glsl, spirv::GLOp::Log2 as u32, [IdRef(val_a)])
.unwrap()
});
},
SSASqrt => {
param_one(float, &mut b, &mut mapping, instruction, |b, val_a| {
b.ext_inst(float, None, glsl, spirv::GLOp::Sqrt as u32, [IdRef(val_a)])
.unwrap()
});
},
SSAInverseSqrt => {
param_one(float, &mut b, &mut mapping, instruction, |b, val_a| {
b.ext_inst(
float,
None,
glsl,
spirv::GLOp::InverseSqrt as u32,
[IdRef(val_a)],
)
.unwrap()
});
},
SSASquare => {
param_one(float, &mut b, &mut mapping, instruction, |b, val_a| {
b.f_mul(float, None, val_a, val_a).unwrap()
@@ -1006,25 +933,21 @@ impl SSATape {
let div_k = b.f_div(float, None, mul_half, k).unwrap();
let add_half = b.f_add(float, None, div_k, half_const).unwrap();
let h = b
.ext_inst(
float,
None,
glsl,
spirv::GLOp::FClamp as u32,
[IdRef(add_half), IdRef(zero_const), IdRef(one_const)],
)
.ext_inst(float, None, glsl, spirv::GLOp::FClamp as u32, [
IdRef(add_half),
IdRef(zero_const),
IdRef(one_const),
])
.unwrap();
let negh = b.f_sub(float, None, one_const, h).unwrap();
let h_negh = b.f_mul(float, None, h, negh).unwrap();
let kh_negh = b.f_mul(float, None, k, h_negh).unwrap();
let mix = b
.ext_inst(
float,
None,
glsl,
spirv::GLOp::FMix as u32,
[IdRef(d2), IdRef(d1), IdRef(h)],
)
.ext_inst(float, None, glsl, spirv::GLOp::FMix as u32, [
IdRef(d2),
IdRef(d1),
IdRef(h),
])
.unwrap();
b.f_sub(float, None, mix, kh_negh).unwrap()
});
@@ -1039,26 +962,22 @@ impl SSATape {
let div_k = b.f_div(float, None, mul_half, k).unwrap();
let add_half = b.f_sub(float, None, half_const, div_k).unwrap();
let h = b
.ext_inst(
float,
None,
glsl,
spirv::GLOp::FClamp as u32,
[IdRef(add_half), IdRef(zero_const), IdRef(one_const)],
)
.ext_inst(float, None, glsl, spirv::GLOp::FClamp as u32, [
IdRef(add_half),
IdRef(zero_const),
IdRef(one_const),
])
.unwrap();
let negh = b.f_sub(float, None, one_const, h).unwrap();
let h_negh = b.f_mul(float, None, h, negh).unwrap();
let kh_negh = b.f_mul(float, None, k, h_negh).unwrap();
let negate = b.f_negate(float, None, d1).unwrap();
let mix = b
.ext_inst(
float,
None,
glsl,
spirv::GLOp::FMix as u32,
[IdRef(d2), IdRef(negate), IdRef(h)],
)
.ext_inst(float, None, glsl, spirv::GLOp::FMix as u32, [
IdRef(d2),
IdRef(negate),
IdRef(h),
])
.unwrap();
b.f_add(float, None, mix, kh_negh).unwrap()
});
@@ -1072,13 +991,11 @@ impl SSATape {
&mut mapping,
instruction,
|b, val_a, val_b, val_c| {
b.ext_inst(
float,
None,
glsl,
spirv::GLOp::FClamp as u32,
[IdRef(val_a), IdRef(val_b), IdRef(val_c)],
)
b.ext_inst(float, None, glsl, spirv::GLOp::FClamp as u32, [
IdRef(val_a),
IdRef(val_b),
IdRef(val_c),
])
.unwrap()
},
);
@@ -1090,49 +1007,11 @@ impl SSATape {
&mut mapping,
instruction,
|b, val_a, val_b, val_c| {
b.ext_inst(
float,
None,
glsl,
spirv::GLOp::FMix as u32,
[IdRef(val_a), IdRef(val_b), IdRef(val_c)],
)
.unwrap()
},
);
},
SSAStep => {
param_three(
float,
&mut b,
&mut mapping,
instruction,
|b, val_a, val_b, val_c| {
b.ext_inst(
float,
None,
glsl,
spirv::GLOp::Step as u32,
[IdRef(val_a), IdRef(val_b), IdRef(val_c)],
)
.unwrap()
},
);
},
SSASmoothStep => {
param_three(
float,
&mut b,
&mut mapping,
instruction,
|b, val_a, val_b, val_c| {
b.ext_inst(
float,
None,
glsl,
spirv::GLOp::SmoothStep as u32,
[IdRef(val_a), IdRef(val_b), IdRef(val_c)],
)
b.ext_inst(float, None, glsl, spirv::GLOp::FMix as u32, [
IdRef(val_a),
IdRef(val_b),
IdRef(val_c),
])
.unwrap()
},
);
@@ -1144,13 +1023,11 @@ impl SSATape {
&mut mapping,
instruction,
|b, val_a, val_b, val_c| {
b.ext_inst(
float,
None,
glsl,
spirv::GLOp::Fma as u32,
[IdRef(val_a), IdRef(val_b), IdRef(val_c)],
)
b.ext_inst(float, None, glsl, spirv::GLOp::Fma as u32, [
IdRef(val_a),
IdRef(val_b),
IdRef(val_c),
])
.unwrap()
},
);
@@ -1201,54 +1078,38 @@ impl SSATape {
.composite_construct(vec3, None, [zero, zero, zero])
.unwrap();
let q_limit = b
.ext_inst(
vec3,
None,
glsl,
spirv::GLOp::FMax as u32,
[IdRef(q), IdRef(zero_vec3)],
)
.ext_inst(vec3, None, glsl, spirv::GLOp::FMax as u32, [
IdRef(q),
IdRef(zero_vec3),
])
.unwrap();
let length = b
.ext_inst(
float,
None,
glsl,
spirv::GLOp::Length as u32,
[IdRef(q_limit)],
)
.ext_inst(float, None, glsl, spirv::GLOp::Length as u32, [IdRef(
q_limit,
)])
.unwrap();
let q_x = b.composite_extract(float, None, q, [0]).unwrap();
let q_y = b.composite_extract(float, None, q, [1]).unwrap();
let q_z = b.composite_extract(float, None, q, [2]).unwrap();
let max1 = b
.ext_inst(
float,
None,
glsl,
spirv::GLOp::FMax as u32,
[IdRef(q_x), IdRef(q_y)],
)
.ext_inst(float, None, glsl, spirv::GLOp::FMax as u32, [
IdRef(q_x),
IdRef(q_y),
])
.unwrap();
let max2 = b
.ext_inst(
float,
None,
glsl,
spirv::GLOp::FMax as u32,
[IdRef(max1), IdRef(q_z)],
)
.ext_inst(float, None, glsl, spirv::GLOp::FMax as u32, [
IdRef(max1),
IdRef(q_z),
])
.unwrap();
let min = b
.ext_inst(
float,
None,
glsl,
spirv::GLOp::FMax as u32,
[IdRef(max2), IdRef(zero)],
)
.ext_inst(float, None, glsl, spirv::GLOp::FMax as u32, [
IdRef(max2),
IdRef(zero),
])
.unwrap();
mapping.insert(
@@ -1291,6 +1152,63 @@ impl SSATape {
b.f_sub(float, None, length, rad2).unwrap(),
);
},
SSACompare => {
param_two(
float,
&mut b,
&mut mapping,
instruction,
|b, val_a, val_b| {
let nan = b.constant_bit32(float, f32::NAN.to_bits());
let zero = b.constant_bit32(float, (0.0f32).to_bits());
let one = b.constant_bit32(float, (1.0f32).to_bits());
let onen = b.constant_bit32(float, (-1.0f32).to_bits());
let equal = b.f_ord_equal(bool, None, val_a, val_b).unwrap();
let less = b.f_ord_less_than(bool, None, val_a, val_b).unwrap();
let more = b.f_ord_greater_than(bool, None, val_a, val_b).unwrap();
let select_less = b.select(float, None, less, onen, nan).unwrap();
let select_more =
b.select(float, None, more, one, select_less).unwrap();
let select_eq =
b.select(float, None, equal, zero, select_more).unwrap();
select_eq
},
);
},
SSAAnd => {
param_two(
float,
&mut b,
&mut mapping,
instruction,
|b, val_a, val_b| {
let zero = b.constant_bit32(float, (0.0f32).to_bits());
let equal = b.f_ord_equal(bool, None, val_a, zero).unwrap();
b.select(float, None, equal, val_a, val_b).unwrap()
},
);
},
SSAOr => {
param_two(
float,
&mut b,
&mut mapping,
instruction,
|b, val_a, val_b| {
let zero = b.constant_bit32(float, (0.0f32).to_bits());
let equal = b.f_ord_equal(bool, None, val_a, zero).unwrap();
b.select(float, None, equal, val_b, val_a).unwrap()
},
);
},
SSANot => {
param_one(float, &mut b, &mut mapping, instruction, |b, val_a| {
let zero = b.constant_bit32(float, (0.0f32).to_bits());
let one = b.constant_bit32(float, (1.0f32).to_bits());
let equal = b.f_ord_equal(bool, None, val_a, zero).unwrap();
b.select(float, None, equal, one, zero).unwrap()
});
},
}
}
b.end_function().unwrap();